Files
EvoScientist-Multi/tests/test_llm.py
T
Xi Zhang b36c19a22a fix(llm): honor openrouter_app_title only alongside a custom referer (#453)
OpenRouter keys app pages by HTTP-Referer; X-Title only renames that
page. A custom openrouter_app_title on the default referer therefore
renamed the shared EvoScientist app page for everyone. Force the default
title whenever the resolved referer is the default, silently, so usage
keeps being attributed to EvoScientist; a private fork still overrides
both together.
2026-09-07 17:22:18 +08:00

3681 lines
144 KiB
Python

"""Tests for EvoScientist LLM module."""
import warnings
from unittest.mock import patch
import pytest
# Side-effect import: applies module-level monkey-patches (e.g.,
# _patch_openai_capture_reasoning_content) before tests reference patched
# functions from langchain_openai.
import EvoScientist.llm.patches # noqa: F401
from EvoScientist.llm import (
DEFAULT_MODEL,
MODELS,
get_chat_model,
get_model_info,
get_models_for_provider,
list_models,
)
from EvoScientist.llm.models import _MODEL_ENTRIES
# =============================================================================
# Test MODELS registry
# =============================================================================
class TestModelsRegistry:
def test_models_is_dict(self):
"""Test that MODELS is a dictionary."""
assert isinstance(MODELS, dict)
def test_entries_has_all_providers(self):
"""Test that _MODEL_ENTRIES covers all registered providers."""
providers = {p for _, _, p in _MODEL_ENTRIES}
assert "anthropic" in providers
assert "openai" in providers
assert "google-genai" in providers
assert "minimax" in providers
assert "nvidia" in providers
assert "siliconflow" in providers
assert "openrouter" in providers
assert "zhipu" in providers
assert "zhipu-code" in providers
assert "volcengine" in providers
assert "volcengine-code" in providers
assert "dashscope" in providers
assert "dashscope-code" in providers
assert "deepseek" in providers
assert "moonshot" in providers
assert "kimi-coding" in providers
assert "atlascloud" in providers
assert "novita" in providers
def test_entries_are_valid_tuples(self):
"""Test that _MODEL_ENTRIES contains valid (name, model_id, provider) tuples."""
valid_providers = {
"anthropic",
"openai",
"google-genai",
"minimax",
"nvidia",
"siliconflow",
"openrouter",
"requesty",
"zhipu",
"zhipu-code",
"volcengine",
"volcengine-code",
"dashscope",
"dashscope-code",
"custom-openai",
"custom-anthropic",
"deepseek",
"moonshot",
"kimi-coding",
"atlascloud",
"novita",
}
for entry in _MODEL_ENTRIES:
assert len(entry) == 3, f"Entry {entry} doesn't have 3 elements"
name, model_id, provider = entry
assert isinstance(name, str)
assert isinstance(model_id, str)
assert provider in valid_providers, (
f"Unknown provider '{provider}' for '{name}'"
)
def test_get_models_for_provider(self):
"""Test that get_models_for_provider returns correct models."""
anthropic_models = get_models_for_provider("anthropic")
assert len(anthropic_models) > 0
for name, model_id in anthropic_models:
assert isinstance(name, str)
assert isinstance(model_id, str)
# Third-party providers now have registered models
openrouter_models = get_models_for_provider("openrouter")
assert len(openrouter_models) > 0
siliconflow_models = get_models_for_provider("siliconflow")
assert len(siliconflow_models) > 0
atlas_models = get_models_for_provider("atlascloud")
assert ("qwen3.5-27b", "qwen/qwen3.5-27b") in atlas_models
assert get_models_for_provider("atlas") == []
novita_models = get_models_for_provider("novita")
assert ("kimi-k3", "moonshotai/kimi-k3") in novita_models
# =============================================================================
# Test DEFAULT_MODEL
# =============================================================================
class TestDefaultModel:
def test_default_model_exists_in_registry(self):
"""Test that DEFAULT_MODEL is a valid model in MODELS."""
assert DEFAULT_MODEL in MODELS
def test_default_model_is_anthropic(self):
"""Test that default model uses Anthropic."""
_, provider = MODELS[DEFAULT_MODEL]
assert provider == "anthropic"
# =============================================================================
# Test list_models
# =============================================================================
class TestListModels:
def test_returns_list(self):
"""Test that list_models returns a list."""
result = list_models()
assert isinstance(result, list)
def test_returns_all_model_names(self):
"""Test that list_models returns all model names."""
result = list_models()
assert set(result) == set(MODELS.keys())
def test_list_is_not_empty(self):
"""Test that the list is not empty."""
assert len(list_models()) > 0
# =============================================================================
# Test get_model_info
# =============================================================================
class TestGetModelInfo:
def test_returns_tuple_for_valid_model(self):
"""Test that get_model_info returns tuple for valid model."""
result = get_model_info("claude-sonnet-4-6")
assert result is not None
assert isinstance(result, tuple)
assert len(result) == 2
def test_returns_none_for_invalid_model(self):
"""Test that get_model_info returns None for invalid model."""
result = get_model_info("nonexistent-model")
assert result is None
def test_returns_correct_info(self):
"""Test that get_model_info returns correct info."""
model_id, provider = get_model_info("gpt-5-nano")
assert model_id == "gpt-5-nano"
assert provider == "openai"
# =============================================================================
# Test get_chat_model
# =============================================================================
class TestGetChatModel:
@patch("EvoScientist.llm.models.init_chat_model")
def test_uses_default_model_when_none(self, mock_init):
"""Test that get_chat_model uses default model when model=None."""
mock_init.return_value = "mock_model"
get_chat_model()
mock_init.assert_called_once()
call_kwargs = mock_init.call_args[1]
# Default model should be resolved from MODELS
expected_model_id, expected_provider = MODELS[DEFAULT_MODEL]
assert call_kwargs["model"] == expected_model_id
assert call_kwargs["model_provider"] == expected_provider
@patch("EvoScientist.llm.models.init_chat_model")
def test_resolves_short_name(self, mock_init):
"""Test that get_chat_model resolves short names correctly."""
mock_init.return_value = "mock_model"
get_chat_model("claude-opus-4-8")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "claude-opus-4-8"
assert call_kwargs["model_provider"] == "anthropic"
@patch("EvoScientist.llm.models.init_chat_model")
def test_resolves_openai_short_name(self, mock_init):
"""Test that get_chat_model resolves OpenAI short names."""
mock_init.return_value = "mock_model"
get_chat_model("gpt-5-mini")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "gpt-5-mini"
assert call_kwargs["model_provider"] == "openai"
@patch("EvoScientist.llm.models.init_chat_model")
def test_uses_full_model_id(self, mock_init):
"""Test that get_chat_model accepts full model IDs."""
mock_init.return_value = "mock_model"
get_chat_model("claude-3-opus-20240229")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "claude-3-opus-20240229"
# Should infer anthropic from the model prefix
assert call_kwargs["model_provider"] == "anthropic"
@patch("EvoScientist.llm.models.init_chat_model")
def test_provider_override(self, mock_init):
"""Test that provider can be overridden."""
mock_init.return_value = "mock_model"
get_chat_model("claude-sonnet-4-6", provider="custom_provider")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "custom_provider"
@patch("EvoScientist.llm.models.init_chat_model")
def test_passes_kwargs(self, mock_init):
"""Test that additional kwargs are passed through."""
mock_init.return_value = "mock_model"
get_chat_model("gpt-5-nano", temperature=0.7, max_tokens=1000)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["temperature"] == 0.7
assert call_kwargs["max_tokens"] == 1000
@patch("EvoScientist.llm.models.init_chat_model")
def test_infers_openai_from_gpt_prefix(self, mock_init):
"""Test that OpenAI is inferred from gpt- prefix."""
mock_init.return_value = "mock_model"
get_chat_model("gpt-4-turbo-preview")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
@patch("EvoScientist.llm.models.init_chat_model")
def test_infers_openai_from_o1_prefix(self, mock_init):
"""Test that OpenAI is inferred from o1 prefix."""
mock_init.return_value = "mock_model"
get_chat_model("o1-preview")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
@patch("EvoScientist.llm.models.init_chat_model")
def test_infers_google_from_gemini_prefix(self, mock_init):
"""Test that google-genai is inferred from gemini prefix."""
mock_init.return_value = "mock_model"
get_chat_model("gemini-2.0-flash")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "google-genai"
@patch("EvoScientist.llm.models.init_chat_model")
def test_defaults_to_anthropic_for_unknown(self, mock_init):
"""Test that anthropic is default for unknown model prefixes."""
mock_init.return_value = "mock_model"
get_chat_model("some-unknown-model")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "anthropic"
# =============================================================================
# Test Ollama provider
# =============================================================================
class TestOllamaProvider:
"""Ollama models are not in the static registry (detected dynamically).
All tests use explicit provider or ollama: prefix."""
@patch("EvoScientist.llm.models.init_chat_model")
def test_explicit_provider(self, mock_init):
"""Test that explicit provider='ollama' routes correctly."""
mock_init.return_value = "mock_model"
get_chat_model("llama3.1:8b", provider="ollama")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "llama3.1:8b"
assert call_kwargs["model_provider"] == "ollama"
@patch("EvoScientist.llm.models.init_chat_model")
def test_ollama_base_url_passthrough(self, mock_init, monkeypatch):
"""Test that OLLAMA_BASE_URL env var is passed to kwargs."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OLLAMA_BASE_URL", "http://gpu-cluster:11434")
get_chat_model("llama3.1:8b", provider="ollama")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["base_url"] == "http://gpu-cluster:11434"
assert call_kwargs["model_provider"] == "ollama"
@patch("EvoScientist.llm.models.init_chat_model")
def test_ollama_no_base_url_when_unset(self, mock_init, monkeypatch):
"""Test that base_url is not set when OLLAMA_BASE_URL is empty."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
get_chat_model("llama3.1:8b", provider="ollama")
call_kwargs = mock_init.call_args[1]
assert "base_url" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_reasoning_auto_enabled_for_ollama(self, mock_init, monkeypatch):
"""Test that reasoning is auto-enabled for Ollama models."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
get_chat_model("llama3.1:8b", provider="ollama")
call_kwargs = mock_init.call_args[1]
assert "thinking" not in call_kwargs
assert call_kwargs["reasoning"] is True
@patch("EvoScientist.llm.models.init_chat_model")
def test_reasoning_not_overridden_for_ollama(self, mock_init, monkeypatch):
"""Test that explicit reasoning=False is not overridden for Ollama."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
get_chat_model("llama3.1:8b", provider="ollama", reasoning=False)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["reasoning"] is False
def test_no_static_registry_entries(self):
"""Test that Ollama has no static registry entries (models detected dynamically)."""
ollama_models = get_models_for_provider("ollama")
assert len(ollama_models) == 0
@patch("EvoScientist.llm.models.init_chat_model")
def test_ollama_prefix_inference(self, mock_init, monkeypatch):
"""Test that ollama: prefix infers ollama provider."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
get_chat_model("ollama:phi3:mini")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "phi3:mini"
assert call_kwargs["model_provider"] == "ollama"
# =============================================================================
# Test slash model ID no longer routes to nvidia
# =============================================================================
class TestSlashModelIdFallback:
@patch("EvoScientist.llm.models.init_chat_model")
def test_slash_model_id_defaults_to_anthropic(self, mock_init):
"""Unregistered model IDs containing '/' should NOT route to nvidia.
They fall through to the default 'anthropic' provider, consistent
with how all other unknown model IDs are handled.
"""
mock_init.return_value = "mock_model"
get_chat_model("some-org/some-model")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "some-org/some-model"
assert call_kwargs["model_provider"] == "anthropic"
# =============================================================================
# Test third-party provider routing
# =============================================================================
def _sdk_retry_supports_status_codes_override() -> bool:
"""openrouter>=0.11 only; the 429 override degrades to a no-op below that."""
from openrouter.utils.retries import RetryConfig
return "status_codes_override" in getattr(RetryConfig, "__annotations__", {})
class TestThirdPartyRouting:
@patch("EvoScientist.llm.models.init_chat_model")
def test_atlascloud_routes_through_openai(self, mock_init, monkeypatch):
"""Atlas Cloud should use OpenAI-compatible routing with its default URL."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("ATLASCLOUD_API_KEY", "atlas-key-123")
get_chat_model("qwen3.5-27b", provider="atlascloud")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "qwen/qwen3.5-27b"
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://api.atlascloud.ai/v1"
assert call_kwargs["api_key"] == "atlas-key-123"
assert "reasoning" not in call_kwargs
def test_atlascloud_host_maps_to_provider(self):
"""Provider error envelopes should identify Atlas Cloud by host."""
from EvoScientist.llm.errors import _lookup_host_or_compat
assert (
_lookup_host_or_compat("https://api.atlascloud.ai/v1", "openai")
== "atlascloud"
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_siliconflow_routes_through_openai(self, mock_init, monkeypatch):
"""SiliconFlow provider should route through OpenAI with correct base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("SILICONFLOW_API_KEY", "sf-key-123")
get_chat_model("Pro/zai-org/GLM-5", provider="siliconflow")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://api.siliconflow.cn/v1"
assert call_kwargs["api_key"] == "sf-key-123"
# SiliconFlow should disable thinking
assert call_kwargs["extra_body"]["enable_thinking"] is False
@patch("EvoScientist.llm.models.init_chat_model")
def test_requesty_routes_through_openai(self, mock_init, monkeypatch):
"""Requesty provider should route through OpenAI with correct base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key-123")
get_chat_model("openai/gpt-4o-mini", provider="requesty")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://router.requesty.ai/v1"
assert call_kwargs["api_key"] == "rq-key-123"
@patch("EvoScientist.llm.models.init_chat_model")
def test_novita_routes_through_openai(self, mock_init, monkeypatch):
"""Novita provider should route through OpenAI with correct base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("NOVITA_API_KEY", "novita-key-123")
get_chat_model("kimi-k3", provider="novita")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "moonshotai/kimi-k3"
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://api.novita.ai/openai/v1"
assert call_kwargs["api_key"] == "novita-key-123"
def test_novita_host_maps_to_provider(self):
"""Provider error envelopes should identify Novita by host."""
from EvoScientist.llm.errors import _lookup_host_or_compat
assert (
_lookup_host_or_compat("https://api.novita.ai/openai/v1", "openai")
== "novita"
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_requesty_anthropic_prompt_cache_enabled_by_default(
self, mock_init, monkeypatch
):
"""Requesty Anthropic prompt caching should be opt-out."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key")
monkeypatch.delenv(
"EVOSCIENTIST_REQUESTY_ANTHROPIC_PROMPT_CACHE", raising=False
)
get_chat_model("anthropic/claude-sonnet-4-6", provider="requesty")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://router.requesty.ai/v1"
assert call_kwargs["model_kwargs"]["cache_control"] == {"type": "ephemeral"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_requesty_anthropic_prompt_cache_opt_out(self, mock_init, monkeypatch):
"""The opt-out flag should skip caching for Requesty Claude models."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key")
monkeypatch.setenv("EVOSCIENTIST_REQUESTY_ANTHROPIC_PROMPT_CACHE", "false")
get_chat_model("anthropic/claude-sonnet-4-6", provider="requesty")
call_kwargs = mock_init.call_args[1]
assert "cache_control" not in call_kwargs
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
@patch("EvoScientist.llm.models.init_chat_model")
def test_requesty_prompt_cache_skips_non_anthropic(self, mock_init, monkeypatch):
"""Requesty caching should not touch non-Anthropic models."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key")
monkeypatch.delenv(
"EVOSCIENTIST_REQUESTY_ANTHROPIC_PROMPT_CACHE", raising=False
)
get_chat_model("openai/gpt-4o-mini", provider="requesty")
call_kwargs = mock_init.call_args[1]
assert "cache_control" not in call_kwargs
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
@patch("EvoScientist.llm.models.init_chat_model")
def test_deepseek_uses_copy_safe_native_model(self, mock_init, monkeypatch):
from EvoScientist.llm.deepseek import EvoChatDeepSeek
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
model = get_chat_model("deepseek-v4-flash", provider="deepseek")
mock_init.assert_not_called()
assert isinstance(model, EvoChatDeepSeek)
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_uses_native_provider(self, mock_init, monkeypatch):
"""OpenRouter should use native 'openrouter' provider via init_chat_model."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key-456")
# Assert the DEFAULT effort, so isolate from any leaked env override.
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
get_chat_model("x-ai/grok-4.3", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openrouter"
assert call_kwargs["api_key"] == "or-key-456"
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_reasoning_user_override(self, mock_init, monkeypatch):
"""User-supplied reasoning config should not be overridden."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
get_chat_model(
"x-ai/grok-4.3",
provider="openrouter",
reasoning={"effort": "low"},
)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["reasoning"] == {"effort": "low"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_reasoning_effort_from_env(self, mock_init, monkeypatch):
"""Reasoning effort should be configurable via env var."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "medium")
get_chat_model("x-ai/grok-4.3", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["reasoning"] == {"effort": "medium", "summary": "auto"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_moonshot_thinking_disable_exempts_kimi_k3(self, mock_init, monkeypatch):
"""Native Moonshot: K3 must not receive the K2.x thinking-disable field.
Moonshot's K3 guide forbids the K2.x `thinking` parameter (K3 is
always-thinking); other Moonshot models keep the disable that guards
against multi-turn error 20015.
"""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MOONSHOT_API_KEY", "ms-key")
get_chat_model("kimi-k3", provider="moonshot")
extra_body = mock_init.call_args[1].get("extra_body") or {}
assert "thinking" not in extra_body
get_chat_model("kimi-k2.6", provider="moonshot")
extra_body = mock_init.call_args[1]["extra_body"]
assert extra_body["thinking"] == {"type": "disabled"}
# --- OpenRouter upstream 429 retry ---
@pytest.mark.skipif(
not _sdk_retry_supports_status_codes_override(),
reason="openrouter<0.11 RetryConfig lacks status_codes_override; "
"the 429 override no-ops there by design (models.py hasattr guard)",
)
def test_openrouter_429_added_to_retryable_status_codes(self, monkeypatch):
"""Upstream 429s must become retryable on the real SDK client.
The openrouter SDK hardcodes per-operation retryable statuses to
["5XX"], so a launch-day "temporarily rate-limited upstream" 429
(Retry-After: 1) fails the run outright instead of being retried.
"""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
retry_config = model.client.sdk_configuration.retry_config
assert retry_config.status_codes_override == ["429", "5XX"]
def test_openrouter_429_override_not_injected_when_retries_disabled(
self, monkeypatch
):
"""max_retries=0 leaves the SDK retry config UNSET — no 429 override.
Note this only asserts our override is absent; the SDK still applies
its own per-operation default (backoff on 5XX) when the config is
UNSET, so retries as such are not fully disabled at the SDK level.
"""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
model = get_chat_model("x-ai/grok-4.3", provider="openrouter", max_retries=0)
retry_config = model.client.sdk_configuration.retry_config
assert getattr(retry_config, "status_codes_override", None) is None
@pytest.mark.skipif(
not _sdk_retry_supports_status_codes_override(),
reason="openrouter<0.11 RetryConfig lacks status_codes_override; "
"the 429 override no-ops there by design (models.py hasattr guard)",
)
def test_openrouter_429_retried_on_the_wire(self, monkeypatch):
"""End-to-end: a 429 with Retry-After is retried and the retry succeeds."""
import httpx
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
calls = {"n": 0}
def handler(request: httpx.Request) -> httpx.Response:
calls["n"] += 1
if calls["n"] == 1:
return httpx.Response(
429,
headers={"Retry-After": "1"},
json={"error": {"message": "Provider returned error", "code": 429}},
)
return httpx.Response(
200,
json={
"id": "gen-1",
"object": "chat.completion",
"created": 1,
"model": "moonshotai/kimi-k3",
"system_fingerprint": "fp-test",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok"},
"finish_reason": "stop",
}
],
},
)
model.client.sdk_configuration.client = httpx.Client(
transport=httpx.MockTransport(handler)
)
result = model.invoke("hi")
assert calls["n"] == 2
assert result.content == "ok"
# --- OpenRouter structured output vs mandatory reasoning ---
@staticmethod
def _capture_structured_request(model, structured, response_message):
"""Invoke a structured-output runnable against a capturing transport."""
import json
import httpx
captured: dict = {}
def handler(request: httpx.Request) -> httpx.Response:
captured.update(json.loads(request.content.decode()))
return httpx.Response(
200,
json={
"id": "gen-1",
"object": "chat.completion",
"created": 1,
"model": "m",
"system_fingerprint": "fp-test",
"choices": [
{
"index": 0,
"message": response_message,
"finish_reason": "stop",
}
],
},
)
model.client.sdk_configuration.client = httpx.Client(
transport=httpx.MockTransport(handler)
)
result = structured.invoke("pick tools")
return captured, result
def test_openrouter_structured_output_json_schema_for_mandatory_model(
self, monkeypatch
):
"""with_structured_output must not force tool_choice on kimi-k3.
Moonshot rejects a forced tool choice with HTTP 400 "tool_choice
'specified' is incompatible with thinking enabled", and kimi-k3's
thinking cannot be disabled — so the default function_calling method
400s every structured-output call (LLMToolSelectorMiddleware included).
The json_schema method (response_format) is supported and needs none.
"""
from pydantic import BaseModel
class ToolSelection(BaseModel):
tools: list[str]
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
# The dated canonical_slug is routable too and must be equally covered.
for model_name in ("moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"):
model = get_chat_model(model_name, provider="openrouter")
structured = model.with_structured_output(ToolSelection)
captured, result = self._capture_structured_request(
model,
structured,
{"role": "assistant", "content": '{"tools": ["tavily_search"]}'},
)
assert "tool_choice" not in captured, model_name
assert captured["response_format"]["type"] == "json_schema", model_name
assert result == ToolSelection(tools=["tavily_search"])
def test_openrouter_structured_output_default_for_other_models(self, monkeypatch):
"""Non-Moonshot models keep the function_calling default.
Includes always-thinking models like grok-4.5 — the forced tool_choice
restriction is Moonshot-specific, so the json_schema rerouting must
stay limited to _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS.
"""
from pydantic import BaseModel
class ToolSelection(BaseModel):
tools: list[str]
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
for model_name in ("x-ai/grok-4.3", "x-ai/grok-4.5"):
model = get_chat_model(model_name, provider="openrouter")
structured = model.with_structured_output(ToolSelection)
captured, result = self._capture_structured_request(
model,
structured,
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {
"name": "ToolSelection",
"arguments": '{"tools": ["tavily_search"]}',
},
}
],
},
)
assert captured.get("tool_choice"), model_name
assert "response_format" not in captured, model_name
assert result == ToolSelection(tools=["tavily_search"])
# --- OpenRouter app attribution (issue #339) ---
_APP_ATTR_ENV = (
"EVOSCIENTIST_OPENROUTER_HTTP_REFERER",
"EVOSCIENTIST_OPENROUTER_APP_TITLE",
"EVOSCIENTIST_OPENROUTER_APP_CATEGORIES",
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_attribution_defaults(self, mock_init, monkeypatch):
"""OpenRouter init should carry EvoScientist's default app attribution."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
# Isolate from any leaked env overrides so we assert the built-in defaults.
for _env in self._APP_ATTR_ENV:
monkeypatch.delenv(_env, raising=False)
get_chat_model("x-ai/grok-4.3", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["app_url"] == "https://github.com/EvoScientist/EvoScientist"
assert call_kwargs["app_title"] == "EvoScientist"
# Must be a list[str] (not the comma string) — langchain-openrouter joins it.
assert call_kwargs["app_categories"] == ["creative-writing", "personal-agent"]
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_attribution_from_env(self, mock_init, monkeypatch):
"""Env vars should override the default app attribution values."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_HTTP_REFERER", "https://acme.test")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_TITLE", "Acme")
# Include a space to prove each category is stripped.
monkeypatch.setenv(
"EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", "cli-agent, programming-app"
)
get_chat_model("x-ai/grok-4.3", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["app_url"] == "https://acme.test"
assert call_kwargs["app_title"] == "Acme"
assert call_kwargs["app_categories"] == ["cli-agent", "programming-app"]
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_attribution_user_override_not_clobbered(
self, mock_init, monkeypatch
):
"""Caller-supplied attribution kwargs must beat both env and defaults."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
# Env is also set, to prove an explicit kwarg outranks the env override
# (not just the built-in default).
monkeypatch.setenv(
"EVOSCIENTIST_OPENROUTER_HTTP_REFERER", "https://env.example"
)
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_TITLE", "EnvTitle")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", "env-cat")
get_chat_model(
"x-ai/grok-4.3",
provider="openrouter",
app_url="https://mine.example",
app_title="MyApp",
app_categories=["only-this"],
)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["app_url"] == "https://mine.example"
assert call_kwargs["app_title"] == "MyApp"
# An explicit list is preserved verbatim, not re-split.
assert call_kwargs["app_categories"] == ["only-this"]
@pytest.mark.parametrize("source", ["env", "kwarg"])
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_title_override_without_referer_falls_back_silently(
self, mock_init, monkeypatch, source
):
"""OpenRouter keys app pages by HTTP-Referer, so a custom title on the
default referer would rename the shared EvoScientist page. It is
replaced by the default title, without any user-facing warning,
whether the title came from the env (config) or an explicit kwarg."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
for _env in self._APP_ATTR_ENV:
monkeypatch.delenv(_env, raising=False)
extra = {}
if source == "env":
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_TITLE", "Acme")
else:
extra["app_title"] = "Acme"
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter("always")
get_chat_model("x-ai/grok-4.3", provider="openrouter", **extra)
assert not [w for w in caught if "openrouter" in str(w.message).lower()]
call_kwargs = mock_init.call_args[1]
assert call_kwargs["app_url"] == "https://github.com/EvoScientist/EvoScientist"
assert call_kwargs["app_title"] == "EvoScientist"
@patch("EvoScientist.llm.models.init_chat_model")
def test_non_openrouter_providers_get_no_app_attribution(
self, mock_init, monkeypatch
):
"""Only the openrouter provider should receive app-attribution kwargs."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-real")
monkeypatch.setenv("OLLAMA_BASE_URL", "http://localhost:11434")
for model, provider in (
("claude-sonnet-4-6", "anthropic"),
("llama3.1:8b", "ollama"),
):
get_chat_model(model, provider=provider)
call_kwargs = mock_init.call_args[1]
assert "app_url" not in call_kwargs
assert "app_title" not in call_kwargs
assert "app_categories" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_attribution_coexists_with_reasoning_and_cache(
self, mock_init, monkeypatch
):
"""Attribution must not disturb reasoning or Anthropic prompt caching."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
monkeypatch.delenv(
"EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", raising=False
)
for _env in self._APP_ATTR_ENV:
monkeypatch.delenv(_env, raising=False)
get_chat_model("claude-sonnet-4.6", provider="openrouter")
call_kwargs = mock_init.call_args[1]
# Existing behavior intact.
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
assert call_kwargs["model_kwargs"]["cache_control"] == {"type": "ephemeral"}
# Attribution added alongside.
assert call_kwargs["app_url"] == "https://github.com/EvoScientist/EvoScientist"
assert call_kwargs["app_title"] == "EvoScientist"
assert call_kwargs["app_categories"] == ["creative-writing", "personal-agent"]
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_categories_env_strips_blank_items(
self, mock_init, monkeypatch
):
"""A messy comma value (stray commas / spaces) yields a clean list."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", "a,, b ")
get_chat_model("x-ai/grok-4.3", provider="openrouter")
assert mock_init.call_args[1]["app_categories"] == ["a", "b"]
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_categories_capped_to_per_request_limit(
self, mock_init, monkeypatch
):
"""Over-configuring categories caps to the first N and warns the user."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv(
"EVOSCIENTIST_OPENROUTER_APP_CATEGORIES",
"cli-agent,programming-app,personal-agent,writing-assistant",
)
with pytest.warns(UserWarning, match="at most 2 app categories"):
get_chat_model("x-ai/grok-4.3", provider="openrouter")
# OpenRouter honors at most 2 per request, so only the first 2 are sent.
assert mock_init.call_args[1]["app_categories"] == [
"cli-agent",
"programming-app",
]
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_app_categories_all_separators_omit_kwarg(
self, mock_init, monkeypatch
):
"""A categories value with no real items omits the kwarg entirely."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", " , , ")
get_chat_model("x-ai/grok-4.3", provider="openrouter")
# No app_categories kwarg at all — not an empty list (which the library
# would reject / send as an empty header).
assert "app_categories" not in mock_init.call_args[1]
def test_openrouter_app_attribution_lands_on_real_model(self, monkeypatch):
"""Build a REAL ChatOpenRouter (no mock) and assert the attribution
values land on the instance rather than being silently dumped into
model_kwargs.
The mocked tests above assert on the kwargs handed to init_chat_model,
so they cannot catch a param-name typo or a langchain-openrouter version
that accepts these only as passthrough model params (which the library
does with a warning, not an error). This test is the guard for both.
"""
from langchain_openrouter import ChatOpenRouter
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
for _env in self._APP_ATTR_ENV:
monkeypatch.delenv(_env, raising=False)
model = get_chat_model("x-ai/grok-4.3", provider="openrouter")
assert isinstance(model, ChatOpenRouter)
assert model.app_url == "https://github.com/EvoScientist/EvoScientist"
assert model.app_title == "EvoScientist"
assert model.app_categories == ["creative-writing", "personal-agent"]
# Not silently swallowed into model_kwargs (the passthrough failure mode).
model_kwargs = model.model_kwargs or {}
assert "app_url" not in model_kwargs
assert "app_title" not in model_kwargs
assert "app_categories" not in model_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_anthropic_prompt_cache_enabled_by_default(
self, mock_init, monkeypatch
):
"""OpenRouter Anthropic prompt caching should be opt-out."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.delenv(
"EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", raising=False
)
get_chat_model("claude-sonnet-4.6", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openrouter"
assert call_kwargs["model"] == "anthropic/claude-sonnet-4.6"
assert call_kwargs["model_kwargs"]["cache_control"] == {"type": "ephemeral"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_anthropic_prompt_cache_opt_out(self, mock_init, monkeypatch):
"""The opt-out flag should skip caching for OpenRouter Claude models."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "false")
get_chat_model("claude-sonnet-4.6", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert "cache_control" not in call_kwargs
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
@patch("EvoScientist.llm.models.init_chat_model")
def test_prompt_cache_default_skips_non_anthropic_openrouter(
self, mock_init, monkeypatch
):
"""OpenRouter models with implicit caching should be left alone."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
get_chat_model("x-ai/grok-4.3", provider="openrouter")
call_kwargs = mock_init.call_args[1]
assert "cache_control" not in call_kwargs
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_anthropic_prompt_cache_preserves_top_level_override(
self, mock_init, monkeypatch
):
"""The default should not duplicate a caller's cache_control kwarg."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
override = {"type": "ephemeral", "ttl": "1h"}
get_chat_model(
"claude-sonnet-4.6",
provider="openrouter",
cache_control=override,
)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["cache_control"] == override
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_anthropic_prompt_cache_preserves_model_kwargs_override(
self, mock_init, monkeypatch
):
"""The default should not duplicate model_kwargs cache_control."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
override = {"type": "ephemeral", "ttl": "1h"}
get_chat_model(
"claude-sonnet-4.6",
provider="openrouter",
model_kwargs={"cache_control": override},
)
call_kwargs = mock_init.call_args[1]
assert "cache_control" not in call_kwargs
assert call_kwargs["model_kwargs"]["cache_control"] == override
@patch("EvoScientist.llm.models.init_chat_model")
def test_openrouter_anthropic_prompt_cache_warns_on_invalid_model_kwargs(
self, mock_init, monkeypatch
):
"""Invalid model_kwargs shape should warn and skip cache injection."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
with pytest.warns(UserWarning, match="model_kwargs` is not a dict"):
get_chat_model(
"claude-sonnet-4.6",
provider="openrouter",
model_kwargs="bad",
)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_kwargs"] == "bad"
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_routes_through_openai(self, mock_init, monkeypatch):
"""Custom provider should route through OpenAI with env-configured base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://my-llm.example.com/v1")
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "custom-key-789")
get_chat_model("my-custom-model", provider="custom-openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://my-llm.example.com/v1"
assert call_kwargs["api_key"] == "custom-key-789"
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_openai_forwards_explicit_reasoning_effort(
self, mock_init, monkeypatch
):
"""User-owned compatible endpoints receive an explicit effort only."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://opencode.example/v1")
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "custom-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "low")
get_chat_model("reasoning-model", provider="custom-openai")
assert mock_init.call_args[1]["reasoning_effort"] == "low"
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_openai_omits_unconfigured_reasoning_effort(
self, mock_init, monkeypatch
):
"""Unknown compatible endpoints stay compatible by default."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://plain.example/v1")
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "custom-key")
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
get_chat_model("plain-model", provider="custom-openai")
assert "reasoning_effort" not in mock_init.call_args[1]
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_base_url_override(self, mock_init, monkeypatch):
"""Anthropic provider should support base_url override (e.g. ccproxy)."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("ANTHROPIC_BASE_URL", "http://localhost:8000/api/v1")
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-dummy")
get_chat_model("claude-sonnet-4-6", provider="anthropic")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "anthropic"
assert call_kwargs["base_url"] == "http://localhost:8000/api/v1"
assert call_kwargs["api_key"] == "sk-dummy"
# Proxy mode: thinking skipped (history round-trip causes 422)
assert "thinking" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_no_base_url_when_unset(self, mock_init, monkeypatch):
"""Anthropic provider should not set base_url when env var is empty."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-real")
get_chat_model("claude-sonnet-4-6", provider="anthropic")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "anthropic"
assert "base_url" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_third_party_no_reasoning(self, mock_init, monkeypatch):
"""Third-party providers routed through OpenAI should NOT get auto-reasoning."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("SILICONFLOW_API_KEY", "sf-key")
get_chat_model("deepseek-v3", provider="siliconflow")
call_kwargs = mock_init.call_args[1]
assert "reasoning" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_volcengine_routes_through_openai(self, mock_init, monkeypatch):
"""Volcengine provider should route through OpenAI with correct base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("VOLCENGINE_API_KEY", "ve-key-123")
get_chat_model("doubao-seed-1.6", provider="volcengine")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://ark.cn-beijing.volces.com/api/v3"
assert call_kwargs["api_key"] == "ve-key-123"
@pytest.mark.parametrize(
("configured_model", "api_model"),
[("glm-5.2", "glm-5-2"), ("kimi-k2.5", "kimi-k2-5")],
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_volcengine_code_routes_through_openai(
self, mock_init, configured_model, api_model, monkeypatch
):
"""Volcengine Coding Plan uses its endpoint, IDs, and vendor API key."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("VOLCENGINE_API_KEY", "ve-code-key-123")
get_chat_model(configured_model, provider="volcengine-code")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["model"] == api_model
assert (
call_kwargs["base_url"] == "https://ark.cn-beijing.volces.com/api/coding/v3"
)
assert call_kwargs["api_key"] == "ve-code-key-123"
@patch("EvoScientist.llm.models.init_chat_model")
def test_dashscope_routes_through_openai(self, mock_init, monkeypatch):
"""DashScope provider should route through OpenAI with correct base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key-456")
get_chat_model("qwen-max", provider="dashscope")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert (
call_kwargs["base_url"]
== "https://dashscope.aliyuncs.com/compatible-mode/v1"
)
assert call_kwargs["api_key"] == "ds-key-456"
assert "reasoning_effort" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_uses_bounded_default(self, mock_init, monkeypatch):
"""Qwen 3.8 avoids the regular endpoint's xhigh default."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
get_chat_model("qwen3.8-max", provider="dashscope")
assert mock_init.call_args[1]["reasoning_effort"] == "medium"
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_respects_configured_reasoning_effort(
self, mock_init, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "medium")
get_chat_model("qwen3.8-max", provider="dashscope")
assert mock_init.call_args[1]["reasoning_effort"] == "medium"
@pytest.mark.parametrize(
"effort",
[
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max",
],
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_accepts_supported_reasoning_effort(
self, mock_init, effort, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", effort)
get_chat_model("qwen3.8-max", provider="dashscope")
assert mock_init.call_args[1]["reasoning_effort"] == effort
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_rejects_unsupported_reasoning_effort(
self, mock_init, monkeypatch
):
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "invalid")
with pytest.raises(ValueError, match="dashscope"):
get_chat_model("qwen3.8-max", provider="dashscope")
mock_init.assert_not_called()
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_explicit_effort_overrides_invalid_environment(
self, mock_init, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "invalid")
get_chat_model("qwen3.8-max", provider="dashscope", reasoning_effort="low")
assert mock_init.call_args[1]["reasoning_effort"] == "low"
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_code_omits_undocumented_reasoning_effort(
self, mock_init, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-sp-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "medium")
get_chat_model("qwen3.8-max", provider="dashscope-code")
assert "reasoning_effort" not in mock_init.call_args[1]
@patch("EvoScientist.llm.models.init_chat_model")
def test_dashscope_code_routes_through_openai(self, mock_init, monkeypatch):
"""DashScope-Code (sk-sp-* subscription keys) routes through OpenAI
with the coding.dashscope.aliyuncs.com base URL, reusing DASHSCOPE_API_KEY.
"""
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-sp-key-789")
get_chat_model("qwen3-coder", provider="dashscope-code")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "https://coding.dashscope.aliyuncs.com/v1"
assert call_kwargs["api_key"] == "sk-sp-key-789"
@patch("EvoScientist.llm.models.init_chat_model")
def test_minimax_routes_through_anthropic(self, mock_init, monkeypatch):
"""MiniMax provider should route through Anthropic with correct base_url."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key-123")
get_chat_model("MiniMax-M2.5", provider="minimax")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "anthropic"
assert call_kwargs["base_url"] == "https://api.minimaxi.com/anthropic"
assert call_kwargs["api_key"] == "mm-key-123"
@patch("EvoScientist.llm.models.init_chat_model")
def test_minimax_base_url_env_override(self, mock_init, monkeypatch):
"""MINIMAX_BASE_URL env var should override the default base URL."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key-123")
monkeypatch.setenv("MINIMAX_BASE_URL", "https://api.minimax.io/anthropic")
get_chat_model("MiniMax-M2.5", provider="minimax")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["base_url"] == "https://api.minimax.io/anthropic"
@patch("EvoScientist.llm.models.init_chat_model")
def test_minimax_gets_thinking(self, mock_init, monkeypatch):
"""MiniMax provider should get auto-thinking (thinking-capable via Anthropic)."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key")
get_chat_model("MiniMax-M2.5", provider="minimax")
call_kwargs = mock_init.call_args[1]
assert "thinking" in call_kwargs
assert "reasoning" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_minimax_short_name_resolution(self, mock_init, monkeypatch):
"""MiniMax short names should resolve to correct model IDs."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key")
get_chat_model("minimax-m2.5", provider="minimax")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "MiniMax-M2.5"
assert call_kwargs["model_provider"] == "anthropic"
@patch("EvoScientist.llm.models.init_chat_model")
def test_minimax_highspeed_model(self, mock_init, monkeypatch):
"""MiniMax M2.5-highspeed model should resolve correctly."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key")
get_chat_model("minimax-m2.5-highspeed", provider="minimax")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model"] == "MiniMax-M2.5-highspeed"
assert call_kwargs["model_provider"] == "anthropic"
assert call_kwargs["base_url"] == "https://api.minimaxi.com/anthropic"
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_anthropic_via_routed_dict(self, mock_init, monkeypatch):
"""custom-anthropic should work via _ANTHROPIC_ROUTED_PROVIDERS dict."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://my-claude.example.com")
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "ca-key-789")
get_chat_model("claude-sonnet-4-6", provider="custom-anthropic")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "anthropic"
assert call_kwargs["base_url"] == "https://my-claude.example.com"
assert call_kwargs["api_key"] == "ca-key-789"
# custom-anthropic is NOT thinking-capable → thinking skipped
assert "thinking" not in call_kwargs
# =============================================================================
# Test MiniMax provider
# =============================================================================
class TestMiniMaxProvider:
def test_minimax_in_anthropic_routed_providers(self):
"""MiniMax should be registered in _ANTHROPIC_ROUTED_PROVIDERS."""
from EvoScientist.llm.models import _ANTHROPIC_ROUTED_PROVIDERS
assert "minimax" in _ANTHROPIC_ROUTED_PROVIDERS
base_url, api_key_env = _ANTHROPIC_ROUTED_PROVIDERS["minimax"]
assert base_url == "https://api.minimaxi.com/anthropic"
assert api_key_env == "MINIMAX_API_KEY"
def test_minimax_not_in_openai_routed_providers(self):
"""MiniMax should NOT be in _OPENAI_ROUTED_PROVIDERS (moved to Anthropic)."""
from EvoScientist.llm.models import _OPENAI_ROUTED_PROVIDERS
assert "minimax" not in _OPENAI_ROUTED_PROVIDERS
def test_minimax_models_registered(self):
"""MiniMax should have 5 direct model entries in _MODEL_ENTRIES."""
minimax_models = get_models_for_provider("minimax")
assert len(minimax_models) == 5
model_names = {name for name, _ in minimax_models}
assert "minimax-m3" in model_names
assert "minimax-m2.7" in model_names
assert "minimax-m2.7-highspeed" in model_names
assert "minimax-m2.5" in model_names
assert "minimax-m2.5-highspeed" in model_names
def test_minimax_model_ids_correct(self):
"""MiniMax model IDs should match the official API model names."""
minimax_models = get_models_for_provider("minimax")
model_dict = dict(minimax_models)
assert model_dict["minimax-m2.7"] == "MiniMax-M2.7"
assert model_dict["minimax-m2.5"] == "MiniMax-M2.5"
assert model_dict["minimax-m2.5-highspeed"] == "MiniMax-M2.5-highspeed"
def test_minimax_short_name_in_models_dict(self):
"""MiniMax short names should be accessible via the MODELS dict."""
# Note: MODELS dict uses last-entry-wins, so direct minimax entries
# may be overridden by nvidia/siliconflow/openrouter entries.
# Use get_models_for_provider() for provider-specific lookups.
minimax_models = get_models_for_provider("minimax")
assert len(minimax_models) > 0
# =============================================================================
# Test _flatten_message_content
# =============================================================================
class TestFlattenMessageContent:
"""Tests for the content-flattening utility used by OpenAI-compatible providers."""
def test_string_passthrough(self):
from EvoScientist.llm.patches import _flatten_message_content
assert _flatten_message_content("hello") == "hello"
def test_non_list_passthrough(self):
from EvoScientist.llm.patches import _flatten_message_content
assert _flatten_message_content(42) == 42
assert _flatten_message_content(None) is None
def test_text_blocks(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [
{"type": "text", "text": "Hello"},
{"type": "text", "text": "World"},
]
assert _flatten_message_content(content) == "Hello\n\nWorld"
def test_skips_thinking_blocks(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [
{"type": "thinking", "text": "Let me think..."},
{"type": "text", "text": "The answer is 42"},
{"type": "reasoning", "text": "internal reasoning"},
{"type": "reasoning_content", "text": "more reasoning"},
]
assert _flatten_message_content(content) == "The answer is 42"
def test_string_blocks(self):
from EvoScientist.llm.patches import _flatten_message_content
content = ["hello", "world"]
assert _flatten_message_content(content) == "hello\n\nworld"
def test_mixed_blocks(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [
{"type": "thinking", "text": "skip me"},
"plain string",
{"type": "text", "text": "dict text"},
]
assert _flatten_message_content(content) == "plain string\n\ndict text"
def test_empty_list(self):
from EvoScientist.llm.patches import _flatten_message_content
assert _flatten_message_content([]) == ""
def test_only_thinking_blocks(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [{"type": "thinking", "text": "thought"}]
assert _flatten_message_content(content) == ""
def test_preserves_image_block(self):
from EvoScientist.llm.patches import _flatten_message_content
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
assert _flatten_message_content([img]) == [img]
def test_preserves_image_url_block(self):
from EvoScientist.llm.patches import _flatten_message_content
img = {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}}
assert _flatten_message_content([img]) == [img]
def test_preserves_file_block(self):
# PDF/document files are preserved (capable models read them).
from EvoScientist.llm.patches import _flatten_message_content
f = {"type": "file", "base64": "FFF", "mime_type": "application/pdf"}
assert _flatten_message_content([f]) == [f]
def test_unsupported_media_dropped(self):
# video/audio are NOT in the allowlist -> dropped, not crashing
# (langchain-openai raises ValueError on `video`).
from EvoScientist.llm.patches import _flatten_message_content
for block in (
{"type": "video", "base64": "VVV", "mime_type": "video/mp4"},
{"type": "audio", "base64": "ZZZ", "mime_type": "audio/wav"},
):
assert _flatten_message_content([block]) == ""
def test_non_image_media_dropped_keeps_text(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [
{"type": "text", "text": "hi"},
{"type": "video", "base64": "VVV", "mime_type": "video/mp4"},
]
# Video dropped, text kept -> plain string (no media list).
assert _flatten_message_content(content) == "hi"
def test_consolidates_text_and_image(self):
from EvoScientist.llm.patches import _flatten_message_content
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
content = [{"type": "text", "text": "a photo"}, img]
assert _flatten_message_content(content) == [
{"type": "text", "text": "a photo"},
img,
]
def test_multiple_text_blocks_with_image(self):
from EvoScientist.llm.patches import _flatten_message_content
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
content = [
{"type": "text", "text": "a"},
{"type": "text", "text": "b"},
img,
]
assert _flatten_message_content(content) == [
{"type": "text", "text": "a\n\nb"},
img,
]
def test_preserves_text_media_ordering(self):
# Text after an image must stay AFTER it (not consolidated to the front).
from EvoScientist.llm.patches import _flatten_message_content
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
content = [
{"type": "text", "text": "before"},
img,
{"type": "text", "text": "after"},
]
assert _flatten_message_content(content) == [
{"type": "text", "text": "before"},
img,
{"type": "text", "text": "after"},
]
def test_thinking_dropped_image_kept(self):
from EvoScientist.llm.patches import _flatten_message_content
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
content = [{"type": "thinking", "text": "hmm"}, img]
assert _flatten_message_content(content) == [img]
def test_pure_text_still_returns_string(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [{"type": "text", "text": "x"}, {"type": "text", "text": "y"}]
result = _flatten_message_content(content)
assert result == "x\n\ny"
assert isinstance(result, str)
def test_unknown_nontext_block_still_dropped(self):
from EvoScientist.llm.patches import _flatten_message_content
content = [{"type": "tool_use", "id": "1", "name": "foo"}]
assert _flatten_message_content(content) == ""
# =============================================================================
# Test _patch_openai_compat_content (all 4 paths)
# =============================================================================
class TestPatchOpenAICompatContent:
"""Verify content flattening covers _generate, _agenerate, _stream, _astream."""
def _make_model(self):
"""Create a minimal mock model with all 4 methods."""
from unittest.mock import AsyncMock, MagicMock
model = MagicMock()
model._generate = MagicMock(return_value="gen_result")
model._agenerate = AsyncMock(return_value="agen_result")
model._stream = MagicMock(return_value=iter(["chunk1"]))
model._astream = AsyncMock()
return model
def test_generate_flattened(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
model._generate([msg])
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == "hello"
async def test_agenerate_flattened(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._agenerate
_patch_openai_compat_content(model)
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
await model._agenerate([msg])
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == "hello"
def test_stream_flattened(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._stream
_patch_openai_compat_content(model)
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
list(model._stream([msg]))
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == "hello"
async def test_astream_flattened(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
received_msgs = []
async def _fake_astream(messages, *args, **kwargs):
received_msgs.extend(messages)
for chunk in ["c1", "c2"]:
yield chunk
model._astream = _fake_astream
_patch_openai_compat_content(model)
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
chunks = []
async for c in model._astream([msg]):
chunks.append(c)
assert chunks == ["c1", "c2"]
assert received_msgs[0].content == "hello"
def test_generate_preserves_media(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
model._generate([msg])
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == [{"type": "text", "text": "see"}, img]
async def test_agenerate_preserves_media(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._agenerate
_patch_openai_compat_content(model)
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
await model._agenerate([msg])
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == [{"type": "text", "text": "see"}, img]
def test_stream_preserves_media(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._stream
_patch_openai_compat_content(model)
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
list(model._stream([msg]))
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == [{"type": "text", "text": "see"}, img]
async def test_astream_preserves_media(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
received_msgs = []
async def _fake_astream(messages, *args, **kwargs):
received_msgs.extend(messages)
for chunk in ["c1", "c2"]:
yield chunk
model._astream = _fake_astream
_patch_openai_compat_content(model)
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
chunks = []
async for c in model._astream([msg]):
chunks.append(c)
assert chunks == ["c1", "c2"]
assert received_msgs[0].content == [{"type": "text", "text": "see"}, img]
def test_toolmessage_image_hoisted_to_human(self):
from langchain_core.messages import ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model) # hoist_tool_media=True (OpenAI-compat)
# deepagents read_file emits this exact shape for an image file.
tm = ToolMessage(
content_blocks=[
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
],
tool_call_id="tc1",
name="read_file",
)
model._generate([tm])
called_msgs = orig.call_args[0][0]
# Tool content becomes a string placeholder (OpenAI-compat requirement) ...
assert isinstance(called_msgs[0].content, str)
# ... and the image is hoisted into a following HumanMessage.
assert len(called_msgs) == 2
hoisted = called_msgs[1]
assert hoisted.type == "human"
assert any(
isinstance(b, dict) and b.get("type") == "image" for b in hoisted.content
)
def test_toolmessage_image_kept_inline_when_no_hoist(self):
from langchain_core.messages import ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model, hoist_tool_media=False) # Anthropic-routed
tm = ToolMessage(
content_blocks=[
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
],
tool_call_id="tc1",
name="read_file",
)
model._generate([tm])
called_msgs = orig.call_args[0][0]
# No hoisting: image stays inline in the tool message content.
assert len(called_msgs) == 1
content = called_msgs[0].content
assert isinstance(content, list)
assert any(isinstance(b, dict) and b.get("type") == "image" for b in content)
def test_parallel_tool_images_hoisted_after_tools(self):
from langchain_core.messages import AIMessage, ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
ai = AIMessage(
content="",
tool_calls=[
{"id": "c1", "name": "read_file", "args": {}},
{"id": "c2", "name": "read_file", "args": {}},
],
)
t1 = ToolMessage(
content_blocks=[
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
],
tool_call_id="c1",
name="read_file",
)
t2 = ToolMessage(
content_blocks=[
{"type": "image", "base64": "BBB", "mime_type": "image/png"}
],
tool_call_id="c2",
name="read_file",
)
model._generate([ai, t1, t2])
called_msgs = orig.call_args[0][0]
# Tool results stay consecutive; one hoisted HumanMessage follows them.
assert [m.type for m in called_msgs] == ["ai", "tool", "tool", "human"]
assert isinstance(called_msgs[1].content, str)
assert isinstance(called_msgs[2].content, str)
imgs = [b for b in called_msgs[3].content if b.get("type") == "image"]
assert len(imgs) == 2
def test_assistant_text_still_flattened_to_string(self):
from langchain_core.messages import AIMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
msg = AIMessage(
content=[
{"type": "text", "text": "hi"},
{"type": "thinking", "text": "t"},
]
)
model._generate([msg])
called_msgs = orig.call_args[0][0]
assert called_msgs[0].content == "hi"
def test_tool_media_flushed_before_next_human(self):
from langchain_core.messages import HumanMessage, ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
tm = ToolMessage(
content_blocks=[
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
],
tool_call_id="tc1",
name="read_file",
)
nxt = HumanMessage(content="thanks")
model._generate([tm, nxt])
called = orig.call_args[0][0]
# tool(placeholder), hoisted image (human), then the original human msg
assert [m.type for m in called] == ["tool", "human", "human"]
assert isinstance(called[0].content, str)
assert any(b.get("type") == "image" for b in called[1].content)
assert called[2].content == "thanks"
def test_tool_message_text_and_image_split(self):
from langchain_core.messages import ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
tm = ToolMessage(
content=[
{"type": "text", "text": "chart description"},
{"type": "image", "base64": "AAA", "mime_type": "image/png"},
],
tool_call_id="tc1",
name="read_file",
)
model._generate([tm])
called = orig.call_args[0][0]
# Tool keeps the text as its string content; image hoisted to a human msg.
assert called[0].content == "chart description"
assert any(b.get("type") == "image" for b in called[1].content)
def test_tool_message_interleaved_text_not_lost(self):
# Interleaved [text, image, text] in a tool result: BOTH text runs must
# survive the hoisting split (not just the first).
from langchain_core.messages import ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
orig = model._generate
_patch_openai_compat_content(model)
tm = ToolMessage(
content=[
{"type": "text", "text": "before"},
{"type": "image", "base64": "AAA", "mime_type": "image/png"},
{"type": "text", "text": "after"},
],
tool_call_id="tc1",
name="read_file",
)
model._generate([tm])
called = orig.call_args[0][0]
# both text runs preserved in the tool placeholder; image hoisted
assert "before" in called[0].content
assert "after" in called[0].content
assert any(b.get("type") == "image" for b in called[1].content)
# =============================================================================
# Test no-vision fallback (models that reject image input)
# =============================================================================
class TestNoVisionFallback:
"""Verify image-rejecting models fall back to a text placeholder."""
def _img_tool(self):
from langchain_core.messages import ToolMessage
return ToolMessage(
content_blocks=[
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
],
tool_call_id="t1",
name="read_file",
)
def _make_model(self):
from unittest.mock import MagicMock
model = MagicMock()
model._agenerate = None
model._stream = None
model._astream = None
return model
def test_media_error_types(self):
from EvoScientist.llm.patches import (
_FILE_CONTENT_TYPES,
_IMAGE_CONTENT_TYPES,
_is_http_400,
_media_error_types,
)
# marker identifies the specific modality
assert (
_media_error_types(Exception("No endpoints found that support image input"))
>= _IMAGE_CONTENT_TYPES
)
assert (
_media_error_types(Exception("file input is not supported"))
== _FILE_CONTENT_TYPES
)
# DeepSeek-style maps to all media (generic "expected text")
assert (
_media_error_types(
Exception("unknown variant `image_url`, expected `text`")
)
>= _IMAGE_CONTENT_TYPES
)
# non-media errors implicate nothing
assert _media_error_types(Exception("rate limit exceeded")) == set()
assert (
_media_error_types(Exception("No endpoints found for some/model")) == set()
)
# bare "expected text" (non-media schema error) must NOT match
assert (
_media_error_types(
Exception("tool schema validation failed: expected text")
)
== set()
)
class _E(Exception):
status_code = 400
assert _is_http_400(_E("bad request"))
assert not _is_http_400(Exception("rate limit exceeded"))
def test_media_types_in(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _media_types_in
img = {"type": "image", "base64": "A", "mime_type": "image/png"}
f = {"type": "file", "base64": "F", "mime_type": "application/pdf"}
assert _media_types_in([HumanMessage(content=[img, f])]) == {"image", "file"}
assert _media_types_in([HumanMessage(content="hi")]) == set()
def test_strip_media_types_replaces_only_given(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _strip_media_types
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
f = {"type": "file", "base64": "FFF", "mime_type": "application/pdf"}
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img, f])
# Strip only files -> image survives, file becomes a placeholder block.
out = _strip_media_types([msg], {"file"})
types = [b.get("type") for b in out[0].content if isinstance(b, dict)]
assert "image" in types # image preserved
assert "file" not in types # file stripped
assert any(
b.get("type") == "text" and "omitted" in b.get("text", "").lower()
for b in out[0].content
)
def test_strip_media_types_preserves_position(self):
# Stripped block is replaced IN PLACE; surrounding text/kept media keep
# their order (placeholder where the image was, file stays last).
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _strip_media_types
img = {"type": "image", "base64": "A", "mime_type": "image/png"}
f = {"type": "file", "base64": "F", "mime_type": "application/pdf"}
msg = HumanMessage(
content=[
{"type": "text", "text": "t1"},
img,
{"type": "text", "text": "t2"},
f,
]
)
out = _strip_media_types([msg], {"image"}) # block only image
content = out[0].content
assert all(b.get("type") != "image" for b in content) # image gone
# order preserved: t1, placeholder (where image was), t2, file
assert content[0]["text"] == "t1"
assert content[1]["type"] == "text"
assert "omitted" in content[1]["text"].lower()
assert content[2]["text"] == "t2"
assert content[3]["type"] == "file" # file kept at its original position
def test_strip_media_types_dedups_consecutive(self):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _strip_media_types
a = {"type": "image", "base64": "A", "mime_type": "image/png"}
b = {"type": "image", "base64": "B", "mime_type": "image/png"}
msg = HumanMessage(content=[a, b])
out = _strip_media_types([msg], {"image"})
# two adjacent stripped blocks collapse into ONE placeholder
assert len(out[0].content) == 1
assert "omitted" in out[0].content[0]["text"].lower()
def test_profile_no_vision_strips_upfront(self):
# Proactive: profile says image_inputs is False -> strip from the start,
# no failing first request.
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
model.profile = {"image_inputs": False}
calls = []
def _gen(msgs, *a, **k):
calls.append(msgs)
return "ok"
model._generate = _gen
_patch_openai_compat_content(model)
assert model._generate([self._img_tool()]) == "ok"
assert len(calls) == 1 # no failed attempt
assert all(isinstance(m.content, str) for m in calls[0])
assert any("omitted" in m.content.lower() for m in calls[0])
def test_profile_with_vision_does_not_strip(self):
# Profile says image_inputs is True -> normal preserve path (no upfront strip).
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
model.profile = {"image_inputs": True}
calls = []
def _gen(msgs, *a, **k):
calls.append(msgs)
return "ok"
model._generate = _gen
_patch_openai_compat_content(model)
assert model._generate([self._img_tool()]) == "ok"
# Image preserved (hoisted), not replaced by a placeholder.
assert any(
isinstance(m.content, list)
and any(b.get("type") == "image" for b in m.content)
for m in calls[0]
)
def test_generate_falls_back_and_caches(self):
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
calls = []
state = {"raised": False}
def _gen(msgs, *a, **k):
calls.append(msgs)
if not state["raised"]: # fail exactly once, ever
state["raised"] = True
raise Exception("unknown variant `image_url`, expected `text`")
return "ok"
model._generate = _gen
_patch_openai_compat_content(model)
tm = self._img_tool()
# 1st turn: preserve attempt fails once -> strip -> ok
assert model._generate([tm]) == "ok"
assert len(calls) == 2
retry = calls[1]
assert all(isinstance(m.content, str) for m in retry)
assert any("omitted" in m.content.lower() for m in retry)
# 2nd turn: cached no-vision -> straight to stripped, single call (no failure)
calls.clear()
assert model._generate([tm]) == "ok"
assert len(calls) == 1
assert all(isinstance(m.content, str) for m in calls[0])
def test_non_image_error_not_retried(self):
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
calls = []
def _gen(msgs, *a, **k):
calls.append(msgs)
raise Exception("rate limit exceeded")
model._generate = _gen
_patch_openai_compat_content(model)
with pytest.raises(Exception, match="rate limit"):
model._generate([self._img_tool()])
assert len(calls) == 1
def test_stream_falls_back(self):
from unittest.mock import MagicMock
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
model._generate = MagicMock(return_value="g")
calls = []
def _stream(msgs, *a, **k):
calls.append(msgs)
if len(calls) == 1:
raise Exception("No endpoints found that support image input")
yield from ["x", "y"]
model._stream = _stream
_patch_openai_compat_content(model)
out = list(model._stream([self._img_tool()]))
assert out == ["x", "y"]
assert len(calls) == 2
async def test_astream_falls_back(self):
from unittest.mock import MagicMock
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
model._generate = MagicMock(return_value="g")
calls = []
async def _astream(msgs, *a, **k):
calls.append(msgs)
if len(calls) == 1:
raise Exception("No endpoints found that support image input")
for c in ["x", "y"]:
yield c
model._astream = _astream
_patch_openai_compat_content(model)
out = [c async for c in model._astream([self._img_tool()])]
assert out == ["x", "y"]
assert len(calls) == 2
def test_unrelated_400_retry_fails_not_cached(self):
# A non-media 400 (e.g. tool schema) whose stripped retry ALSO fails must
# surface the original error and must NOT permanently flip to no-media.
from EvoScientist.llm.patches import _patch_openai_compat_content
class _E(Exception):
status_code = 400
model = self._make_model()
calls = []
def _gen(msgs, *a, **k):
calls.append(msgs)
raise _E("invalid tool schema") # 400, not media; fails every time
model._generate = _gen
_patch_openai_compat_content(model)
tm = self._img_tool()
with pytest.raises(_E):
model._generate([tm])
assert len(calls) == 2 # preserve attempt + stripped retry (both fail)
# Not cached: the next call attempts preserve again (not straight-to-stripped)
calls.clear()
with pytest.raises(_E):
model._generate([tm])
assert len(calls) == 2
def test_pdf_rejection_does_not_disable_images(self):
# Per-modality: a PDF/file rejection caches only file types; a later
# image must still be preserved (not stripped).
from langchain_core.messages import HumanMessage, ToolMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
calls = []
state = {"raised": False}
def _gen(msgs, *a, **k):
calls.append(msgs)
has_file = any(
isinstance(m.content, list)
and any(
isinstance(b, dict) and b.get("type") == "file" for b in m.content
)
for m in msgs
)
if has_file and not state["raised"]:
state["raised"] = True
raise Exception("file input is not supported")
return "ok"
model._generate = _gen
_patch_openai_compat_content(model)
pdf_tm = ToolMessage(
content_blocks=[
{"type": "file", "base64": "F", "mime_type": "application/pdf"}
],
tool_call_id="t1",
name="read_file",
)
assert model._generate([pdf_tm]) == "ok" # file rejected -> stripped -> ok
# Now an image: must still be preserved (images not blocked by a PDF reject)
calls.clear()
img_msg = HumanMessage(
content=[{"type": "image", "base64": "A", "mime_type": "image/png"}]
)
assert model._generate([img_msg]) == "ok"
assert len(calls) == 1 # single attempt, no failure
assert any(
isinstance(m.content, list)
and any(isinstance(b, dict) and b.get("type") == "image" for b in m.content)
for m in calls[0]
)
def test_bare_400_recovers_but_not_cached(self):
# A bare 400 with NO media marker recovers this request (stripped retry)
# but must NOT cache (no permanent degradation) — High #1.
from EvoScientist.llm.patches import _patch_openai_compat_content
class _E(Exception):
status_code = 400
model = self._make_model()
calls = []
state = {"raised": False}
def _gen(msgs, *a, **k):
calls.append(msgs)
if not state["raised"]:
state["raised"] = True
raise _E("transient bad request") # 400, no media marker
return "ok"
model._generate = _gen
_patch_openai_compat_content(model)
tm = self._img_tool()
assert model._generate([tm]) == "ok" # bare 400 -> stripped retry -> ok
assert len(calls) == 2
# NOT cached: the next call still attempts preserve (image kept, not stripped)
calls.clear()
assert model._generate([tm]) == "ok"
assert len(calls) == 1
assert any(
isinstance(m.content, list)
and any(isinstance(b, dict) and b.get("type") == "image" for b in m.content)
for m in calls[0]
)
def test_mixed_modality_caches_only_culprit(self):
# image+file message; provider rejects only the file -> cache file only,
# images stay preserved on later turns — High #2.
from langchain_core.messages import HumanMessage
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
calls = []
state = {"raised": False}
def _gen(msgs, *a, **k):
calls.append(msgs)
if not state["raised"]:
state["raised"] = True
raise Exception("file input is not supported")
return "ok"
model._generate = _gen
_patch_openai_compat_content(model)
mixed = HumanMessage(
content=[
{"type": "image", "base64": "A", "mime_type": "image/png"},
{"type": "file", "base64": "F", "mime_type": "application/pdf"},
]
)
assert model._generate([mixed]) == "ok" # file rejected -> retry -> cache file
# later image-only request: image must still be preserved
calls.clear()
img = HumanMessage(
content=[{"type": "image", "base64": "A", "mime_type": "image/png"}]
)
assert model._generate([img]) == "ok"
assert len(calls) == 1
assert any(
isinstance(m.content, list)
and any(isinstance(b, dict) and b.get("type") == "image" for b in m.content)
for m in calls[0]
)
def test_stream_empty_retry_raises_original(self):
# If the stripped streaming retry yields ZERO chunks, surface the
# original error instead of silently returning an empty stream.
from unittest.mock import MagicMock
from EvoScientist.llm.patches import _patch_openai_compat_content
model = self._make_model()
model._generate = MagicMock(return_value="g")
calls = []
def _stream(msgs, *a, **k):
calls.append(msgs)
if len(calls) == 1:
raise Exception("No endpoints found that support image input")
return # retry yields nothing
yield # pragma: no cover (makes this a generator)
model._stream = _stream
_patch_openai_compat_content(model)
with pytest.raises(Exception, match="support image"):
list(model._stream([self._img_tool()]))
assert len(calls) == 2
# =============================================================================
# Test DeepSeek model integration
# =============================================================================
def test_deepseek_model_strips_unsupported_tool_media(monkeypatch):
import json
import httpx
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
captured = {}
def respond(request: httpx.Request) -> httpx.Response:
captured.update(json.loads(request.content))
return httpx.Response(
200,
json={
"id": "chatcmpl-1",
"object": "chat.completion",
"created": 1,
"model": "deepseek-v4-flash",
"choices": [
{
"index": 0,
"finish_reason": "stop",
"message": {"role": "assistant", "content": "ok"},
}
],
"usage": {
"prompt_tokens": 1,
"completion_tokens": 1,
"total_tokens": 2,
},
},
)
with httpx.Client(transport=httpx.MockTransport(respond)) as client:
model = get_chat_model(
"deepseek-v4-flash",
provider="deepseek",
http_client=client,
)
model.invoke(
[
HumanMessage("inspect the file"),
AIMessage(
"",
tool_calls=[{"name": "read_file", "args": {}, "id": "call_1"}],
),
ToolMessage(
content_blocks=[
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
],
tool_call_id="call_1",
),
]
)
assert captured["messages"][2]["content"] == (
"[attachment omitted: this model does not support this input type]"
)
class TestDeepseekReasoningPassback:
"""Verify reasoning_content is retained in serialized DeepSeek history."""
def test_request_payload_preserves_reasoning_for_tool_history(self, monkeypatch):
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
from EvoScientist.llm.deepseek import EvoChatDeepSeek
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
model = EvoChatDeepSeek(model="deepseek-v4-flash")
messages = [
HumanMessage("q1"),
AIMessage(
"",
additional_kwargs={"reasoning_content": "rc1"},
tool_calls=[{"name": "read_file", "args": {}, "id": "call_1"}],
),
ToolMessage("result", tool_call_id="call_1"),
HumanMessage("q2"),
AIMessage("a2"),
HumanMessage("q3"),
AIMessage("a3", additional_kwargs={"reasoning_content": "rc3"}),
HumanMessage("q4"),
]
payload = model._get_request_payload(messages)
assert payload["messages"][1]["reasoning_content"] == "rc1"
assert "tool_calls" in payload["messages"][1]
assert "reasoning_content" not in payload["messages"][2]
assert payload["messages"][4]["reasoning_content"] == ""
assert payload["messages"][6]["reasoning_content"] == "rc3"
def test_thinking_disabled_copy_omits_reasoning_passback(self, monkeypatch):
from langchain_core.messages import AIMessage, HumanMessage
from EvoScientist.llm.deepseek import EvoChatDeepSeek
from EvoScientist.middleware.utils import disable_thinking
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
model = disable_thinking(EvoChatDeepSeek(model="deepseek-v4-flash"))
messages = [
HumanMessage("q1"),
AIMessage("a1", additional_kwargs={"reasoning_content": "rc1"}),
HumanMessage("q2"),
]
payload = model._get_request_payload(messages)
assert "reasoning_content" not in payload["messages"][1]
# =============================================================================
# Test _patch_openai_capture_reasoning_content (module-level monkey-patch)
# =============================================================================
class TestPatchOpenAICaptureReasoningContent:
"""Verify reasoning_content is captured into AIMessage.additional_kwargs.
This patch is applied at import time and globally affects langchain-openai's
_convert_dict_to_message and _convert_delta_to_message_chunk.
"""
def test_capture_from_non_streaming_response(self):
"""reasoning_content in OpenAI response dict → AIMessage.additional_kwargs."""
from langchain_openai.chat_models.base import _convert_dict_to_message
msg = _convert_dict_to_message(
{
"role": "assistant",
"content": "hi",
"reasoning_content": "let me think...",
}
)
assert msg.additional_kwargs.get("reasoning_content") == "let me think..."
def test_capture_absent_when_field_missing(self):
"""No reasoning_content in response → not added to additional_kwargs."""
from langchain_openai.chat_models.base import _convert_dict_to_message
msg = _convert_dict_to_message({"role": "assistant", "content": "hi"})
assert "reasoning_content" not in msg.additional_kwargs
def test_capture_from_streaming_chunk(self):
"""reasoning_content delta is captured onto the chunk's additional_kwargs."""
from langchain_core.messages import AIMessageChunk
from langchain_openai.chat_models.base import (
_convert_delta_to_message_chunk,
)
chunk = _convert_delta_to_message_chunk(
{"role": "assistant", "content": "", "reasoning_content": "thinking"},
AIMessageChunk,
)
assert chunk.additional_kwargs.get("reasoning_content") == "thinking"
def test_capture_does_not_affect_other_fields(self):
"""Existing tool_calls / function_call extraction unaffected."""
from langchain_openai.chat_models.base import _convert_dict_to_message
msg = _convert_dict_to_message(
{
"role": "assistant",
"content": "calling tool",
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {"name": "get_weather", "arguments": "{}"},
}
],
"reasoning_content": "use the tool",
}
)
assert len(msg.tool_calls) == 1
assert msg.tool_calls[0]["name"] == "get_weather"
assert msg.additional_kwargs.get("reasoning_content") == "use the tool"
class TestIsResponsesReasoningItem:
"""_is_responses_reasoning_item flags encrypted OpenAI-Responses items."""
def test_rs_id_is_responses_item(self):
from EvoScientist.llm.patches import _is_responses_reasoning_item
assert _is_responses_reasoning_item({"id": "rs_09363d42", "type": "x"})
def test_encrypted_data_is_responses_item(self):
from EvoScientist.llm.patches import _is_responses_reasoning_item
assert _is_responses_reasoning_item({"data": "gAAAAAB...", "type": "x"})
def test_plain_text_reasoning_is_not_responses_item(self):
from EvoScientist.llm.patches import _is_responses_reasoning_item
assert not _is_responses_reasoning_item(
{"type": "reasoning.text", "text": "thinking", "index": 0}
)
assert not _is_responses_reasoning_item("not a dict")
class TestPatchOpenrouterStripResponsesReasoning:
"""OpenAI-Responses encrypted reasoning items (`rs_` id / encrypted data)
are stripped from outgoing OpenRouter assistant messages, preventing the
multi-turn "Item with id 'rs_...' not found" 400 (store=false; #37777).
"""
def _apply(self):
import langchain_openrouter.chat_models as mod
import EvoScientist.llm.patches as patches
orig = mod._convert_message_to_dict
orig_flag = patches._openrouter_reasoning_strip_patched
patches._openrouter_reasoning_strip_patched = False
patches._patch_openrouter_strip_responses_reasoning()
return patches, mod, orig, orig_flag
@staticmethod
def _restore(patches, mod, orig, orig_flag):
mod._convert_message_to_dict = orig
patches._openrouter_reasoning_strip_patched = orig_flag
def test_strips_encrypted_item_drops_key_when_empty(self):
from langchain_core.messages import AIMessage
patches, mod, orig, orig_flag = self._apply()
try:
msg = AIMessage(
content="done",
additional_kwargs={
"reasoning_details": [
{
"type": "reasoning.summary",
"format": "openai-responses-v1",
"id": "rs_09363d42b054",
"data": "gAAAAAB...",
"summary": "real reasoning text",
"index": 0,
}
],
},
)
result = mod._convert_message_to_dict(msg)
# sole entry was an rs_ item → reasoning_details removed entirely.
assert "reasoning_details" not in result
finally:
self._restore(patches, mod, orig, orig_flag)
def test_keeps_plain_text_reasoning(self):
from langchain_core.messages import AIMessage
patches, mod, orig, orig_flag = self._apply()
try:
msg = AIMessage(
content="done",
additional_kwargs={
"reasoning_details": [
{"type": "reasoning.text", "text": "thinking", "index": 0},
{"id": "rs_abc", "data": "blob", "index": 1},
],
},
)
result = mod._convert_message_to_dict(msg)
kept = result["reasoning_details"]
assert len(kept) == 1
assert kept[0]["type"] == "reasoning.text"
finally:
self._restore(patches, mod, orig, orig_flag)
def test_does_not_mutate_original_message(self):
from langchain_core.messages import AIMessage
patches, mod, orig, orig_flag = self._apply()
try:
details = [{"id": "rs_abc", "data": "blob"}]
msg = AIMessage(
content="x", additional_kwargs={"reasoning_details": details}
)
mod._convert_message_to_dict(msg)
# stored history untouched — we filter a fresh list, not in place.
assert details == [{"id": "rs_abc", "data": "blob"}]
finally:
self._restore(patches, mod, orig, orig_flag)
def test_patch_is_idempotent(self):
patches, mod, orig, orig_flag = self._apply()
try:
wrapper = mod._convert_message_to_dict
# Second call is guarded by the flag → must not re-wrap.
patches._patch_openrouter_strip_responses_reasoning()
assert mod._convert_message_to_dict is wrapper
finally:
self._restore(patches, mod, orig, orig_flag)
def test_non_dict_entry_is_kept(self):
from langchain_core.messages import AIMessage
patches, mod, orig, orig_flag = self._apply()
try:
msg = AIMessage(
content="done",
additional_kwargs={
"reasoning_details": [
"opaque", # non-dict slipped in → kept, not crashed on
{"id": "rs_abc", "data": "blob", "index": 1},
],
},
)
result = mod._convert_message_to_dict(msg)
assert result["reasoning_details"] == ["opaque"]
finally:
self._restore(patches, mod, orig, orig_flag)
# =============================================================================
# Test _patch_anthropic_strip_foreign_reasoning
# =============================================================================
def _anthropic_httpx():
"""Return the httpx flavour the installed anthropic SDK accepts as ``http_client``.
anthropic >= 1.0 is built on ``httpx2`` and rejects an ``httpx.Client``.
"""
import anthropic
from packaging.version import Version
if Version(anthropic.__version__) >= Version("1"):
import httpx2 as httpx
else:
import httpx
return httpx
class TestAnthropicStripForeignReasoning:
def test_strip_removes_reasoning_content_blocks(self):
"""reasoning_content blocks are dropped; text and thinking survive."""
from langchain_core.messages import AIMessage, HumanMessage
from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
messages = [
HumanMessage("hello"),
AIMessage(
content=[
{"type": "reasoning_content", "reasoning_content": {"text": "hm"}},
{"type": "thinking", "thinking": "hm", "signature": ""},
{"type": "text", "text": "hi"},
]
),
]
result = _normalize_anthropic_replay_messages(messages)
types = [b["type"] for b in result[1].content]
assert types == ["thinking", "text"]
def test_missing_thinking_signature_defaulted(self):
"""Streamed thinking blocks without a signature key get signature ''."""
from langchain_core.messages import AIMessage
from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
messages = [
AIMessage(
content=[
{"type": "thinking", "thinking": "hm", "index": 0},
{"type": "text", "text": "hi", "index": 1},
]
),
]
result = _normalize_anthropic_replay_messages(messages)
assert result[0].content[0]["signature"] == ""
assert "signature" not in result[0].content[1]
def test_strip_no_change_returns_same_object(self):
"""Clean histories pass through without copying."""
from langchain_core.messages import AIMessage, HumanMessage
from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
messages = [
HumanMessage("hello"),
AIMessage(content=[{"type": "text", "text": "hi"}]),
AIMessage(content="plain string content"),
]
assert _normalize_anthropic_replay_messages(messages) is messages
def test_kimi_k3_exempt_from_flatten_patch(self, monkeypatch):
"""K3 on custom-anthropic gets no instance flatten closures; others do."""
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
kimi = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
assert "_generate" not in vars(kimi)
other = get_chat_model(
"claude-sonnet-4-6", provider="custom-anthropic", max_tokens=1024
)
assert "_generate" in vars(other)
def test_reasoning_content_stripped_on_the_wire(self, monkeypatch):
"""End-to-end: foreign reasoning blocks never reach the Anthropic wire."""
import json
import anthropic
from langchain_core.messages import AIMessage, HumanMessage
httpx = _anthropic_httpx()
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
captured: dict = {}
def handler(request: httpx.Request) -> httpx.Response:
captured.update(json.loads(request.content.decode()))
return httpx.Response(
200,
json={
"id": "msg_test",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "ok"}],
"model": "moonshotai/kimi-k3",
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 1, "output_tokens": 1},
},
)
model._client = anthropic.Anthropic(
api_key="test-key",
base_url="https://compat.example.com",
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
)
history = [
HumanMessage("hello"),
AIMessage(
content=[
{"type": "reasoning_content", "reasoning_content": {"text": "hm"}},
{"type": "thinking", "thinking": "hm", "index": 0},
{"type": "text", "text": "hi there"},
]
),
HumanMessage("say ok"),
]
result = model.invoke(history)
sent_blocks = [
block
for message in captured["messages"]
for block in (
message["content"] if isinstance(message["content"], list) else []
)
]
sent_types = [block["type"] for block in sent_blocks]
assert "reasoning_content" not in sent_types
assert "text" in sent_types
thinking_blocks = [b for b in sent_blocks if b["type"] == "thinking"]
assert thinking_blocks
assert thinking_blocks[0]["signature"] == ""
assert result.content == "ok"
# =============================================================================
# Test _patch_anthropic_structured_output
# =============================================================================
class TestAnthropicStructuredOutput:
@staticmethod
def _capture_structured_request(model, response_text):
"""Invoke a structured-output runnable against a capturing transport."""
import json
import anthropic
from pydantic import BaseModel
httpx = _anthropic_httpx()
class Pick(BaseModel):
answer: str
captured: dict = {}
def handler(request: httpx.Request) -> httpx.Response:
captured.update(json.loads(request.content.decode()))
return httpx.Response(
200,
json={
"id": "msg_test",
"type": "message",
"role": "assistant",
"content": response_text,
"model": "test",
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 1, "output_tokens": 1},
},
)
model._client = anthropic.Anthropic(
api_key="test-key",
base_url="https://compat.example.com",
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
)
result = model.with_structured_output(Pick).invoke("Reply with answer='ok'")
return captured, result
def test_kimi_k3_defaults_to_json_schema(self, monkeypatch):
"""K3 structured output binds output_config.format, no forced tool_choice."""
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
captured, result = self._capture_structured_request(
model, [{"type": "text", "text": '{"answer": "ok"}'}]
)
assert captured["output_config"]["format"]["type"] == "json_schema"
assert "tool_choice" not in captured
assert result.answer == "ok"
def test_claude_keeps_function_calling(self, monkeypatch):
"""Claude models keep tool-based structured output (no json_schema flip)."""
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key")
model = get_chat_model(
"claude-haiku-4-5", provider="anthropic", max_tokens=1024
)
captured, result = self._capture_structured_request(
model,
[
{
"type": "tool_use",
"id": "toolu_1",
"name": "Pick",
"input": {"answer": "ok"},
}
],
)
assert "output_config" not in captured
assert [t["name"] for t in captured["tools"]] == ["Pick"]
assert result.answer == "ok"
# =============================================================================
# Test _apply_auto_config
# =============================================================================
class TestAutoConfig:
@pytest.fixture(autouse=True)
def _clear_reasoning_effort_env(self, monkeypatch):
"""Keep auto-config tests independent of the developer environment."""
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_5_thinking(self, mock_init, monkeypatch):
"""Anthropic 4-5 models get enabled thinking with budget."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
get_chat_model("claude-haiku-4-5")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_6_adaptive_thinking(self, mock_init, monkeypatch):
"""Anthropic 4-6 models get adaptive thinking with max effort."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
get_chat_model("claude-sonnet-4-6")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
assert call_kwargs["effort"] == "max"
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_8_adaptive_thinking(self, mock_init, monkeypatch):
"""Anthropic 4-8 models get adaptive thinking with max effort."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
get_chat_model("claude-opus-4-8")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
assert call_kwargs["effort"] == "max"
@pytest.mark.parametrize("model", ["claude-opus-5", "claude-sonnet-5"])
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_5_series_adaptive_thinking(self, mock_init, model, monkeypatch):
"""Anthropic 5-series models get adaptive thinking (budget_tokens would 400)."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
get_chat_model(model, provider="anthropic")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
assert call_kwargs["effort"] == "max"
@pytest.mark.parametrize("model", ["moonshotai/kimi-k3", "kimi-k3"])
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_anthropic_kimi_k3_declares_thinking(
self, mock_init, model, monkeypatch
):
"""K3 via custom-anthropic declares thinking (else forced tool_choice 400s)."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
get_chat_model(model, provider="custom-anthropic")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
assert call_kwargs["max_tokens"] == 16000
@patch("EvoScientist.llm.models.init_chat_model")
def test_kimi_coding_declares_thinking(self, mock_init):
"""Kimi For Coding plan models declare thinking on the kimi-coding provider."""
mock_init.return_value = "mock_model"
get_chat_model("kimi-for-coding", provider="kimi-coding")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
assert call_kwargs["max_tokens"] == 16000
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_anthropic_non_kimi_no_thinking(self, mock_init, monkeypatch):
"""Non-Kimi models on custom-anthropic still skip thinking injection."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
get_chat_model("glm-4.7", provider="custom-anthropic")
call_kwargs = mock_init.call_args[1]
assert "thinking" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_6_proxy_no_thinking(self, mock_init, monkeypatch):
"""Anthropic 4-6 models via proxy skip thinking (history round-trip 422)."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("ANTHROPIC_BASE_URL", "http://127.0.0.1:8000")
monkeypatch.setenv("ANTHROPIC_API_KEY", "ccproxy-oauth")
get_chat_model("claude-sonnet-4-6")
call_kwargs = mock_init.call_args[1]
assert "thinking" not in call_kwargs
assert "effort" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_5_proxy_no_thinking(self, mock_init, monkeypatch):
"""Anthropic 4-5 models via proxy also skip thinking (history round-trip 422)."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("ANTHROPIC_BASE_URL", "http://127.0.0.1:8000")
monkeypatch.setenv("ANTHROPIC_API_KEY", "ccproxy-oauth")
get_chat_model("claude-haiku-4-5")
call_kwargs = mock_init.call_args[1]
assert "thinking" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_6_no_proxy_no_downgrade(self, mock_init, monkeypatch):
"""Anthropic 4-6 models without proxy still get adaptive thinking."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("ANTHROPIC_BASE_URL", "https://api.anthropic.com")
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-real")
get_chat_model("claude-sonnet-4-6")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
assert call_kwargs["effort"] == "max"
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_thinking_not_overridden(self, mock_init):
"""User-supplied thinking config should not be overridden."""
mock_init.return_value = "mock_model"
custom_thinking = {"type": "enabled", "budget_tokens": 500}
get_chat_model("claude-sonnet-4-6", thinking=custom_thinking)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["thinking"] == custom_thinking
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_reasoning_xhigh(self, mock_init, monkeypatch):
"""gpt-5.4+ and codex models get xhigh reasoning."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
get_chat_model("gpt-5.4", provider="openai")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "xhigh",
"summary": "auto",
}
get_chat_model("gpt-5.3-codex", provider="openai")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "xhigh",
"summary": "auto",
}
get_chat_model("gpt-5.5", provider="openai")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "xhigh",
"summary": "auto",
}
get_chat_model("gpt-5.6-sol", provider="openai")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "xhigh",
"summary": "auto",
}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_reasoning_effort_from_env(self, mock_init, monkeypatch):
"""Native OpenAI reasoning effort should be configurable via env var."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "high")
get_chat_model("gpt-5.5", provider="openai")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "high",
"summary": "auto",
}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_reasoning_high_fallback(self, mock_init, monkeypatch):
"""Other OpenAI models get high reasoning effort."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
get_chat_model("gpt-5-nano")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "high",
"summary": "auto",
}
get_chat_model("gpt-5.2", provider="openai")
assert mock_init.call_args[1]["reasoning"] == {
"effort": "high",
"summary": "auto",
}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_base_url_override(self, mock_init, monkeypatch):
"""OpenAI provider should support base_url override (e.g. ccproxy Codex)."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert call_kwargs["base_url"] == "http://127.0.0.1:8000/codex/v1"
assert call_kwargs["api_key"] == "ccproxy-oauth"
# ccproxy uses the Responses API, so reasoning configuration is valid.
assert call_kwargs["reasoning"] == {
"effort": "high",
"summary": "auto",
"context": "all_turns",
}
# Proxy mode: Responses API (bypasses format chain), streaming ON
assert call_kwargs["use_responses_api"] is True
assert "streaming" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_localhost_non_ccproxy_not_downgraded(self, mock_init, monkeypatch):
"""Local OpenAI-compatible endpoints (vLLM, etc.) are not affected by ccproxy workarounds."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8080/v1")
monkeypatch.setenv("OPENAI_API_KEY", "sk-local-key")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["base_url"] == "http://127.0.0.1:8080/v1"
# NOT ccproxy: reasoning should be applied, no forced Chat Completions
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
assert "use_responses_api" not in call_kwargs
assert "streaming" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_codex_path_but_wrong_key_not_ccproxy(self, mock_init, monkeypatch):
"""ccproxy detection requires both /codex/ path AND ccproxy-oauth key."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "sk-real-key")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
assert "use_responses_api" not in call_kwargs
assert "default_headers" not in call_kwargs
@patch(
"EvoScientist.llm.models._installed_codex_client_version",
return_value="0.144.1",
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_client_headers(
self, mock_init, mock_installed_version, monkeypatch
):
"""ccproxy Codex mode sends Codex-CLI-shaped client headers."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
monkeypatch.delenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", raising=False)
get_chat_model("gpt-5.5", provider="openai")
headers = mock_init.call_args[1]["default_headers"]
assert headers["originator"] == "codex_cli_rs"
assert headers["version"] == "0.144.1"
assert headers["User-Agent"].startswith("codex_cli_rs/0.144.1")
mock_installed_version.assert_called_once_with()
assert mock_init.call_args[1]["reasoning"]["effort"] == "xhigh"
assert mock_init.call_args[1]["reasoning"]["context"] == "all_turns"
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_client_version_env(self, mock_init, monkeypatch):
"""EVOSCIENTIST_CODEX_CLIENT_VERSION overrides the pinned version."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
monkeypatch.setenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", "9.9.9")
get_chat_model("gpt-5.5", provider="openai")
headers = mock_init.call_args[1]["default_headers"]
assert headers["version"] == "9.9.9"
assert headers["User-Agent"].startswith("codex_cli_rs/9.9.9")
@patch("EvoScientist.llm.models.subprocess.run")
def test_installed_codex_client_version(self, mock_run):
"""The advertised version follows the installed Codex CLI."""
from EvoScientist.llm.models import _installed_codex_client_version
mock_run.return_value.returncode = 0
mock_run.return_value.stdout = "codex-cli 0.144.1\n"
mock_run.return_value.stderr = ""
_installed_codex_client_version.cache_clear()
try:
assert _installed_codex_client_version() == "0.144.1"
assert _installed_codex_client_version() == "0.144.1"
finally:
_installed_codex_client_version.cache_clear()
mock_run.assert_called_once_with(
["codex", "--version"],
capture_output=True,
text=True,
timeout=2,
check=False,
)
@patch(
"EvoScientist.llm.models._installed_codex_client_version",
return_value="0.140.0",
)
def test_older_installed_codex_uses_fallback(
self, mock_installed_version, monkeypatch
):
"""An outdated installed CLI must not undercut the safe fallback."""
from EvoScientist.llm.models import (
_CODEX_CLIENT_VERSION_FALLBACK,
_resolve_codex_client_version,
)
monkeypatch.delenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", raising=False)
assert _resolve_codex_client_version() == _CODEX_CLIENT_VERSION_FALLBACK
mock_installed_version.assert_called_once_with()
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_headers_respect_caller(self, mock_init, monkeypatch):
"""Caller-supplied default_headers keys are not overridden."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
get_chat_model(
"gpt-5.5",
provider="openai",
default_headers={"originator": "codex_vscode", "version": "9.9.9"},
)
headers = mock_init.call_args[1]["default_headers"]
assert headers["originator"] == "codex_vscode"
assert headers["version"] == "9.9.9"
assert headers["User-Agent"].startswith("codex_cli_rs/9.9.9")
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_reasoning_context_respects_caller(
self, mock_init, monkeypatch
):
"""Caller-supplied reasoning.context is not overridden."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
reasoning = {"effort": "low", "context": "previous_response_id"}
get_chat_model(
"gpt-5.5",
provider="openai",
reasoning=reasoning,
)
assert mock_init.call_args[1]["reasoning"] == {
"effort": "low",
"context": "previous_response_id",
}
assert mock_init.call_args[1]["reasoning"] is not reasoning
assert reasoning == {"effort": "low", "context": "previous_response_id"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_reasoning_context_requires_responses_api(
self, mock_init, monkeypatch
):
"""Responses-only reasoning.context is not sent when Chat Completions is forced."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
monkeypatch.delenv("EVOSCIENTIST_USE_RESPONSES_API", raising=False)
get_chat_model(
"gpt-5.5",
provider="openai",
use_responses_api=False,
reasoning={"effort": "low"},
)
call_kwargs = mock_init.call_args[1]
assert call_kwargs["use_responses_api"] is False
assert call_kwargs["reasoning"] == {"effort": "low"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_env_false_drops_reasoning(
self, mock_init, monkeypatch
):
"""The global Chat Completions override still removes reasoning entirely."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "false")
get_chat_model("gpt-5.5", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["use_responses_api"] is False
assert "reasoning" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_codex_none_headers(self, mock_init, monkeypatch):
"""An explicit default_headers=None is normalized before gap-filling."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
monkeypatch.setenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", "9.9.9")
get_chat_model(
"gpt-5.5",
provider="openai",
default_headers=None,
)
headers = mock_init.call_args[1]["default_headers"]
assert headers["originator"] == "codex_cli_rs"
assert headers["version"] == "9.9.9"
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_ccproxy_key_but_wrong_path_not_ccproxy(
self, mock_init, monkeypatch
):
"""ccproxy detection requires both /codex/ path AND ccproxy-oauth key."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/v1")
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
assert "use_responses_api" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_openai_no_base_url_when_unset(self, mock_init, monkeypatch):
"""OpenAI provider should not set base_url when env var is empty."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.setenv("OPENAI_API_KEY", "sk-real")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["model_provider"] == "openai"
assert "base_url" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_google_thoughts(self, mock_init):
"""Google GenAI models get include_thoughts=True by default."""
mock_init.return_value = "mock_model"
get_chat_model("gemini-2.5-flash")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["include_thoughts"] is True
@patch("EvoScientist.llm.models.init_chat_model")
def test_use_responses_api_false(self, mock_init, monkeypatch):
"""use_responses_api=false forces Chat Completions and drops reasoning."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "false")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["use_responses_api"] is False
assert "reasoning" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_use_responses_api_true(self, mock_init, monkeypatch):
"""use_responses_api=true explicitly enables the Responses API."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "true")
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["use_responses_api"] is True
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_use_responses_api_default_unchanged(self, mock_init, monkeypatch):
"""Empty use_responses_api preserves default behavior (no kwarg set)."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.delenv("EVOSCIENTIST_USE_RESPONSES_API", raising=False)
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert "use_responses_api" not in call_kwargs
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
def test_responses_api_env_ignored_for_host_routed_deepseek(self, monkeypatch):
from langchain_core.messages import HumanMessage
from EvoScientist.llm.deepseek import EvoChatDeepSeek
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "sk-test")
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://api.deepseek.com")
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "true")
model = get_chat_model("deepseek-chat", provider="custom-openai")
assert isinstance(model, EvoChatDeepSeek)
assert model.use_responses_api is not True
assert "messages" in model._get_request_payload([HumanMessage("hi")])
@pytest.mark.parametrize("provider", ["deepseek", "custom-openai"])
def test_deepseek_rejects_explicit_responses_api(self, monkeypatch, provider):
if provider == "deepseek":
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
else:
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "sk-test")
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://api.deepseek.com")
with pytest.raises(ValueError, match="does not support the OpenAI Responses"):
get_chat_model(
"deepseek-chat",
provider=provider,
use_responses_api=True,
)
@pytest.mark.parametrize("env_value", ["FALSE", " false ", "False"])
@patch("EvoScientist.llm.models.init_chat_model")
def test_use_responses_api_false_normalization(
self, mock_init, monkeypatch, env_value
):
"""Case/whitespace variants of 'false' are normalized correctly."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", env_value)
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["use_responses_api"] is False
assert "reasoning" not in call_kwargs
@pytest.mark.parametrize("env_value", ["TRUE", " true ", "True"])
@patch("EvoScientist.llm.models.init_chat_model")
def test_use_responses_api_true_normalization(
self, mock_init, monkeypatch, env_value
):
"""Case/whitespace variants of 'true' are normalized correctly."""
mock_init.return_value = "mock_model"
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", env_value)
get_chat_model("gpt-5-nano", provider="openai")
call_kwargs = mock_init.call_args[1]
assert call_kwargs["use_responses_api"] is True
# =============================================================================
# Test validate_requesty_key
# =============================================================================
class TestValidateRequestyKey:
"""The Requesty key validator probes the router's auth layer.
Validation targets a deliberately nonexistent sentinel model
(``requesty/auth-preflight``) so key checks don't depend on any real
model staying available: the router resolves auth *before* the model,
so a valid key returns 404 (model-not-found) while a bad key returns
401/403. All HTTP calls are mocked — no network in unit tests.
"""
def test_empty_key_skipped(self):
"""No key provided is skipped, not an error."""
from EvoScientist.config.onboard.validators import validate_requesty_key
is_valid, msg = validate_requesty_key("")
assert is_valid is True
assert "Skipped" in msg
def test_uses_sentinel_model_not_a_real_one(self):
"""The probe targets a nonexistent sentinel model, not a real model."""
from EvoScientist.config.onboard.validators import validate_requesty_key
with patch("httpx.post") as mock_post:
mock_post.return_value.status_code = 404
validate_requesty_key("rq-key")
payload = mock_post.call_args.kwargs["json"]
assert payload["model"] == "requesty/auth-preflight"
assert payload["max_tokens"] == 1
def test_model_not_found_means_auth_passed(self):
"""404 (model not found) means auth was accepted → key is valid."""
from EvoScientist.config.onboard.validators import validate_requesty_key
with patch("httpx.post") as mock_post:
mock_post.return_value.status_code = 404
is_valid, msg = validate_requesty_key("rq-key")
assert is_valid is True
assert msg == "Valid"
def test_success_means_valid(self):
"""200 (accepted) also means the key is valid."""
from EvoScientist.config.onboard.validators import validate_requesty_key
with patch("httpx.post") as mock_post:
mock_post.return_value.status_code = 200
is_valid, msg = validate_requesty_key("rq-key")
assert is_valid is True
assert msg == "Valid"
@pytest.mark.parametrize("status", [401, 403])
def test_auth_rejected_means_invalid(self, status):
"""401/403 mean the key was rejected by the auth layer."""
from EvoScientist.config.onboard.validators import validate_requesty_key
with patch("httpx.post") as mock_post:
mock_post.return_value.status_code = status
is_valid, msg = validate_requesty_key("bad-key")
assert is_valid is False
assert msg == "Invalid API key"
@pytest.mark.parametrize("status", [429, 500, 502, 503])
def test_transient_status_is_inconclusive(self, status):
"""Rate-limit / 5xx leave key validity unknown, not rejected."""
from EvoScientist.config.onboard.validators import validate_requesty_key
with patch("httpx.post") as mock_post:
mock_post.return_value.status_code = status
is_valid, msg = validate_requesty_key("rq-key")
assert is_valid is False
assert "inconclusive" in msg.lower()
assert str(status) in msg