33979e5371
* fix(llm): support Volcengine Coding model aliases * refactor(llm): add Volcengine Coding provider * style: format Volcengine Coding test
3496 lines
137 KiB
Python
3496 lines
137 KiB
Python
"""Tests for EvoScientist LLM module."""
|
|
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
# Side-effect import: applies module-level monkey-patches (e.g.,
|
|
# _patch_openai_capture_reasoning_content) before tests reference patched
|
|
# functions from langchain_openai.
|
|
import EvoScientist.llm.patches # noqa: F401
|
|
from EvoScientist.llm import (
|
|
DEFAULT_MODEL,
|
|
MODELS,
|
|
get_chat_model,
|
|
get_model_info,
|
|
get_models_for_provider,
|
|
list_models,
|
|
)
|
|
from EvoScientist.llm.models import _MODEL_ENTRIES
|
|
|
|
# =============================================================================
|
|
# Test MODELS registry
|
|
# =============================================================================
|
|
|
|
|
|
class TestModelsRegistry:
|
|
def test_models_is_dict(self):
|
|
"""Test that MODELS is a dictionary."""
|
|
assert isinstance(MODELS, dict)
|
|
|
|
def test_entries_has_all_providers(self):
|
|
"""Test that _MODEL_ENTRIES covers all registered providers."""
|
|
providers = {p for _, _, p in _MODEL_ENTRIES}
|
|
assert "anthropic" in providers
|
|
assert "openai" in providers
|
|
assert "google-genai" in providers
|
|
assert "minimax" in providers
|
|
assert "nvidia" in providers
|
|
assert "siliconflow" in providers
|
|
assert "openrouter" in providers
|
|
assert "zhipu" in providers
|
|
assert "zhipu-code" in providers
|
|
assert "volcengine" in providers
|
|
assert "volcengine-code" in providers
|
|
assert "dashscope" in providers
|
|
assert "dashscope-code" in providers
|
|
assert "deepseek" in providers
|
|
assert "moonshot" in providers
|
|
assert "kimi-coding" in providers
|
|
assert "atlascloud" in providers
|
|
|
|
def test_entries_are_valid_tuples(self):
|
|
"""Test that _MODEL_ENTRIES contains valid (name, model_id, provider) tuples."""
|
|
valid_providers = {
|
|
"anthropic",
|
|
"openai",
|
|
"google-genai",
|
|
"minimax",
|
|
"nvidia",
|
|
"siliconflow",
|
|
"openrouter",
|
|
"requesty",
|
|
"zhipu",
|
|
"zhipu-code",
|
|
"volcengine",
|
|
"volcengine-code",
|
|
"dashscope",
|
|
"dashscope-code",
|
|
"custom-openai",
|
|
"custom-anthropic",
|
|
"deepseek",
|
|
"moonshot",
|
|
"kimi-coding",
|
|
"atlascloud",
|
|
}
|
|
for entry in _MODEL_ENTRIES:
|
|
assert len(entry) == 3, f"Entry {entry} doesn't have 3 elements"
|
|
name, model_id, provider = entry
|
|
assert isinstance(name, str)
|
|
assert isinstance(model_id, str)
|
|
assert provider in valid_providers, (
|
|
f"Unknown provider '{provider}' for '{name}'"
|
|
)
|
|
|
|
def test_get_models_for_provider(self):
|
|
"""Test that get_models_for_provider returns correct models."""
|
|
anthropic_models = get_models_for_provider("anthropic")
|
|
assert len(anthropic_models) > 0
|
|
for name, model_id in anthropic_models:
|
|
assert isinstance(name, str)
|
|
assert isinstance(model_id, str)
|
|
|
|
# Third-party providers now have registered models
|
|
openrouter_models = get_models_for_provider("openrouter")
|
|
assert len(openrouter_models) > 0
|
|
siliconflow_models = get_models_for_provider("siliconflow")
|
|
assert len(siliconflow_models) > 0
|
|
atlas_models = get_models_for_provider("atlascloud")
|
|
assert ("qwen3.5-27b", "qwen/qwen3.5-27b") in atlas_models
|
|
assert get_models_for_provider("atlas") == []
|
|
|
|
|
|
# =============================================================================
|
|
# Test DEFAULT_MODEL
|
|
# =============================================================================
|
|
|
|
|
|
class TestDefaultModel:
|
|
def test_default_model_exists_in_registry(self):
|
|
"""Test that DEFAULT_MODEL is a valid model in MODELS."""
|
|
assert DEFAULT_MODEL in MODELS
|
|
|
|
def test_default_model_is_anthropic(self):
|
|
"""Test that default model uses Anthropic."""
|
|
_, provider = MODELS[DEFAULT_MODEL]
|
|
assert provider == "anthropic"
|
|
|
|
|
|
# =============================================================================
|
|
# Test list_models
|
|
# =============================================================================
|
|
|
|
|
|
class TestListModels:
|
|
def test_returns_list(self):
|
|
"""Test that list_models returns a list."""
|
|
result = list_models()
|
|
assert isinstance(result, list)
|
|
|
|
def test_returns_all_model_names(self):
|
|
"""Test that list_models returns all model names."""
|
|
result = list_models()
|
|
assert set(result) == set(MODELS.keys())
|
|
|
|
def test_list_is_not_empty(self):
|
|
"""Test that the list is not empty."""
|
|
assert len(list_models()) > 0
|
|
|
|
|
|
# =============================================================================
|
|
# Test get_model_info
|
|
# =============================================================================
|
|
|
|
|
|
class TestGetModelInfo:
|
|
def test_returns_tuple_for_valid_model(self):
|
|
"""Test that get_model_info returns tuple for valid model."""
|
|
result = get_model_info("claude-sonnet-4-6")
|
|
assert result is not None
|
|
assert isinstance(result, tuple)
|
|
assert len(result) == 2
|
|
|
|
def test_returns_none_for_invalid_model(self):
|
|
"""Test that get_model_info returns None for invalid model."""
|
|
result = get_model_info("nonexistent-model")
|
|
assert result is None
|
|
|
|
def test_returns_correct_info(self):
|
|
"""Test that get_model_info returns correct info."""
|
|
model_id, provider = get_model_info("gpt-5-nano")
|
|
assert model_id == "gpt-5-nano"
|
|
assert provider == "openai"
|
|
|
|
|
|
# =============================================================================
|
|
# Test get_chat_model
|
|
# =============================================================================
|
|
|
|
|
|
class TestGetChatModel:
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_uses_default_model_when_none(self, mock_init):
|
|
"""Test that get_chat_model uses default model when model=None."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model()
|
|
|
|
mock_init.assert_called_once()
|
|
call_kwargs = mock_init.call_args[1]
|
|
# Default model should be resolved from MODELS
|
|
expected_model_id, expected_provider = MODELS[DEFAULT_MODEL]
|
|
assert call_kwargs["model"] == expected_model_id
|
|
assert call_kwargs["model_provider"] == expected_provider
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_resolves_short_name(self, mock_init):
|
|
"""Test that get_chat_model resolves short names correctly."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("claude-opus-4-8")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "claude-opus-4-8"
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_resolves_openai_short_name(self, mock_init):
|
|
"""Test that get_chat_model resolves OpenAI short names."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("gpt-5-mini")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "gpt-5-mini"
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_uses_full_model_id(self, mock_init):
|
|
"""Test that get_chat_model accepts full model IDs."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("claude-3-opus-20240229")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "claude-3-opus-20240229"
|
|
# Should infer anthropic from the model prefix
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_provider_override(self, mock_init):
|
|
"""Test that provider can be overridden."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("claude-sonnet-4-6", provider="custom_provider")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "custom_provider"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_passes_kwargs(self, mock_init):
|
|
"""Test that additional kwargs are passed through."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("gpt-5-nano", temperature=0.7, max_tokens=1000)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["temperature"] == 0.7
|
|
assert call_kwargs["max_tokens"] == 1000
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_infers_openai_from_gpt_prefix(self, mock_init):
|
|
"""Test that OpenAI is inferred from gpt- prefix."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("gpt-4-turbo-preview")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_infers_openai_from_o1_prefix(self, mock_init):
|
|
"""Test that OpenAI is inferred from o1 prefix."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("o1-preview")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_infers_google_from_gemini_prefix(self, mock_init):
|
|
"""Test that google-genai is inferred from gemini prefix."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("gemini-2.0-flash")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "google-genai"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_defaults_to_anthropic_for_unknown(self, mock_init):
|
|
"""Test that anthropic is default for unknown model prefixes."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("some-unknown-model")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
|
|
|
|
# =============================================================================
|
|
# Test Ollama provider
|
|
# =============================================================================
|
|
|
|
|
|
class TestOllamaProvider:
|
|
"""Ollama models are not in the static registry (detected dynamically).
|
|
All tests use explicit provider or ollama: prefix."""
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_explicit_provider(self, mock_init):
|
|
"""Test that explicit provider='ollama' routes correctly."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("llama3.1:8b", provider="ollama")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "llama3.1:8b"
|
|
assert call_kwargs["model_provider"] == "ollama"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_ollama_base_url_passthrough(self, mock_init, monkeypatch):
|
|
"""Test that OLLAMA_BASE_URL env var is passed to kwargs."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OLLAMA_BASE_URL", "http://gpu-cluster:11434")
|
|
|
|
get_chat_model("llama3.1:8b", provider="ollama")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["base_url"] == "http://gpu-cluster:11434"
|
|
assert call_kwargs["model_provider"] == "ollama"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_ollama_no_base_url_when_unset(self, mock_init, monkeypatch):
|
|
"""Test that base_url is not set when OLLAMA_BASE_URL is empty."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
|
|
|
|
get_chat_model("llama3.1:8b", provider="ollama")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "base_url" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_reasoning_auto_enabled_for_ollama(self, mock_init, monkeypatch):
|
|
"""Test that reasoning is auto-enabled for Ollama models."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
|
|
|
|
get_chat_model("llama3.1:8b", provider="ollama")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "thinking" not in call_kwargs
|
|
assert call_kwargs["reasoning"] is True
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_reasoning_not_overridden_for_ollama(self, mock_init, monkeypatch):
|
|
"""Test that explicit reasoning=False is not overridden for Ollama."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
|
|
|
|
get_chat_model("llama3.1:8b", provider="ollama", reasoning=False)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["reasoning"] is False
|
|
|
|
def test_no_static_registry_entries(self):
|
|
"""Test that Ollama has no static registry entries (models detected dynamically)."""
|
|
ollama_models = get_models_for_provider("ollama")
|
|
assert len(ollama_models) == 0
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_ollama_prefix_inference(self, mock_init, monkeypatch):
|
|
"""Test that ollama: prefix infers ollama provider."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OLLAMA_BASE_URL", raising=False)
|
|
|
|
get_chat_model("ollama:phi3:mini")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "phi3:mini"
|
|
assert call_kwargs["model_provider"] == "ollama"
|
|
|
|
|
|
# =============================================================================
|
|
# Test slash model ID no longer routes to nvidia
|
|
# =============================================================================
|
|
|
|
|
|
class TestSlashModelIdFallback:
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_slash_model_id_defaults_to_anthropic(self, mock_init):
|
|
"""Unregistered model IDs containing '/' should NOT route to nvidia.
|
|
|
|
They fall through to the default 'anthropic' provider, consistent
|
|
with how all other unknown model IDs are handled.
|
|
"""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("some-org/some-model")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "some-org/some-model"
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
|
|
|
|
# =============================================================================
|
|
# Test third-party provider routing
|
|
# =============================================================================
|
|
|
|
|
|
def _sdk_retry_supports_status_codes_override() -> bool:
|
|
"""openrouter>=0.11 only; the 429 override degrades to a no-op below that."""
|
|
from openrouter.utils.retries import RetryConfig
|
|
|
|
return "status_codes_override" in getattr(RetryConfig, "__annotations__", {})
|
|
|
|
|
|
class TestThirdPartyRouting:
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_atlascloud_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""Atlas Cloud should use OpenAI-compatible routing with its default URL."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("ATLASCLOUD_API_KEY", "atlas-key-123")
|
|
|
|
get_chat_model("qwen3.5-27b", provider="atlascloud")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "qwen/qwen3.5-27b"
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://api.atlascloud.ai/v1"
|
|
assert call_kwargs["api_key"] == "atlas-key-123"
|
|
assert "reasoning" not in call_kwargs
|
|
|
|
def test_atlascloud_host_maps_to_provider(self):
|
|
"""Provider error envelopes should identify Atlas Cloud by host."""
|
|
from EvoScientist.llm.errors import _lookup_host_or_compat
|
|
|
|
assert (
|
|
_lookup_host_or_compat("https://api.atlascloud.ai/v1", "openai")
|
|
== "atlascloud"
|
|
)
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_siliconflow_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""SiliconFlow provider should route through OpenAI with correct base_url."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("SILICONFLOW_API_KEY", "sf-key-123")
|
|
|
|
get_chat_model("Pro/zai-org/GLM-5", provider="siliconflow")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://api.siliconflow.cn/v1"
|
|
assert call_kwargs["api_key"] == "sf-key-123"
|
|
# SiliconFlow should disable thinking
|
|
assert call_kwargs["extra_body"]["enable_thinking"] is False
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_requesty_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""Requesty provider should route through OpenAI with correct base_url."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key-123")
|
|
|
|
get_chat_model("openai/gpt-4o-mini", provider="requesty")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://router.requesty.ai/v1"
|
|
assert call_kwargs["api_key"] == "rq-key-123"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_requesty_anthropic_prompt_cache_enabled_by_default(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Requesty Anthropic prompt caching should be opt-out."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key")
|
|
monkeypatch.delenv(
|
|
"EVOSCIENTIST_REQUESTY_ANTHROPIC_PROMPT_CACHE", raising=False
|
|
)
|
|
|
|
get_chat_model("anthropic/claude-sonnet-4-6", provider="requesty")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://router.requesty.ai/v1"
|
|
assert call_kwargs["model_kwargs"]["cache_control"] == {"type": "ephemeral"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_requesty_anthropic_prompt_cache_opt_out(self, mock_init, monkeypatch):
|
|
"""The opt-out flag should skip caching for Requesty Claude models."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_REQUESTY_ANTHROPIC_PROMPT_CACHE", "false")
|
|
|
|
get_chat_model("anthropic/claude-sonnet-4-6", provider="requesty")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "cache_control" not in call_kwargs
|
|
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_requesty_prompt_cache_skips_non_anthropic(self, mock_init, monkeypatch):
|
|
"""Requesty caching should not touch non-Anthropic models."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("REQUESTY_API_KEY", "rq-key")
|
|
monkeypatch.delenv(
|
|
"EVOSCIENTIST_REQUESTY_ANTHROPIC_PROMPT_CACHE", raising=False
|
|
)
|
|
|
|
get_chat_model("openai/gpt-4o-mini", provider="requesty")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "cache_control" not in call_kwargs
|
|
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_deepseek_uses_copy_safe_native_model(self, mock_init, monkeypatch):
|
|
from EvoScientist.llm.deepseek import EvoChatDeepSeek
|
|
|
|
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
|
|
|
|
model = get_chat_model("deepseek-v4-flash", provider="deepseek")
|
|
|
|
mock_init.assert_not_called()
|
|
assert isinstance(model, EvoChatDeepSeek)
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_uses_native_provider(self, mock_init, monkeypatch):
|
|
"""OpenRouter should use native 'openrouter' provider via init_chat_model."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key-456")
|
|
# Assert the DEFAULT effort, so isolate from any leaked env override.
|
|
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openrouter"
|
|
assert call_kwargs["api_key"] == "or-key-456"
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_reasoning_user_override(self, mock_init, monkeypatch):
|
|
"""User-supplied reasoning config should not be overridden."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
|
|
get_chat_model(
|
|
"x-ai/grok-4.3",
|
|
provider="openrouter",
|
|
reasoning={"effort": "low"},
|
|
)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["reasoning"] == {"effort": "low"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_reasoning_effort_from_env(self, mock_init, monkeypatch):
|
|
"""Reasoning effort should be configurable via env var."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "medium")
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["reasoning"] == {"effort": "medium", "summary": "auto"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_moonshot_thinking_disable_exempts_kimi_k3(self, mock_init, monkeypatch):
|
|
"""Native Moonshot: K3 must not receive the K2.x thinking-disable field.
|
|
|
|
Moonshot's K3 guide forbids the K2.x `thinking` parameter (K3 is
|
|
always-thinking); other Moonshot models keep the disable that guards
|
|
against multi-turn error 20015.
|
|
"""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("MOONSHOT_API_KEY", "ms-key")
|
|
|
|
get_chat_model("kimi-k3", provider="moonshot")
|
|
extra_body = mock_init.call_args[1].get("extra_body") or {}
|
|
assert "thinking" not in extra_body
|
|
|
|
get_chat_model("kimi-k2.6", provider="moonshot")
|
|
extra_body = mock_init.call_args[1]["extra_body"]
|
|
assert extra_body["thinking"] == {"type": "disabled"}
|
|
|
|
# --- OpenRouter upstream 429 retry ---
|
|
|
|
@pytest.mark.skipif(
|
|
not _sdk_retry_supports_status_codes_override(),
|
|
reason="openrouter<0.11 RetryConfig lacks status_codes_override; "
|
|
"the 429 override no-ops there by design (models.py hasattr guard)",
|
|
)
|
|
def test_openrouter_429_added_to_retryable_status_codes(self, monkeypatch):
|
|
"""Upstream 429s must become retryable on the real SDK client.
|
|
|
|
The openrouter SDK hardcodes per-operation retryable statuses to
|
|
["5XX"], so a launch-day "temporarily rate-limited upstream" 429
|
|
(Retry-After: 1) fails the run outright instead of being retried.
|
|
"""
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
|
|
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
|
|
|
|
retry_config = model.client.sdk_configuration.retry_config
|
|
assert retry_config.status_codes_override == ["429", "5XX"]
|
|
|
|
def test_openrouter_429_override_not_injected_when_retries_disabled(
|
|
self, monkeypatch
|
|
):
|
|
"""max_retries=0 leaves the SDK retry config UNSET — no 429 override.
|
|
|
|
Note this only asserts our override is absent; the SDK still applies
|
|
its own per-operation default (backoff on 5XX) when the config is
|
|
UNSET, so retries as such are not fully disabled at the SDK level.
|
|
"""
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
|
|
model = get_chat_model("x-ai/grok-4.3", provider="openrouter", max_retries=0)
|
|
|
|
retry_config = model.client.sdk_configuration.retry_config
|
|
assert getattr(retry_config, "status_codes_override", None) is None
|
|
|
|
@pytest.mark.skipif(
|
|
not _sdk_retry_supports_status_codes_override(),
|
|
reason="openrouter<0.11 RetryConfig lacks status_codes_override; "
|
|
"the 429 override no-ops there by design (models.py hasattr guard)",
|
|
)
|
|
def test_openrouter_429_retried_on_the_wire(self, monkeypatch):
|
|
"""End-to-end: a 429 with Retry-After is retried and the retry succeeds."""
|
|
import httpx
|
|
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
|
|
|
|
calls = {"n": 0}
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
calls["n"] += 1
|
|
if calls["n"] == 1:
|
|
return httpx.Response(
|
|
429,
|
|
headers={"Retry-After": "1"},
|
|
json={"error": {"message": "Provider returned error", "code": 429}},
|
|
)
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "gen-1",
|
|
"object": "chat.completion",
|
|
"created": 1,
|
|
"model": "moonshotai/kimi-k3",
|
|
"system_fingerprint": "fp-test",
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
},
|
|
)
|
|
|
|
model.client.sdk_configuration.client = httpx.Client(
|
|
transport=httpx.MockTransport(handler)
|
|
)
|
|
|
|
result = model.invoke("hi")
|
|
|
|
assert calls["n"] == 2
|
|
assert result.content == "ok"
|
|
|
|
# --- OpenRouter structured output vs mandatory reasoning ---
|
|
|
|
@staticmethod
|
|
def _capture_structured_request(model, structured, response_message):
|
|
"""Invoke a structured-output runnable against a capturing transport."""
|
|
import json
|
|
|
|
import httpx
|
|
|
|
captured: dict = {}
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
captured.update(json.loads(request.content.decode()))
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "gen-1",
|
|
"object": "chat.completion",
|
|
"created": 1,
|
|
"model": "m",
|
|
"system_fingerprint": "fp-test",
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": response_message,
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
},
|
|
)
|
|
|
|
model.client.sdk_configuration.client = httpx.Client(
|
|
transport=httpx.MockTransport(handler)
|
|
)
|
|
result = structured.invoke("pick tools")
|
|
return captured, result
|
|
|
|
def test_openrouter_structured_output_json_schema_for_mandatory_model(
|
|
self, monkeypatch
|
|
):
|
|
"""with_structured_output must not force tool_choice on kimi-k3.
|
|
|
|
Moonshot rejects a forced tool choice with HTTP 400 "tool_choice
|
|
'specified' is incompatible with thinking enabled", and kimi-k3's
|
|
thinking cannot be disabled — so the default function_calling method
|
|
400s every structured-output call (LLMToolSelectorMiddleware included).
|
|
The json_schema method (response_format) is supported and needs none.
|
|
"""
|
|
from pydantic import BaseModel
|
|
|
|
class ToolSelection(BaseModel):
|
|
tools: list[str]
|
|
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
|
|
# The dated canonical_slug is routable too and must be equally covered.
|
|
for model_name in ("moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"):
|
|
model = get_chat_model(model_name, provider="openrouter")
|
|
structured = model.with_structured_output(ToolSelection)
|
|
|
|
captured, result = self._capture_structured_request(
|
|
model,
|
|
structured,
|
|
{"role": "assistant", "content": '{"tools": ["tavily_search"]}'},
|
|
)
|
|
|
|
assert "tool_choice" not in captured, model_name
|
|
assert captured["response_format"]["type"] == "json_schema", model_name
|
|
assert result == ToolSelection(tools=["tavily_search"])
|
|
|
|
def test_openrouter_structured_output_default_for_other_models(self, monkeypatch):
|
|
"""Non-Moonshot models keep the function_calling default.
|
|
|
|
Includes always-thinking models like grok-4.5 — the forced tool_choice
|
|
restriction is Moonshot-specific, so the json_schema rerouting must
|
|
stay limited to _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS.
|
|
"""
|
|
from pydantic import BaseModel
|
|
|
|
class ToolSelection(BaseModel):
|
|
tools: list[str]
|
|
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
|
|
for model_name in ("x-ai/grok-4.3", "x-ai/grok-4.5"):
|
|
model = get_chat_model(model_name, provider="openrouter")
|
|
structured = model.with_structured_output(ToolSelection)
|
|
|
|
captured, result = self._capture_structured_request(
|
|
model,
|
|
structured,
|
|
{
|
|
"role": "assistant",
|
|
"content": "",
|
|
"tool_calls": [
|
|
{
|
|
"id": "call_1",
|
|
"type": "function",
|
|
"function": {
|
|
"name": "ToolSelection",
|
|
"arguments": '{"tools": ["tavily_search"]}',
|
|
},
|
|
}
|
|
],
|
|
},
|
|
)
|
|
|
|
assert captured.get("tool_choice"), model_name
|
|
assert "response_format" not in captured, model_name
|
|
assert result == ToolSelection(tools=["tavily_search"])
|
|
|
|
# --- OpenRouter app attribution (issue #339) ---
|
|
|
|
_APP_ATTR_ENV = (
|
|
"EVOSCIENTIST_OPENROUTER_HTTP_REFERER",
|
|
"EVOSCIENTIST_OPENROUTER_APP_TITLE",
|
|
"EVOSCIENTIST_OPENROUTER_APP_CATEGORIES",
|
|
)
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_attribution_defaults(self, mock_init, monkeypatch):
|
|
"""OpenRouter init should carry EvoScientist's default app attribution."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
# Isolate from any leaked env overrides so we assert the built-in defaults.
|
|
for _env in self._APP_ATTR_ENV:
|
|
monkeypatch.delenv(_env, raising=False)
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["app_url"] == "https://github.com/EvoScientist/EvoScientist"
|
|
assert call_kwargs["app_title"] == "EvoScientist"
|
|
# Must be a list[str] (not the comma string) — langchain-openrouter joins it.
|
|
assert call_kwargs["app_categories"] == ["creative-writing", "personal-agent"]
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_attribution_from_env(self, mock_init, monkeypatch):
|
|
"""Env vars should override the default app attribution values."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_HTTP_REFERER", "https://acme.test")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_TITLE", "Acme")
|
|
# Include a space to prove each category is stripped.
|
|
monkeypatch.setenv(
|
|
"EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", "cli-agent, programming-app"
|
|
)
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["app_url"] == "https://acme.test"
|
|
assert call_kwargs["app_title"] == "Acme"
|
|
assert call_kwargs["app_categories"] == ["cli-agent", "programming-app"]
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_attribution_user_override_not_clobbered(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Caller-supplied attribution kwargs must beat both env and defaults."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
# Env is also set, to prove an explicit kwarg outranks the env override
|
|
# (not just the built-in default).
|
|
monkeypatch.setenv(
|
|
"EVOSCIENTIST_OPENROUTER_HTTP_REFERER", "https://env.example"
|
|
)
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_TITLE", "EnvTitle")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", "env-cat")
|
|
|
|
get_chat_model(
|
|
"x-ai/grok-4.3",
|
|
provider="openrouter",
|
|
app_url="https://mine.example",
|
|
app_title="MyApp",
|
|
app_categories=["only-this"],
|
|
)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["app_url"] == "https://mine.example"
|
|
assert call_kwargs["app_title"] == "MyApp"
|
|
# An explicit list is preserved verbatim, not re-split.
|
|
assert call_kwargs["app_categories"] == ["only-this"]
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_non_openrouter_providers_get_no_app_attribution(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Only the openrouter provider should receive app-attribution kwargs."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-real")
|
|
monkeypatch.setenv("OLLAMA_BASE_URL", "http://localhost:11434")
|
|
|
|
for model, provider in (
|
|
("claude-sonnet-4-6", "anthropic"),
|
|
("llama3.1:8b", "ollama"),
|
|
):
|
|
get_chat_model(model, provider=provider)
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "app_url" not in call_kwargs
|
|
assert "app_title" not in call_kwargs
|
|
assert "app_categories" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_attribution_coexists_with_reasoning_and_cache(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Attribution must not disturb reasoning or Anthropic prompt caching."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
|
|
monkeypatch.delenv(
|
|
"EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", raising=False
|
|
)
|
|
for _env in self._APP_ATTR_ENV:
|
|
monkeypatch.delenv(_env, raising=False)
|
|
|
|
get_chat_model("claude-sonnet-4.6", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
# Existing behavior intact.
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
assert call_kwargs["model_kwargs"]["cache_control"] == {"type": "ephemeral"}
|
|
# Attribution added alongside.
|
|
assert call_kwargs["app_url"] == "https://github.com/EvoScientist/EvoScientist"
|
|
assert call_kwargs["app_title"] == "EvoScientist"
|
|
assert call_kwargs["app_categories"] == ["creative-writing", "personal-agent"]
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_categories_env_strips_blank_items(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""A messy comma value (stray commas / spaces) yields a clean list."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", "a,, b ")
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
assert mock_init.call_args[1]["app_categories"] == ["a", "b"]
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_categories_capped_to_per_request_limit(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Over-configuring categories caps to the first N and warns the user."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv(
|
|
"EVOSCIENTIST_OPENROUTER_APP_CATEGORIES",
|
|
"cli-agent,programming-app,personal-agent,writing-assistant",
|
|
)
|
|
|
|
with pytest.warns(UserWarning, match="at most 2 app categories"):
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
# OpenRouter honors at most 2 per request, so only the first 2 are sent.
|
|
assert mock_init.call_args[1]["app_categories"] == [
|
|
"cli-agent",
|
|
"programming-app",
|
|
]
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_app_categories_all_separators_omit_kwarg(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""A categories value with no real items omits the kwarg entirely."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_APP_CATEGORIES", " , , ")
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
# No app_categories kwarg at all — not an empty list (which the library
|
|
# would reject / send as an empty header).
|
|
assert "app_categories" not in mock_init.call_args[1]
|
|
|
|
def test_openrouter_app_attribution_lands_on_real_model(self, monkeypatch):
|
|
"""Build a REAL ChatOpenRouter (no mock) and assert the attribution
|
|
values land on the instance rather than being silently dumped into
|
|
model_kwargs.
|
|
|
|
The mocked tests above assert on the kwargs handed to init_chat_model,
|
|
so they cannot catch a param-name typo or a langchain-openrouter version
|
|
that accepts these only as passthrough model params (which the library
|
|
does with a warning, not an error). This test is the guard for both.
|
|
"""
|
|
from langchain_openrouter import ChatOpenRouter
|
|
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
for _env in self._APP_ATTR_ENV:
|
|
monkeypatch.delenv(_env, raising=False)
|
|
|
|
model = get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
assert isinstance(model, ChatOpenRouter)
|
|
assert model.app_url == "https://github.com/EvoScientist/EvoScientist"
|
|
assert model.app_title == "EvoScientist"
|
|
assert model.app_categories == ["creative-writing", "personal-agent"]
|
|
# Not silently swallowed into model_kwargs (the passthrough failure mode).
|
|
model_kwargs = model.model_kwargs or {}
|
|
assert "app_url" not in model_kwargs
|
|
assert "app_title" not in model_kwargs
|
|
assert "app_categories" not in model_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_anthropic_prompt_cache_enabled_by_default(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""OpenRouter Anthropic prompt caching should be opt-out."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.delenv(
|
|
"EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", raising=False
|
|
)
|
|
|
|
get_chat_model("claude-sonnet-4.6", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openrouter"
|
|
assert call_kwargs["model"] == "anthropic/claude-sonnet-4.6"
|
|
assert call_kwargs["model_kwargs"]["cache_control"] == {"type": "ephemeral"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_anthropic_prompt_cache_opt_out(self, mock_init, monkeypatch):
|
|
"""The opt-out flag should skip caching for OpenRouter Claude models."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "false")
|
|
|
|
get_chat_model("claude-sonnet-4.6", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "cache_control" not in call_kwargs
|
|
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_prompt_cache_default_skips_non_anthropic_openrouter(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""OpenRouter models with implicit caching should be left alone."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
|
|
|
|
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "cache_control" not in call_kwargs
|
|
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_anthropic_prompt_cache_preserves_top_level_override(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""The default should not duplicate a caller's cache_control kwarg."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
|
|
override = {"type": "ephemeral", "ttl": "1h"}
|
|
|
|
get_chat_model(
|
|
"claude-sonnet-4.6",
|
|
provider="openrouter",
|
|
cache_control=override,
|
|
)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["cache_control"] == override
|
|
assert "cache_control" not in call_kwargs.get("model_kwargs", {})
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_anthropic_prompt_cache_preserves_model_kwargs_override(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""The default should not duplicate model_kwargs cache_control."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
|
|
override = {"type": "ephemeral", "ttl": "1h"}
|
|
|
|
get_chat_model(
|
|
"claude-sonnet-4.6",
|
|
provider="openrouter",
|
|
model_kwargs={"cache_control": override},
|
|
)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "cache_control" not in call_kwargs
|
|
assert call_kwargs["model_kwargs"]["cache_control"] == override
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openrouter_anthropic_prompt_cache_warns_on_invalid_model_kwargs(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Invalid model_kwargs shape should warn and skip cache injection."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
|
monkeypatch.setenv("EVOSCIENTIST_OPENROUTER_ANTHROPIC_PROMPT_CACHE", "true")
|
|
|
|
with pytest.warns(UserWarning, match="model_kwargs` is not a dict"):
|
|
get_chat_model(
|
|
"claude-sonnet-4.6",
|
|
provider="openrouter",
|
|
model_kwargs="bad",
|
|
)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_kwargs"] == "bad"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_custom_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""Custom provider should route through OpenAI with env-configured base_url."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://my-llm.example.com/v1")
|
|
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "custom-key-789")
|
|
|
|
get_chat_model("my-custom-model", provider="custom-openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://my-llm.example.com/v1"
|
|
assert call_kwargs["api_key"] == "custom-key-789"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_base_url_override(self, mock_init, monkeypatch):
|
|
"""Anthropic provider should support base_url override (e.g. ccproxy)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("ANTHROPIC_BASE_URL", "http://localhost:8000/api/v1")
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-dummy")
|
|
|
|
get_chat_model("claude-sonnet-4-6", provider="anthropic")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
assert call_kwargs["base_url"] == "http://localhost:8000/api/v1"
|
|
assert call_kwargs["api_key"] == "sk-dummy"
|
|
# Proxy mode: thinking skipped (history round-trip causes 422)
|
|
assert "thinking" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_no_base_url_when_unset(self, mock_init, monkeypatch):
|
|
"""Anthropic provider should not set base_url when env var is empty."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-real")
|
|
|
|
get_chat_model("claude-sonnet-4-6", provider="anthropic")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
assert "base_url" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_third_party_no_reasoning(self, mock_init, monkeypatch):
|
|
"""Third-party providers routed through OpenAI should NOT get auto-reasoning."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("SILICONFLOW_API_KEY", "sf-key")
|
|
|
|
get_chat_model("deepseek-v3", provider="siliconflow")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "reasoning" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_volcengine_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""Volcengine provider should route through OpenAI with correct base_url."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("VOLCENGINE_API_KEY", "ve-key-123")
|
|
|
|
get_chat_model("doubao-seed-1.6", provider="volcengine")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://ark.cn-beijing.volces.com/api/v3"
|
|
assert call_kwargs["api_key"] == "ve-key-123"
|
|
|
|
@pytest.mark.parametrize(
|
|
("configured_model", "api_model"),
|
|
[("glm-5.2", "glm-5-2"), ("kimi-k2.5", "kimi-k2-5")],
|
|
)
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_volcengine_code_routes_through_openai(
|
|
self, mock_init, configured_model, api_model, monkeypatch
|
|
):
|
|
"""Volcengine Coding Plan uses its endpoint, IDs, and vendor API key."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("VOLCENGINE_API_KEY", "ve-code-key-123")
|
|
|
|
get_chat_model(configured_model, provider="volcengine-code")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["model"] == api_model
|
|
assert (
|
|
call_kwargs["base_url"] == "https://ark.cn-beijing.volces.com/api/coding/v3"
|
|
)
|
|
assert call_kwargs["api_key"] == "ve-code-key-123"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_dashscope_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""DashScope provider should route through OpenAI with correct base_url."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key-456")
|
|
|
|
get_chat_model("qwen-max", provider="dashscope")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert (
|
|
call_kwargs["base_url"]
|
|
== "https://dashscope.aliyuncs.com/compatible-mode/v1"
|
|
)
|
|
assert call_kwargs["api_key"] == "ds-key-456"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_dashscope_code_routes_through_openai(self, mock_init, monkeypatch):
|
|
"""DashScope-Code (sk-sp-* subscription keys) routes through OpenAI
|
|
with the coding.dashscope.aliyuncs.com base URL, reusing DASHSCOPE_API_KEY.
|
|
"""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-sp-key-789")
|
|
|
|
get_chat_model("qwen3-coder", provider="dashscope-code")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "https://coding.dashscope.aliyuncs.com/v1"
|
|
assert call_kwargs["api_key"] == "sk-sp-key-789"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_minimax_routes_through_anthropic(self, mock_init, monkeypatch):
|
|
"""MiniMax provider should route through Anthropic with correct base_url."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key-123")
|
|
|
|
get_chat_model("MiniMax-M2.5", provider="minimax")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
assert call_kwargs["base_url"] == "https://api.minimaxi.com/anthropic"
|
|
assert call_kwargs["api_key"] == "mm-key-123"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_minimax_base_url_env_override(self, mock_init, monkeypatch):
|
|
"""MINIMAX_BASE_URL env var should override the default base URL."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key-123")
|
|
monkeypatch.setenv("MINIMAX_BASE_URL", "https://api.minimax.io/anthropic")
|
|
|
|
get_chat_model("MiniMax-M2.5", provider="minimax")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["base_url"] == "https://api.minimax.io/anthropic"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_minimax_gets_thinking(self, mock_init, monkeypatch):
|
|
"""MiniMax provider should get auto-thinking (thinking-capable via Anthropic)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key")
|
|
|
|
get_chat_model("MiniMax-M2.5", provider="minimax")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "thinking" in call_kwargs
|
|
assert "reasoning" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_minimax_short_name_resolution(self, mock_init, monkeypatch):
|
|
"""MiniMax short names should resolve to correct model IDs."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key")
|
|
|
|
get_chat_model("minimax-m2.5", provider="minimax")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "MiniMax-M2.5"
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_minimax_highspeed_model(self, mock_init, monkeypatch):
|
|
"""MiniMax M2.5-highspeed model should resolve correctly."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("MINIMAX_API_KEY", "mm-key")
|
|
|
|
get_chat_model("minimax-m2.5-highspeed", provider="minimax")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model"] == "MiniMax-M2.5-highspeed"
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
assert call_kwargs["base_url"] == "https://api.minimaxi.com/anthropic"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_custom_anthropic_via_routed_dict(self, mock_init, monkeypatch):
|
|
"""custom-anthropic should work via _ANTHROPIC_ROUTED_PROVIDERS dict."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://my-claude.example.com")
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "ca-key-789")
|
|
|
|
get_chat_model("claude-sonnet-4-6", provider="custom-anthropic")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "anthropic"
|
|
assert call_kwargs["base_url"] == "https://my-claude.example.com"
|
|
assert call_kwargs["api_key"] == "ca-key-789"
|
|
# custom-anthropic is NOT thinking-capable → thinking skipped
|
|
assert "thinking" not in call_kwargs
|
|
|
|
|
|
# =============================================================================
|
|
# Test MiniMax provider
|
|
# =============================================================================
|
|
|
|
|
|
class TestMiniMaxProvider:
|
|
def test_minimax_in_anthropic_routed_providers(self):
|
|
"""MiniMax should be registered in _ANTHROPIC_ROUTED_PROVIDERS."""
|
|
from EvoScientist.llm.models import _ANTHROPIC_ROUTED_PROVIDERS
|
|
|
|
assert "minimax" in _ANTHROPIC_ROUTED_PROVIDERS
|
|
base_url, api_key_env = _ANTHROPIC_ROUTED_PROVIDERS["minimax"]
|
|
assert base_url == "https://api.minimaxi.com/anthropic"
|
|
assert api_key_env == "MINIMAX_API_KEY"
|
|
|
|
def test_minimax_not_in_openai_routed_providers(self):
|
|
"""MiniMax should NOT be in _OPENAI_ROUTED_PROVIDERS (moved to Anthropic)."""
|
|
from EvoScientist.llm.models import _OPENAI_ROUTED_PROVIDERS
|
|
|
|
assert "minimax" not in _OPENAI_ROUTED_PROVIDERS
|
|
|
|
def test_minimax_models_registered(self):
|
|
"""MiniMax should have 5 direct model entries in _MODEL_ENTRIES."""
|
|
minimax_models = get_models_for_provider("minimax")
|
|
assert len(minimax_models) == 5
|
|
model_names = {name for name, _ in minimax_models}
|
|
assert "minimax-m3" in model_names
|
|
assert "minimax-m2.7" in model_names
|
|
assert "minimax-m2.7-highspeed" in model_names
|
|
assert "minimax-m2.5" in model_names
|
|
assert "minimax-m2.5-highspeed" in model_names
|
|
|
|
def test_minimax_model_ids_correct(self):
|
|
"""MiniMax model IDs should match the official API model names."""
|
|
minimax_models = get_models_for_provider("minimax")
|
|
model_dict = dict(minimax_models)
|
|
assert model_dict["minimax-m2.7"] == "MiniMax-M2.7"
|
|
assert model_dict["minimax-m2.5"] == "MiniMax-M2.5"
|
|
assert model_dict["minimax-m2.5-highspeed"] == "MiniMax-M2.5-highspeed"
|
|
|
|
def test_minimax_short_name_in_models_dict(self):
|
|
"""MiniMax short names should be accessible via the MODELS dict."""
|
|
# Note: MODELS dict uses last-entry-wins, so direct minimax entries
|
|
# may be overridden by nvidia/siliconflow/openrouter entries.
|
|
# Use get_models_for_provider() for provider-specific lookups.
|
|
minimax_models = get_models_for_provider("minimax")
|
|
assert len(minimax_models) > 0
|
|
|
|
|
|
# =============================================================================
|
|
# Test _flatten_message_content
|
|
# =============================================================================
|
|
|
|
|
|
class TestFlattenMessageContent:
|
|
"""Tests for the content-flattening utility used by OpenAI-compatible providers."""
|
|
|
|
def test_string_passthrough(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
assert _flatten_message_content("hello") == "hello"
|
|
|
|
def test_non_list_passthrough(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
assert _flatten_message_content(42) == 42
|
|
assert _flatten_message_content(None) is None
|
|
|
|
def test_text_blocks(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [
|
|
{"type": "text", "text": "Hello"},
|
|
{"type": "text", "text": "World"},
|
|
]
|
|
assert _flatten_message_content(content) == "Hello\n\nWorld"
|
|
|
|
def test_skips_thinking_blocks(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [
|
|
{"type": "thinking", "text": "Let me think..."},
|
|
{"type": "text", "text": "The answer is 42"},
|
|
{"type": "reasoning", "text": "internal reasoning"},
|
|
{"type": "reasoning_content", "text": "more reasoning"},
|
|
]
|
|
assert _flatten_message_content(content) == "The answer is 42"
|
|
|
|
def test_string_blocks(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = ["hello", "world"]
|
|
assert _flatten_message_content(content) == "hello\n\nworld"
|
|
|
|
def test_mixed_blocks(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [
|
|
{"type": "thinking", "text": "skip me"},
|
|
"plain string",
|
|
{"type": "text", "text": "dict text"},
|
|
]
|
|
assert _flatten_message_content(content) == "plain string\n\ndict text"
|
|
|
|
def test_empty_list(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
assert _flatten_message_content([]) == ""
|
|
|
|
def test_only_thinking_blocks(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [{"type": "thinking", "text": "thought"}]
|
|
assert _flatten_message_content(content) == ""
|
|
|
|
def test_preserves_image_block(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
assert _flatten_message_content([img]) == [img]
|
|
|
|
def test_preserves_image_url_block(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
img = {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}}
|
|
assert _flatten_message_content([img]) == [img]
|
|
|
|
def test_preserves_file_block(self):
|
|
# PDF/document files are preserved (capable models read them).
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
f = {"type": "file", "base64": "FFF", "mime_type": "application/pdf"}
|
|
assert _flatten_message_content([f]) == [f]
|
|
|
|
def test_unsupported_media_dropped(self):
|
|
# video/audio are NOT in the allowlist -> dropped, not crashing
|
|
# (langchain-openai raises ValueError on `video`).
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
for block in (
|
|
{"type": "video", "base64": "VVV", "mime_type": "video/mp4"},
|
|
{"type": "audio", "base64": "ZZZ", "mime_type": "audio/wav"},
|
|
):
|
|
assert _flatten_message_content([block]) == ""
|
|
|
|
def test_non_image_media_dropped_keeps_text(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [
|
|
{"type": "text", "text": "hi"},
|
|
{"type": "video", "base64": "VVV", "mime_type": "video/mp4"},
|
|
]
|
|
# Video dropped, text kept -> plain string (no media list).
|
|
assert _flatten_message_content(content) == "hi"
|
|
|
|
def test_consolidates_text_and_image(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
content = [{"type": "text", "text": "a photo"}, img]
|
|
assert _flatten_message_content(content) == [
|
|
{"type": "text", "text": "a photo"},
|
|
img,
|
|
]
|
|
|
|
def test_multiple_text_blocks_with_image(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
content = [
|
|
{"type": "text", "text": "a"},
|
|
{"type": "text", "text": "b"},
|
|
img,
|
|
]
|
|
assert _flatten_message_content(content) == [
|
|
{"type": "text", "text": "a\n\nb"},
|
|
img,
|
|
]
|
|
|
|
def test_preserves_text_media_ordering(self):
|
|
# Text after an image must stay AFTER it (not consolidated to the front).
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
content = [
|
|
{"type": "text", "text": "before"},
|
|
img,
|
|
{"type": "text", "text": "after"},
|
|
]
|
|
assert _flatten_message_content(content) == [
|
|
{"type": "text", "text": "before"},
|
|
img,
|
|
{"type": "text", "text": "after"},
|
|
]
|
|
|
|
def test_thinking_dropped_image_kept(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
content = [{"type": "thinking", "text": "hmm"}, img]
|
|
assert _flatten_message_content(content) == [img]
|
|
|
|
def test_pure_text_still_returns_string(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [{"type": "text", "text": "x"}, {"type": "text", "text": "y"}]
|
|
result = _flatten_message_content(content)
|
|
assert result == "x\n\ny"
|
|
assert isinstance(result, str)
|
|
|
|
def test_unknown_nontext_block_still_dropped(self):
|
|
from EvoScientist.llm.patches import _flatten_message_content
|
|
|
|
content = [{"type": "tool_use", "id": "1", "name": "foo"}]
|
|
assert _flatten_message_content(content) == ""
|
|
|
|
|
|
# =============================================================================
|
|
# Test _patch_openai_compat_content (all 4 paths)
|
|
# =============================================================================
|
|
|
|
|
|
class TestPatchOpenAICompatContent:
|
|
"""Verify content flattening covers _generate, _agenerate, _stream, _astream."""
|
|
|
|
def _make_model(self):
|
|
"""Create a minimal mock model with all 4 methods."""
|
|
from unittest.mock import AsyncMock, MagicMock
|
|
|
|
model = MagicMock()
|
|
model._generate = MagicMock(return_value="gen_result")
|
|
model._agenerate = AsyncMock(return_value="agen_result")
|
|
model._stream = MagicMock(return_value=iter(["chunk1"]))
|
|
model._astream = AsyncMock()
|
|
return model
|
|
|
|
def test_generate_flattened(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
|
|
model._generate([msg])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == "hello"
|
|
|
|
async def test_agenerate_flattened(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._agenerate
|
|
_patch_openai_compat_content(model)
|
|
|
|
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
|
|
await model._agenerate([msg])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == "hello"
|
|
|
|
def test_stream_flattened(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._stream
|
|
_patch_openai_compat_content(model)
|
|
|
|
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
|
|
list(model._stream([msg]))
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == "hello"
|
|
|
|
async def test_astream_flattened(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
received_msgs = []
|
|
|
|
async def _fake_astream(messages, *args, **kwargs):
|
|
received_msgs.extend(messages)
|
|
for chunk in ["c1", "c2"]:
|
|
yield chunk
|
|
|
|
model._astream = _fake_astream
|
|
_patch_openai_compat_content(model)
|
|
|
|
msg = HumanMessage(content=[{"type": "text", "text": "hello"}])
|
|
chunks = []
|
|
async for c in model._astream([msg]):
|
|
chunks.append(c)
|
|
|
|
assert chunks == ["c1", "c2"]
|
|
assert received_msgs[0].content == "hello"
|
|
|
|
def test_generate_preserves_media(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
|
|
model._generate([msg])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == [{"type": "text", "text": "see"}, img]
|
|
|
|
async def test_agenerate_preserves_media(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._agenerate
|
|
_patch_openai_compat_content(model)
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
|
|
await model._agenerate([msg])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == [{"type": "text", "text": "see"}, img]
|
|
|
|
def test_stream_preserves_media(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._stream
|
|
_patch_openai_compat_content(model)
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
|
|
list(model._stream([msg]))
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == [{"type": "text", "text": "see"}, img]
|
|
|
|
async def test_astream_preserves_media(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
received_msgs = []
|
|
|
|
async def _fake_astream(messages, *args, **kwargs):
|
|
received_msgs.extend(messages)
|
|
for chunk in ["c1", "c2"]:
|
|
yield chunk
|
|
|
|
model._astream = _fake_astream
|
|
_patch_openai_compat_content(model)
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img])
|
|
chunks = []
|
|
async for c in model._astream([msg]):
|
|
chunks.append(c)
|
|
|
|
assert chunks == ["c1", "c2"]
|
|
assert received_msgs[0].content == [{"type": "text", "text": "see"}, img]
|
|
|
|
def test_toolmessage_image_hoisted_to_human(self):
|
|
from langchain_core.messages import ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model) # hoist_tool_media=True (OpenAI-compat)
|
|
|
|
# deepagents read_file emits this exact shape for an image file.
|
|
tm = ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="tc1",
|
|
name="read_file",
|
|
)
|
|
model._generate([tm])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
# Tool content becomes a string placeholder (OpenAI-compat requirement) ...
|
|
assert isinstance(called_msgs[0].content, str)
|
|
# ... and the image is hoisted into a following HumanMessage.
|
|
assert len(called_msgs) == 2
|
|
hoisted = called_msgs[1]
|
|
assert hoisted.type == "human"
|
|
assert any(
|
|
isinstance(b, dict) and b.get("type") == "image" for b in hoisted.content
|
|
)
|
|
|
|
def test_toolmessage_image_kept_inline_when_no_hoist(self):
|
|
from langchain_core.messages import ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model, hoist_tool_media=False) # Anthropic-routed
|
|
|
|
tm = ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="tc1",
|
|
name="read_file",
|
|
)
|
|
model._generate([tm])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
# No hoisting: image stays inline in the tool message content.
|
|
assert len(called_msgs) == 1
|
|
content = called_msgs[0].content
|
|
assert isinstance(content, list)
|
|
assert any(isinstance(b, dict) and b.get("type") == "image" for b in content)
|
|
|
|
def test_parallel_tool_images_hoisted_after_tools(self):
|
|
from langchain_core.messages import AIMessage, ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
ai = AIMessage(
|
|
content="",
|
|
tool_calls=[
|
|
{"id": "c1", "name": "read_file", "args": {}},
|
|
{"id": "c2", "name": "read_file", "args": {}},
|
|
],
|
|
)
|
|
t1 = ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="c1",
|
|
name="read_file",
|
|
)
|
|
t2 = ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "BBB", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="c2",
|
|
name="read_file",
|
|
)
|
|
model._generate([ai, t1, t2])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
# Tool results stay consecutive; one hoisted HumanMessage follows them.
|
|
assert [m.type for m in called_msgs] == ["ai", "tool", "tool", "human"]
|
|
assert isinstance(called_msgs[1].content, str)
|
|
assert isinstance(called_msgs[2].content, str)
|
|
imgs = [b for b in called_msgs[3].content if b.get("type") == "image"]
|
|
assert len(imgs) == 2
|
|
|
|
def test_assistant_text_still_flattened_to_string(self):
|
|
from langchain_core.messages import AIMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
msg = AIMessage(
|
|
content=[
|
|
{"type": "text", "text": "hi"},
|
|
{"type": "thinking", "text": "t"},
|
|
]
|
|
)
|
|
model._generate([msg])
|
|
|
|
called_msgs = orig.call_args[0][0]
|
|
assert called_msgs[0].content == "hi"
|
|
|
|
def test_tool_media_flushed_before_next_human(self):
|
|
from langchain_core.messages import HumanMessage, ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
tm = ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="tc1",
|
|
name="read_file",
|
|
)
|
|
nxt = HumanMessage(content="thanks")
|
|
model._generate([tm, nxt])
|
|
|
|
called = orig.call_args[0][0]
|
|
# tool(placeholder), hoisted image (human), then the original human msg
|
|
assert [m.type for m in called] == ["tool", "human", "human"]
|
|
assert isinstance(called[0].content, str)
|
|
assert any(b.get("type") == "image" for b in called[1].content)
|
|
assert called[2].content == "thanks"
|
|
|
|
def test_tool_message_text_and_image_split(self):
|
|
from langchain_core.messages import ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
tm = ToolMessage(
|
|
content=[
|
|
{"type": "text", "text": "chart description"},
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"},
|
|
],
|
|
tool_call_id="tc1",
|
|
name="read_file",
|
|
)
|
|
model._generate([tm])
|
|
|
|
called = orig.call_args[0][0]
|
|
# Tool keeps the text as its string content; image hoisted to a human msg.
|
|
assert called[0].content == "chart description"
|
|
assert any(b.get("type") == "image" for b in called[1].content)
|
|
|
|
def test_tool_message_interleaved_text_not_lost(self):
|
|
# Interleaved [text, image, text] in a tool result: BOTH text runs must
|
|
# survive the hoisting split (not just the first).
|
|
from langchain_core.messages import ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
orig = model._generate
|
|
_patch_openai_compat_content(model)
|
|
|
|
tm = ToolMessage(
|
|
content=[
|
|
{"type": "text", "text": "before"},
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"},
|
|
{"type": "text", "text": "after"},
|
|
],
|
|
tool_call_id="tc1",
|
|
name="read_file",
|
|
)
|
|
model._generate([tm])
|
|
|
|
called = orig.call_args[0][0]
|
|
# both text runs preserved in the tool placeholder; image hoisted
|
|
assert "before" in called[0].content
|
|
assert "after" in called[0].content
|
|
assert any(b.get("type") == "image" for b in called[1].content)
|
|
|
|
|
|
# =============================================================================
|
|
# Test no-vision fallback (models that reject image input)
|
|
# =============================================================================
|
|
|
|
|
|
class TestNoVisionFallback:
|
|
"""Verify image-rejecting models fall back to a text placeholder."""
|
|
|
|
def _img_tool(self):
|
|
from langchain_core.messages import ToolMessage
|
|
|
|
return ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="t1",
|
|
name="read_file",
|
|
)
|
|
|
|
def _make_model(self):
|
|
from unittest.mock import MagicMock
|
|
|
|
model = MagicMock()
|
|
model._agenerate = None
|
|
model._stream = None
|
|
model._astream = None
|
|
return model
|
|
|
|
def test_media_error_types(self):
|
|
from EvoScientist.llm.patches import (
|
|
_FILE_CONTENT_TYPES,
|
|
_IMAGE_CONTENT_TYPES,
|
|
_is_http_400,
|
|
_media_error_types,
|
|
)
|
|
|
|
# marker identifies the specific modality
|
|
assert (
|
|
_media_error_types(Exception("No endpoints found that support image input"))
|
|
>= _IMAGE_CONTENT_TYPES
|
|
)
|
|
assert (
|
|
_media_error_types(Exception("file input is not supported"))
|
|
== _FILE_CONTENT_TYPES
|
|
)
|
|
# DeepSeek-style maps to all media (generic "expected text")
|
|
assert (
|
|
_media_error_types(
|
|
Exception("unknown variant `image_url`, expected `text`")
|
|
)
|
|
>= _IMAGE_CONTENT_TYPES
|
|
)
|
|
# non-media errors implicate nothing
|
|
assert _media_error_types(Exception("rate limit exceeded")) == set()
|
|
assert (
|
|
_media_error_types(Exception("No endpoints found for some/model")) == set()
|
|
)
|
|
# bare "expected text" (non-media schema error) must NOT match
|
|
assert (
|
|
_media_error_types(
|
|
Exception("tool schema validation failed: expected text")
|
|
)
|
|
== set()
|
|
)
|
|
|
|
class _E(Exception):
|
|
status_code = 400
|
|
|
|
assert _is_http_400(_E("bad request"))
|
|
assert not _is_http_400(Exception("rate limit exceeded"))
|
|
|
|
def test_media_types_in(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _media_types_in
|
|
|
|
img = {"type": "image", "base64": "A", "mime_type": "image/png"}
|
|
f = {"type": "file", "base64": "F", "mime_type": "application/pdf"}
|
|
assert _media_types_in([HumanMessage(content=[img, f])]) == {"image", "file"}
|
|
assert _media_types_in([HumanMessage(content="hi")]) == set()
|
|
|
|
def test_strip_media_types_replaces_only_given(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _strip_media_types
|
|
|
|
img = {"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
f = {"type": "file", "base64": "FFF", "mime_type": "application/pdf"}
|
|
msg = HumanMessage(content=[{"type": "text", "text": "see"}, img, f])
|
|
# Strip only files -> image survives, file becomes a placeholder block.
|
|
out = _strip_media_types([msg], {"file"})
|
|
types = [b.get("type") for b in out[0].content if isinstance(b, dict)]
|
|
assert "image" in types # image preserved
|
|
assert "file" not in types # file stripped
|
|
assert any(
|
|
b.get("type") == "text" and "omitted" in b.get("text", "").lower()
|
|
for b in out[0].content
|
|
)
|
|
|
|
def test_strip_media_types_preserves_position(self):
|
|
# Stripped block is replaced IN PLACE; surrounding text/kept media keep
|
|
# their order (placeholder where the image was, file stays last).
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _strip_media_types
|
|
|
|
img = {"type": "image", "base64": "A", "mime_type": "image/png"}
|
|
f = {"type": "file", "base64": "F", "mime_type": "application/pdf"}
|
|
msg = HumanMessage(
|
|
content=[
|
|
{"type": "text", "text": "t1"},
|
|
img,
|
|
{"type": "text", "text": "t2"},
|
|
f,
|
|
]
|
|
)
|
|
out = _strip_media_types([msg], {"image"}) # block only image
|
|
content = out[0].content
|
|
assert all(b.get("type") != "image" for b in content) # image gone
|
|
# order preserved: t1, placeholder (where image was), t2, file
|
|
assert content[0]["text"] == "t1"
|
|
assert content[1]["type"] == "text"
|
|
assert "omitted" in content[1]["text"].lower()
|
|
assert content[2]["text"] == "t2"
|
|
assert content[3]["type"] == "file" # file kept at its original position
|
|
|
|
def test_strip_media_types_dedups_consecutive(self):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _strip_media_types
|
|
|
|
a = {"type": "image", "base64": "A", "mime_type": "image/png"}
|
|
b = {"type": "image", "base64": "B", "mime_type": "image/png"}
|
|
msg = HumanMessage(content=[a, b])
|
|
out = _strip_media_types([msg], {"image"})
|
|
# two adjacent stripped blocks collapse into ONE placeholder
|
|
assert len(out[0].content) == 1
|
|
assert "omitted" in out[0].content[0]["text"].lower()
|
|
|
|
def test_profile_no_vision_strips_upfront(self):
|
|
# Proactive: profile says image_inputs is False -> strip from the start,
|
|
# no failing first request.
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
model.profile = {"image_inputs": False}
|
|
calls = []
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
return "ok"
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
assert model._generate([self._img_tool()]) == "ok"
|
|
assert len(calls) == 1 # no failed attempt
|
|
assert all(isinstance(m.content, str) for m in calls[0])
|
|
assert any("omitted" in m.content.lower() for m in calls[0])
|
|
|
|
def test_profile_with_vision_does_not_strip(self):
|
|
# Profile says image_inputs is True -> normal preserve path (no upfront strip).
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
model.profile = {"image_inputs": True}
|
|
calls = []
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
return "ok"
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
assert model._generate([self._img_tool()]) == "ok"
|
|
# Image preserved (hoisted), not replaced by a placeholder.
|
|
assert any(
|
|
isinstance(m.content, list)
|
|
and any(b.get("type") == "image" for b in m.content)
|
|
for m in calls[0]
|
|
)
|
|
|
|
def test_generate_falls_back_and_caches(self):
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
calls = []
|
|
state = {"raised": False}
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
if not state["raised"]: # fail exactly once, ever
|
|
state["raised"] = True
|
|
raise Exception("unknown variant `image_url`, expected `text`")
|
|
return "ok"
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
tm = self._img_tool()
|
|
# 1st turn: preserve attempt fails once -> strip -> ok
|
|
assert model._generate([tm]) == "ok"
|
|
assert len(calls) == 2
|
|
retry = calls[1]
|
|
assert all(isinstance(m.content, str) for m in retry)
|
|
assert any("omitted" in m.content.lower() for m in retry)
|
|
|
|
# 2nd turn: cached no-vision -> straight to stripped, single call (no failure)
|
|
calls.clear()
|
|
assert model._generate([tm]) == "ok"
|
|
assert len(calls) == 1
|
|
assert all(isinstance(m.content, str) for m in calls[0])
|
|
|
|
def test_non_image_error_not_retried(self):
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
calls = []
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
raise Exception("rate limit exceeded")
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
with pytest.raises(Exception, match="rate limit"):
|
|
model._generate([self._img_tool()])
|
|
assert len(calls) == 1
|
|
|
|
def test_stream_falls_back(self):
|
|
from unittest.mock import MagicMock
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
model._generate = MagicMock(return_value="g")
|
|
calls = []
|
|
|
|
def _stream(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
if len(calls) == 1:
|
|
raise Exception("No endpoints found that support image input")
|
|
yield from ["x", "y"]
|
|
|
|
model._stream = _stream
|
|
_patch_openai_compat_content(model)
|
|
|
|
out = list(model._stream([self._img_tool()]))
|
|
assert out == ["x", "y"]
|
|
assert len(calls) == 2
|
|
|
|
async def test_astream_falls_back(self):
|
|
from unittest.mock import MagicMock
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
model._generate = MagicMock(return_value="g")
|
|
calls = []
|
|
|
|
async def _astream(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
if len(calls) == 1:
|
|
raise Exception("No endpoints found that support image input")
|
|
for c in ["x", "y"]:
|
|
yield c
|
|
|
|
model._astream = _astream
|
|
_patch_openai_compat_content(model)
|
|
|
|
out = [c async for c in model._astream([self._img_tool()])]
|
|
assert out == ["x", "y"]
|
|
assert len(calls) == 2
|
|
|
|
def test_unrelated_400_retry_fails_not_cached(self):
|
|
# A non-media 400 (e.g. tool schema) whose stripped retry ALSO fails must
|
|
# surface the original error and must NOT permanently flip to no-media.
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
class _E(Exception):
|
|
status_code = 400
|
|
|
|
model = self._make_model()
|
|
calls = []
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
raise _E("invalid tool schema") # 400, not media; fails every time
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
tm = self._img_tool()
|
|
with pytest.raises(_E):
|
|
model._generate([tm])
|
|
assert len(calls) == 2 # preserve attempt + stripped retry (both fail)
|
|
|
|
# Not cached: the next call attempts preserve again (not straight-to-stripped)
|
|
calls.clear()
|
|
with pytest.raises(_E):
|
|
model._generate([tm])
|
|
assert len(calls) == 2
|
|
|
|
def test_pdf_rejection_does_not_disable_images(self):
|
|
# Per-modality: a PDF/file rejection caches only file types; a later
|
|
# image must still be preserved (not stripped).
|
|
from langchain_core.messages import HumanMessage, ToolMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
calls = []
|
|
state = {"raised": False}
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
has_file = any(
|
|
isinstance(m.content, list)
|
|
and any(
|
|
isinstance(b, dict) and b.get("type") == "file" for b in m.content
|
|
)
|
|
for m in msgs
|
|
)
|
|
if has_file and not state["raised"]:
|
|
state["raised"] = True
|
|
raise Exception("file input is not supported")
|
|
return "ok"
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
pdf_tm = ToolMessage(
|
|
content_blocks=[
|
|
{"type": "file", "base64": "F", "mime_type": "application/pdf"}
|
|
],
|
|
tool_call_id="t1",
|
|
name="read_file",
|
|
)
|
|
assert model._generate([pdf_tm]) == "ok" # file rejected -> stripped -> ok
|
|
|
|
# Now an image: must still be preserved (images not blocked by a PDF reject)
|
|
calls.clear()
|
|
img_msg = HumanMessage(
|
|
content=[{"type": "image", "base64": "A", "mime_type": "image/png"}]
|
|
)
|
|
assert model._generate([img_msg]) == "ok"
|
|
assert len(calls) == 1 # single attempt, no failure
|
|
assert any(
|
|
isinstance(m.content, list)
|
|
and any(isinstance(b, dict) and b.get("type") == "image" for b in m.content)
|
|
for m in calls[0]
|
|
)
|
|
|
|
def test_bare_400_recovers_but_not_cached(self):
|
|
# A bare 400 with NO media marker recovers this request (stripped retry)
|
|
# but must NOT cache (no permanent degradation) — High #1.
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
class _E(Exception):
|
|
status_code = 400
|
|
|
|
model = self._make_model()
|
|
calls = []
|
|
state = {"raised": False}
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
if not state["raised"]:
|
|
state["raised"] = True
|
|
raise _E("transient bad request") # 400, no media marker
|
|
return "ok"
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
tm = self._img_tool()
|
|
assert model._generate([tm]) == "ok" # bare 400 -> stripped retry -> ok
|
|
assert len(calls) == 2
|
|
|
|
# NOT cached: the next call still attempts preserve (image kept, not stripped)
|
|
calls.clear()
|
|
assert model._generate([tm]) == "ok"
|
|
assert len(calls) == 1
|
|
assert any(
|
|
isinstance(m.content, list)
|
|
and any(isinstance(b, dict) and b.get("type") == "image" for b in m.content)
|
|
for m in calls[0]
|
|
)
|
|
|
|
def test_mixed_modality_caches_only_culprit(self):
|
|
# image+file message; provider rejects only the file -> cache file only,
|
|
# images stay preserved on later turns — High #2.
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
calls = []
|
|
state = {"raised": False}
|
|
|
|
def _gen(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
if not state["raised"]:
|
|
state["raised"] = True
|
|
raise Exception("file input is not supported")
|
|
return "ok"
|
|
|
|
model._generate = _gen
|
|
_patch_openai_compat_content(model)
|
|
|
|
mixed = HumanMessage(
|
|
content=[
|
|
{"type": "image", "base64": "A", "mime_type": "image/png"},
|
|
{"type": "file", "base64": "F", "mime_type": "application/pdf"},
|
|
]
|
|
)
|
|
assert model._generate([mixed]) == "ok" # file rejected -> retry -> cache file
|
|
|
|
# later image-only request: image must still be preserved
|
|
calls.clear()
|
|
img = HumanMessage(
|
|
content=[{"type": "image", "base64": "A", "mime_type": "image/png"}]
|
|
)
|
|
assert model._generate([img]) == "ok"
|
|
assert len(calls) == 1
|
|
assert any(
|
|
isinstance(m.content, list)
|
|
and any(isinstance(b, dict) and b.get("type") == "image" for b in m.content)
|
|
for m in calls[0]
|
|
)
|
|
|
|
def test_stream_empty_retry_raises_original(self):
|
|
# If the stripped streaming retry yields ZERO chunks, surface the
|
|
# original error instead of silently returning an empty stream.
|
|
from unittest.mock import MagicMock
|
|
|
|
from EvoScientist.llm.patches import _patch_openai_compat_content
|
|
|
|
model = self._make_model()
|
|
model._generate = MagicMock(return_value="g")
|
|
calls = []
|
|
|
|
def _stream(msgs, *a, **k):
|
|
calls.append(msgs)
|
|
if len(calls) == 1:
|
|
raise Exception("No endpoints found that support image input")
|
|
return # retry yields nothing
|
|
yield # pragma: no cover (makes this a generator)
|
|
|
|
model._stream = _stream
|
|
_patch_openai_compat_content(model)
|
|
|
|
with pytest.raises(Exception, match="support image"):
|
|
list(model._stream([self._img_tool()]))
|
|
assert len(calls) == 2
|
|
|
|
|
|
# =============================================================================
|
|
# Test DeepSeek model integration
|
|
# =============================================================================
|
|
|
|
|
|
def test_deepseek_model_strips_unsupported_tool_media(monkeypatch):
|
|
import json
|
|
|
|
import httpx
|
|
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
|
|
|
|
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
|
|
captured = {}
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
captured.update(json.loads(request.content))
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "chatcmpl-1",
|
|
"object": "chat.completion",
|
|
"created": 1,
|
|
"model": "deepseek-v4-flash",
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"finish_reason": "stop",
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 1,
|
|
"completion_tokens": 1,
|
|
"total_tokens": 2,
|
|
},
|
|
},
|
|
)
|
|
|
|
with httpx.Client(transport=httpx.MockTransport(respond)) as client:
|
|
model = get_chat_model(
|
|
"deepseek-v4-flash",
|
|
provider="deepseek",
|
|
http_client=client,
|
|
)
|
|
model.invoke(
|
|
[
|
|
HumanMessage("inspect the file"),
|
|
AIMessage(
|
|
"",
|
|
tool_calls=[{"name": "read_file", "args": {}, "id": "call_1"}],
|
|
),
|
|
ToolMessage(
|
|
content_blocks=[
|
|
{"type": "image", "base64": "AAA", "mime_type": "image/png"}
|
|
],
|
|
tool_call_id="call_1",
|
|
),
|
|
]
|
|
)
|
|
|
|
assert captured["messages"][2]["content"] == (
|
|
"[attachment omitted: this model does not support this input type]"
|
|
)
|
|
|
|
|
|
class TestDeepseekReasoningPassback:
|
|
"""Verify reasoning_content is retained in serialized DeepSeek history."""
|
|
|
|
def test_request_payload_preserves_reasoning_for_tool_history(self, monkeypatch):
|
|
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
|
|
|
|
from EvoScientist.llm.deepseek import EvoChatDeepSeek
|
|
|
|
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
|
|
model = EvoChatDeepSeek(model="deepseek-v4-flash")
|
|
messages = [
|
|
HumanMessage("q1"),
|
|
AIMessage(
|
|
"",
|
|
additional_kwargs={"reasoning_content": "rc1"},
|
|
tool_calls=[{"name": "read_file", "args": {}, "id": "call_1"}],
|
|
),
|
|
ToolMessage("result", tool_call_id="call_1"),
|
|
HumanMessage("q2"),
|
|
AIMessage("a2"),
|
|
HumanMessage("q3"),
|
|
AIMessage("a3", additional_kwargs={"reasoning_content": "rc3"}),
|
|
HumanMessage("q4"),
|
|
]
|
|
payload = model._get_request_payload(messages)
|
|
|
|
assert payload["messages"][1]["reasoning_content"] == "rc1"
|
|
assert "tool_calls" in payload["messages"][1]
|
|
assert "reasoning_content" not in payload["messages"][2]
|
|
assert payload["messages"][4]["reasoning_content"] == ""
|
|
assert payload["messages"][6]["reasoning_content"] == "rc3"
|
|
|
|
def test_thinking_disabled_copy_omits_reasoning_passback(self, monkeypatch):
|
|
from langchain_core.messages import AIMessage, HumanMessage
|
|
|
|
from EvoScientist.llm.deepseek import EvoChatDeepSeek
|
|
from EvoScientist.middleware.utils import disable_thinking
|
|
|
|
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
|
|
model = disable_thinking(EvoChatDeepSeek(model="deepseek-v4-flash"))
|
|
messages = [
|
|
HumanMessage("q1"),
|
|
AIMessage("a1", additional_kwargs={"reasoning_content": "rc1"}),
|
|
HumanMessage("q2"),
|
|
]
|
|
|
|
payload = model._get_request_payload(messages)
|
|
|
|
assert "reasoning_content" not in payload["messages"][1]
|
|
|
|
|
|
# =============================================================================
|
|
# Test _patch_openai_capture_reasoning_content (module-level monkey-patch)
|
|
# =============================================================================
|
|
|
|
|
|
class TestPatchOpenAICaptureReasoningContent:
|
|
"""Verify reasoning_content is captured into AIMessage.additional_kwargs.
|
|
|
|
This patch is applied at import time and globally affects langchain-openai's
|
|
_convert_dict_to_message and _convert_delta_to_message_chunk.
|
|
"""
|
|
|
|
def test_capture_from_non_streaming_response(self):
|
|
"""reasoning_content in OpenAI response dict → AIMessage.additional_kwargs."""
|
|
from langchain_openai.chat_models.base import _convert_dict_to_message
|
|
|
|
msg = _convert_dict_to_message(
|
|
{
|
|
"role": "assistant",
|
|
"content": "hi",
|
|
"reasoning_content": "let me think...",
|
|
}
|
|
)
|
|
assert msg.additional_kwargs.get("reasoning_content") == "let me think..."
|
|
|
|
def test_capture_absent_when_field_missing(self):
|
|
"""No reasoning_content in response → not added to additional_kwargs."""
|
|
from langchain_openai.chat_models.base import _convert_dict_to_message
|
|
|
|
msg = _convert_dict_to_message({"role": "assistant", "content": "hi"})
|
|
assert "reasoning_content" not in msg.additional_kwargs
|
|
|
|
def test_capture_from_streaming_chunk(self):
|
|
"""reasoning_content delta is captured onto the chunk's additional_kwargs."""
|
|
from langchain_core.messages import AIMessageChunk
|
|
from langchain_openai.chat_models.base import (
|
|
_convert_delta_to_message_chunk,
|
|
)
|
|
|
|
chunk = _convert_delta_to_message_chunk(
|
|
{"role": "assistant", "content": "", "reasoning_content": "thinking"},
|
|
AIMessageChunk,
|
|
)
|
|
assert chunk.additional_kwargs.get("reasoning_content") == "thinking"
|
|
|
|
def test_capture_does_not_affect_other_fields(self):
|
|
"""Existing tool_calls / function_call extraction unaffected."""
|
|
from langchain_openai.chat_models.base import _convert_dict_to_message
|
|
|
|
msg = _convert_dict_to_message(
|
|
{
|
|
"role": "assistant",
|
|
"content": "calling tool",
|
|
"tool_calls": [
|
|
{
|
|
"id": "call_1",
|
|
"type": "function",
|
|
"function": {"name": "get_weather", "arguments": "{}"},
|
|
}
|
|
],
|
|
"reasoning_content": "use the tool",
|
|
}
|
|
)
|
|
assert len(msg.tool_calls) == 1
|
|
assert msg.tool_calls[0]["name"] == "get_weather"
|
|
assert msg.additional_kwargs.get("reasoning_content") == "use the tool"
|
|
|
|
|
|
class TestIsResponsesReasoningItem:
|
|
"""_is_responses_reasoning_item flags encrypted OpenAI-Responses items."""
|
|
|
|
def test_rs_id_is_responses_item(self):
|
|
from EvoScientist.llm.patches import _is_responses_reasoning_item
|
|
|
|
assert _is_responses_reasoning_item({"id": "rs_09363d42", "type": "x"})
|
|
|
|
def test_encrypted_data_is_responses_item(self):
|
|
from EvoScientist.llm.patches import _is_responses_reasoning_item
|
|
|
|
assert _is_responses_reasoning_item({"data": "gAAAAAB...", "type": "x"})
|
|
|
|
def test_plain_text_reasoning_is_not_responses_item(self):
|
|
from EvoScientist.llm.patches import _is_responses_reasoning_item
|
|
|
|
assert not _is_responses_reasoning_item(
|
|
{"type": "reasoning.text", "text": "thinking", "index": 0}
|
|
)
|
|
assert not _is_responses_reasoning_item("not a dict")
|
|
|
|
|
|
class TestPatchOpenrouterStripResponsesReasoning:
|
|
"""OpenAI-Responses encrypted reasoning items (`rs_` id / encrypted data)
|
|
are stripped from outgoing OpenRouter assistant messages, preventing the
|
|
multi-turn "Item with id 'rs_...' not found" 400 (store=false; #37777).
|
|
"""
|
|
|
|
def _apply(self):
|
|
import langchain_openrouter.chat_models as mod
|
|
|
|
import EvoScientist.llm.patches as patches
|
|
|
|
orig = mod._convert_message_to_dict
|
|
orig_flag = patches._openrouter_reasoning_strip_patched
|
|
patches._openrouter_reasoning_strip_patched = False
|
|
patches._patch_openrouter_strip_responses_reasoning()
|
|
return patches, mod, orig, orig_flag
|
|
|
|
@staticmethod
|
|
def _restore(patches, mod, orig, orig_flag):
|
|
mod._convert_message_to_dict = orig
|
|
patches._openrouter_reasoning_strip_patched = orig_flag
|
|
|
|
def test_strips_encrypted_item_drops_key_when_empty(self):
|
|
from langchain_core.messages import AIMessage
|
|
|
|
patches, mod, orig, orig_flag = self._apply()
|
|
try:
|
|
msg = AIMessage(
|
|
content="done",
|
|
additional_kwargs={
|
|
"reasoning_details": [
|
|
{
|
|
"type": "reasoning.summary",
|
|
"format": "openai-responses-v1",
|
|
"id": "rs_09363d42b054",
|
|
"data": "gAAAAAB...",
|
|
"summary": "real reasoning text",
|
|
"index": 0,
|
|
}
|
|
],
|
|
},
|
|
)
|
|
result = mod._convert_message_to_dict(msg)
|
|
# sole entry was an rs_ item → reasoning_details removed entirely.
|
|
assert "reasoning_details" not in result
|
|
finally:
|
|
self._restore(patches, mod, orig, orig_flag)
|
|
|
|
def test_keeps_plain_text_reasoning(self):
|
|
from langchain_core.messages import AIMessage
|
|
|
|
patches, mod, orig, orig_flag = self._apply()
|
|
try:
|
|
msg = AIMessage(
|
|
content="done",
|
|
additional_kwargs={
|
|
"reasoning_details": [
|
|
{"type": "reasoning.text", "text": "thinking", "index": 0},
|
|
{"id": "rs_abc", "data": "blob", "index": 1},
|
|
],
|
|
},
|
|
)
|
|
result = mod._convert_message_to_dict(msg)
|
|
kept = result["reasoning_details"]
|
|
assert len(kept) == 1
|
|
assert kept[0]["type"] == "reasoning.text"
|
|
finally:
|
|
self._restore(patches, mod, orig, orig_flag)
|
|
|
|
def test_does_not_mutate_original_message(self):
|
|
from langchain_core.messages import AIMessage
|
|
|
|
patches, mod, orig, orig_flag = self._apply()
|
|
try:
|
|
details = [{"id": "rs_abc", "data": "blob"}]
|
|
msg = AIMessage(
|
|
content="x", additional_kwargs={"reasoning_details": details}
|
|
)
|
|
mod._convert_message_to_dict(msg)
|
|
# stored history untouched — we filter a fresh list, not in place.
|
|
assert details == [{"id": "rs_abc", "data": "blob"}]
|
|
finally:
|
|
self._restore(patches, mod, orig, orig_flag)
|
|
|
|
def test_patch_is_idempotent(self):
|
|
patches, mod, orig, orig_flag = self._apply()
|
|
try:
|
|
wrapper = mod._convert_message_to_dict
|
|
# Second call is guarded by the flag → must not re-wrap.
|
|
patches._patch_openrouter_strip_responses_reasoning()
|
|
assert mod._convert_message_to_dict is wrapper
|
|
finally:
|
|
self._restore(patches, mod, orig, orig_flag)
|
|
|
|
def test_non_dict_entry_is_kept(self):
|
|
from langchain_core.messages import AIMessage
|
|
|
|
patches, mod, orig, orig_flag = self._apply()
|
|
try:
|
|
msg = AIMessage(
|
|
content="done",
|
|
additional_kwargs={
|
|
"reasoning_details": [
|
|
"opaque", # non-dict slipped in → kept, not crashed on
|
|
{"id": "rs_abc", "data": "blob", "index": 1},
|
|
],
|
|
},
|
|
)
|
|
result = mod._convert_message_to_dict(msg)
|
|
assert result["reasoning_details"] == ["opaque"]
|
|
finally:
|
|
self._restore(patches, mod, orig, orig_flag)
|
|
|
|
|
|
# =============================================================================
|
|
# Test _patch_anthropic_strip_foreign_reasoning
|
|
# =============================================================================
|
|
|
|
|
|
class TestAnthropicStripForeignReasoning:
|
|
def test_strip_removes_reasoning_content_blocks(self):
|
|
"""reasoning_content blocks are dropped; text and thinking survive."""
|
|
from langchain_core.messages import AIMessage, HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
|
|
|
|
messages = [
|
|
HumanMessage("hello"),
|
|
AIMessage(
|
|
content=[
|
|
{"type": "reasoning_content", "reasoning_content": {"text": "hm"}},
|
|
{"type": "thinking", "thinking": "hm", "signature": ""},
|
|
{"type": "text", "text": "hi"},
|
|
]
|
|
),
|
|
]
|
|
|
|
result = _normalize_anthropic_replay_messages(messages)
|
|
|
|
types = [b["type"] for b in result[1].content]
|
|
assert types == ["thinking", "text"]
|
|
|
|
def test_missing_thinking_signature_defaulted(self):
|
|
"""Streamed thinking blocks without a signature key get signature ''."""
|
|
from langchain_core.messages import AIMessage
|
|
|
|
from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
|
|
|
|
messages = [
|
|
AIMessage(
|
|
content=[
|
|
{"type": "thinking", "thinking": "hm", "index": 0},
|
|
{"type": "text", "text": "hi", "index": 1},
|
|
]
|
|
),
|
|
]
|
|
|
|
result = _normalize_anthropic_replay_messages(messages)
|
|
|
|
assert result[0].content[0]["signature"] == ""
|
|
assert "signature" not in result[0].content[1]
|
|
|
|
def test_strip_no_change_returns_same_object(self):
|
|
"""Clean histories pass through without copying."""
|
|
from langchain_core.messages import AIMessage, HumanMessage
|
|
|
|
from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
|
|
|
|
messages = [
|
|
HumanMessage("hello"),
|
|
AIMessage(content=[{"type": "text", "text": "hi"}]),
|
|
AIMessage(content="plain string content"),
|
|
]
|
|
|
|
assert _normalize_anthropic_replay_messages(messages) is messages
|
|
|
|
def test_kimi_k3_exempt_from_flatten_patch(self, monkeypatch):
|
|
"""K3 on custom-anthropic gets no instance flatten closures; others do."""
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
|
|
|
|
kimi = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
|
|
assert "_generate" not in vars(kimi)
|
|
|
|
other = get_chat_model(
|
|
"claude-sonnet-4-6", provider="custom-anthropic", max_tokens=1024
|
|
)
|
|
assert "_generate" in vars(other)
|
|
|
|
def test_reasoning_content_stripped_on_the_wire(self, monkeypatch):
|
|
"""End-to-end: foreign reasoning blocks never reach the Anthropic wire."""
|
|
import json
|
|
|
|
import anthropic
|
|
import httpx
|
|
from langchain_core.messages import AIMessage, HumanMessage
|
|
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
|
|
model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
|
|
|
|
captured: dict = {}
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
captured.update(json.loads(request.content.decode()))
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "msg_test",
|
|
"type": "message",
|
|
"role": "assistant",
|
|
"content": [{"type": "text", "text": "ok"}],
|
|
"model": "moonshotai/kimi-k3",
|
|
"stop_reason": "end_turn",
|
|
"stop_sequence": None,
|
|
"usage": {"input_tokens": 1, "output_tokens": 1},
|
|
},
|
|
)
|
|
|
|
model._client = anthropic.Anthropic(
|
|
api_key="test-key",
|
|
base_url="https://compat.example.com",
|
|
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
|
|
)
|
|
|
|
history = [
|
|
HumanMessage("hello"),
|
|
AIMessage(
|
|
content=[
|
|
{"type": "reasoning_content", "reasoning_content": {"text": "hm"}},
|
|
{"type": "thinking", "thinking": "hm", "index": 0},
|
|
{"type": "text", "text": "hi there"},
|
|
]
|
|
),
|
|
HumanMessage("say ok"),
|
|
]
|
|
result = model.invoke(history)
|
|
|
|
sent_blocks = [
|
|
block
|
|
for message in captured["messages"]
|
|
for block in (
|
|
message["content"] if isinstance(message["content"], list) else []
|
|
)
|
|
]
|
|
sent_types = [block["type"] for block in sent_blocks]
|
|
assert "reasoning_content" not in sent_types
|
|
assert "text" in sent_types
|
|
thinking_blocks = [b for b in sent_blocks if b["type"] == "thinking"]
|
|
assert thinking_blocks
|
|
assert thinking_blocks[0]["signature"] == ""
|
|
assert result.content == "ok"
|
|
|
|
|
|
# =============================================================================
|
|
# Test _patch_anthropic_structured_output
|
|
# =============================================================================
|
|
|
|
|
|
class TestAnthropicStructuredOutput:
|
|
@staticmethod
|
|
def _capture_structured_request(model, response_text):
|
|
"""Invoke a structured-output runnable against a capturing transport."""
|
|
import json
|
|
|
|
import anthropic
|
|
import httpx
|
|
from pydantic import BaseModel
|
|
|
|
class Pick(BaseModel):
|
|
answer: str
|
|
|
|
captured: dict = {}
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
captured.update(json.loads(request.content.decode()))
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "msg_test",
|
|
"type": "message",
|
|
"role": "assistant",
|
|
"content": response_text,
|
|
"model": "test",
|
|
"stop_reason": "end_turn",
|
|
"stop_sequence": None,
|
|
"usage": {"input_tokens": 1, "output_tokens": 1},
|
|
},
|
|
)
|
|
|
|
model._client = anthropic.Anthropic(
|
|
api_key="test-key",
|
|
base_url="https://compat.example.com",
|
|
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
|
|
)
|
|
result = model.with_structured_output(Pick).invoke("Reply with answer='ok'")
|
|
return captured, result
|
|
|
|
def test_kimi_k3_defaults_to_json_schema(self, monkeypatch):
|
|
"""K3 structured output binds output_config.format, no forced tool_choice."""
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
|
|
model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
|
|
|
|
captured, result = self._capture_structured_request(
|
|
model, [{"type": "text", "text": '{"answer": "ok"}'}]
|
|
)
|
|
|
|
assert captured["output_config"]["format"]["type"] == "json_schema"
|
|
assert "tool_choice" not in captured
|
|
assert result.answer == "ok"
|
|
|
|
def test_claude_keeps_function_calling(self, monkeypatch):
|
|
"""Claude models keep tool-based structured output (no json_schema flip)."""
|
|
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key")
|
|
model = get_chat_model(
|
|
"claude-haiku-4-5", provider="anthropic", max_tokens=1024
|
|
)
|
|
|
|
captured, result = self._capture_structured_request(
|
|
model,
|
|
[
|
|
{
|
|
"type": "tool_use",
|
|
"id": "toolu_1",
|
|
"name": "Pick",
|
|
"input": {"answer": "ok"},
|
|
}
|
|
],
|
|
)
|
|
|
|
assert "output_config" not in captured
|
|
assert [t["name"] for t in captured["tools"]] == ["Pick"]
|
|
assert result.answer == "ok"
|
|
|
|
|
|
# =============================================================================
|
|
# Test _apply_auto_config
|
|
# =============================================================================
|
|
|
|
|
|
class TestAutoConfig:
|
|
@pytest.fixture(autouse=True)
|
|
def _clear_reasoning_effort_env(self, monkeypatch):
|
|
"""Keep auto-config tests independent of the developer environment."""
|
|
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_4_5_thinking(self, mock_init, monkeypatch):
|
|
"""Anthropic 4-5 models get enabled thinking with budget."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
|
|
|
|
get_chat_model("claude-haiku-4-5")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_4_6_adaptive_thinking(self, mock_init, monkeypatch):
|
|
"""Anthropic 4-6 models get adaptive thinking with max effort."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
|
|
|
|
get_chat_model("claude-sonnet-4-6")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
|
|
assert call_kwargs["effort"] == "max"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_4_8_adaptive_thinking(self, mock_init, monkeypatch):
|
|
"""Anthropic 4-8 models get adaptive thinking with max effort."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
|
|
|
|
get_chat_model("claude-opus-4-8")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
|
|
assert call_kwargs["effort"] == "max"
|
|
|
|
@pytest.mark.parametrize("model", ["claude-opus-5", "claude-sonnet-5"])
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_5_series_adaptive_thinking(self, mock_init, model, monkeypatch):
|
|
"""Anthropic 5-series models get adaptive thinking (budget_tokens would 400)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
|
|
|
|
get_chat_model(model, provider="anthropic")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
|
|
assert call_kwargs["effort"] == "max"
|
|
|
|
@pytest.mark.parametrize("model", ["moonshotai/kimi-k3", "kimi-k3"])
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_custom_anthropic_kimi_k3_declares_thinking(
|
|
self, mock_init, model, monkeypatch
|
|
):
|
|
"""K3 via custom-anthropic declares thinking (else forced tool_choice 400s)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
|
|
|
|
get_chat_model(model, provider="custom-anthropic")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
|
|
assert call_kwargs["max_tokens"] == 16000
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_kimi_coding_declares_thinking(self, mock_init):
|
|
"""Kimi For Coding plan models declare thinking on the kimi-coding provider."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("kimi-for-coding", provider="kimi-coding")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
|
|
assert call_kwargs["max_tokens"] == 16000
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_custom_anthropic_non_kimi_no_thinking(self, mock_init, monkeypatch):
|
|
"""Non-Kimi models on custom-anthropic still skip thinking injection."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
|
|
monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
|
|
|
|
get_chat_model("glm-4.7", provider="custom-anthropic")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "thinking" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_4_6_proxy_no_thinking(self, mock_init, monkeypatch):
|
|
"""Anthropic 4-6 models via proxy skip thinking (history round-trip 422)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("ANTHROPIC_BASE_URL", "http://127.0.0.1:8000")
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "ccproxy-oauth")
|
|
|
|
get_chat_model("claude-sonnet-4-6")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "thinking" not in call_kwargs
|
|
assert "effort" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_4_5_proxy_no_thinking(self, mock_init, monkeypatch):
|
|
"""Anthropic 4-5 models via proxy also skip thinking (history round-trip 422)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("ANTHROPIC_BASE_URL", "http://127.0.0.1:8000")
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "ccproxy-oauth")
|
|
|
|
get_chat_model("claude-haiku-4-5")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "thinking" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_4_6_no_proxy_no_downgrade(self, mock_init, monkeypatch):
|
|
"""Anthropic 4-6 models without proxy still get adaptive thinking."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("ANTHROPIC_BASE_URL", "https://api.anthropic.com")
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-real")
|
|
|
|
get_chat_model("claude-sonnet-4-6")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
|
|
assert call_kwargs["effort"] == "max"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_anthropic_thinking_not_overridden(self, mock_init):
|
|
"""User-supplied thinking config should not be overridden."""
|
|
mock_init.return_value = "mock_model"
|
|
custom_thinking = {"type": "enabled", "budget_tokens": 500}
|
|
|
|
get_chat_model("claude-sonnet-4-6", thinking=custom_thinking)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["thinking"] == custom_thinking
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_reasoning_xhigh(self, mock_init, monkeypatch):
|
|
"""gpt-5.4+ and codex models get xhigh reasoning."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
|
|
|
|
get_chat_model("gpt-5.4", provider="openai")
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "xhigh",
|
|
"summary": "auto",
|
|
}
|
|
|
|
get_chat_model("gpt-5.3-codex", provider="openai")
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "xhigh",
|
|
"summary": "auto",
|
|
}
|
|
|
|
get_chat_model("gpt-5.5", provider="openai")
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "xhigh",
|
|
"summary": "auto",
|
|
}
|
|
|
|
get_chat_model("gpt-5.6-sol", provider="openai")
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "xhigh",
|
|
"summary": "auto",
|
|
}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_reasoning_effort_from_env(self, mock_init, monkeypatch):
|
|
"""Native OpenAI reasoning effort should be configurable via env var."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "high")
|
|
|
|
get_chat_model("gpt-5.5", provider="openai")
|
|
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "high",
|
|
"summary": "auto",
|
|
}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_reasoning_high_fallback(self, mock_init, monkeypatch):
|
|
"""Other OpenAI models get high reasoning effort."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
|
|
get_chat_model("gpt-5-nano")
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "high",
|
|
"summary": "auto",
|
|
}
|
|
|
|
get_chat_model("gpt-5.2", provider="openai")
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "high",
|
|
"summary": "auto",
|
|
}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_base_url_override(self, mock_init, monkeypatch):
|
|
"""OpenAI provider should support base_url override (e.g. ccproxy Codex)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert call_kwargs["base_url"] == "http://127.0.0.1:8000/codex/v1"
|
|
assert call_kwargs["api_key"] == "ccproxy-oauth"
|
|
# ccproxy uses the Responses API, so reasoning configuration is valid.
|
|
assert call_kwargs["reasoning"] == {
|
|
"effort": "high",
|
|
"summary": "auto",
|
|
"context": "all_turns",
|
|
}
|
|
# Proxy mode: Responses API (bypasses format chain), streaming ON
|
|
assert call_kwargs["use_responses_api"] is True
|
|
assert "streaming" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_localhost_non_ccproxy_not_downgraded(self, mock_init, monkeypatch):
|
|
"""Local OpenAI-compatible endpoints (vLLM, etc.) are not affected by ccproxy workarounds."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8080/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "sk-local-key")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["base_url"] == "http://127.0.0.1:8080/v1"
|
|
# NOT ccproxy: reasoning should be applied, no forced Chat Completions
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
assert "use_responses_api" not in call_kwargs
|
|
assert "streaming" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_codex_path_but_wrong_key_not_ccproxy(self, mock_init, monkeypatch):
|
|
"""ccproxy detection requires both /codex/ path AND ccproxy-oauth key."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "sk-real-key")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
assert "use_responses_api" not in call_kwargs
|
|
assert "default_headers" not in call_kwargs
|
|
|
|
@patch(
|
|
"EvoScientist.llm.models._installed_codex_client_version",
|
|
return_value="0.144.1",
|
|
)
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_client_headers(
|
|
self, mock_init, mock_installed_version, monkeypatch
|
|
):
|
|
"""ccproxy Codex mode sends Codex-CLI-shaped client headers."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
monkeypatch.delenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", raising=False)
|
|
|
|
get_chat_model("gpt-5.5", provider="openai")
|
|
|
|
headers = mock_init.call_args[1]["default_headers"]
|
|
assert headers["originator"] == "codex_cli_rs"
|
|
assert headers["version"] == "0.144.1"
|
|
assert headers["User-Agent"].startswith("codex_cli_rs/0.144.1")
|
|
mock_installed_version.assert_called_once_with()
|
|
assert mock_init.call_args[1]["reasoning"]["effort"] == "xhigh"
|
|
assert mock_init.call_args[1]["reasoning"]["context"] == "all_turns"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_client_version_env(self, mock_init, monkeypatch):
|
|
"""EVOSCIENTIST_CODEX_CLIENT_VERSION overrides the pinned version."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
monkeypatch.setenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", "9.9.9")
|
|
|
|
get_chat_model("gpt-5.5", provider="openai")
|
|
|
|
headers = mock_init.call_args[1]["default_headers"]
|
|
assert headers["version"] == "9.9.9"
|
|
assert headers["User-Agent"].startswith("codex_cli_rs/9.9.9")
|
|
|
|
@patch("EvoScientist.llm.models.subprocess.run")
|
|
def test_installed_codex_client_version(self, mock_run):
|
|
"""The advertised version follows the installed Codex CLI."""
|
|
from EvoScientist.llm.models import _installed_codex_client_version
|
|
|
|
mock_run.return_value.returncode = 0
|
|
mock_run.return_value.stdout = "codex-cli 0.144.1\n"
|
|
mock_run.return_value.stderr = ""
|
|
_installed_codex_client_version.cache_clear()
|
|
try:
|
|
assert _installed_codex_client_version() == "0.144.1"
|
|
assert _installed_codex_client_version() == "0.144.1"
|
|
finally:
|
|
_installed_codex_client_version.cache_clear()
|
|
mock_run.assert_called_once_with(
|
|
["codex", "--version"],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=2,
|
|
check=False,
|
|
)
|
|
|
|
@patch(
|
|
"EvoScientist.llm.models._installed_codex_client_version",
|
|
return_value="0.140.0",
|
|
)
|
|
def test_older_installed_codex_uses_fallback(
|
|
self, mock_installed_version, monkeypatch
|
|
):
|
|
"""An outdated installed CLI must not undercut the safe fallback."""
|
|
from EvoScientist.llm.models import (
|
|
_CODEX_CLIENT_VERSION_FALLBACK,
|
|
_resolve_codex_client_version,
|
|
)
|
|
|
|
monkeypatch.delenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", raising=False)
|
|
|
|
assert _resolve_codex_client_version() == _CODEX_CLIENT_VERSION_FALLBACK
|
|
mock_installed_version.assert_called_once_with()
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_headers_respect_caller(self, mock_init, monkeypatch):
|
|
"""Caller-supplied default_headers keys are not overridden."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
|
|
get_chat_model(
|
|
"gpt-5.5",
|
|
provider="openai",
|
|
default_headers={"originator": "codex_vscode", "version": "9.9.9"},
|
|
)
|
|
|
|
headers = mock_init.call_args[1]["default_headers"]
|
|
assert headers["originator"] == "codex_vscode"
|
|
assert headers["version"] == "9.9.9"
|
|
assert headers["User-Agent"].startswith("codex_cli_rs/9.9.9")
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_reasoning_context_respects_caller(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Caller-supplied reasoning.context is not overridden."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
reasoning = {"effort": "low", "context": "previous_response_id"}
|
|
|
|
get_chat_model(
|
|
"gpt-5.5",
|
|
provider="openai",
|
|
reasoning=reasoning,
|
|
)
|
|
|
|
assert mock_init.call_args[1]["reasoning"] == {
|
|
"effort": "low",
|
|
"context": "previous_response_id",
|
|
}
|
|
assert mock_init.call_args[1]["reasoning"] is not reasoning
|
|
assert reasoning == {"effort": "low", "context": "previous_response_id"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_reasoning_context_requires_responses_api(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""Responses-only reasoning.context is not sent when Chat Completions is forced."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
monkeypatch.delenv("EVOSCIENTIST_USE_RESPONSES_API", raising=False)
|
|
|
|
get_chat_model(
|
|
"gpt-5.5",
|
|
provider="openai",
|
|
use_responses_api=False,
|
|
reasoning={"effort": "low"},
|
|
)
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["use_responses_api"] is False
|
|
assert call_kwargs["reasoning"] == {"effort": "low"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_env_false_drops_reasoning(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""The global Chat Completions override still removes reasoning entirely."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "false")
|
|
|
|
get_chat_model("gpt-5.5", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["use_responses_api"] is False
|
|
assert "reasoning" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_codex_none_headers(self, mock_init, monkeypatch):
|
|
"""An explicit default_headers=None is normalized before gap-filling."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/codex/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
monkeypatch.setenv("EVOSCIENTIST_CODEX_CLIENT_VERSION", "9.9.9")
|
|
|
|
get_chat_model(
|
|
"gpt-5.5",
|
|
provider="openai",
|
|
default_headers=None,
|
|
)
|
|
|
|
headers = mock_init.call_args[1]["default_headers"]
|
|
assert headers["originator"] == "codex_cli_rs"
|
|
assert headers["version"] == "9.9.9"
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_ccproxy_key_but_wrong_path_not_ccproxy(
|
|
self, mock_init, monkeypatch
|
|
):
|
|
"""ccproxy detection requires both /codex/ path AND ccproxy-oauth key."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "http://127.0.0.1:8000/v1")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "ccproxy-oauth")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
assert "use_responses_api" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_openai_no_base_url_when_unset(self, mock_init, monkeypatch):
|
|
"""OpenAI provider should not set base_url when env var is empty."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.setenv("OPENAI_API_KEY", "sk-real")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["model_provider"] == "openai"
|
|
assert "base_url" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_google_thoughts(self, mock_init):
|
|
"""Google GenAI models get include_thoughts=True by default."""
|
|
mock_init.return_value = "mock_model"
|
|
|
|
get_chat_model("gemini-2.5-flash")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["include_thoughts"] is True
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_use_responses_api_false(self, mock_init, monkeypatch):
|
|
"""use_responses_api=false forces Chat Completions and drops reasoning."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "false")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["use_responses_api"] is False
|
|
assert "reasoning" not in call_kwargs
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_use_responses_api_true(self, mock_init, monkeypatch):
|
|
"""use_responses_api=true explicitly enables the Responses API."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "true")
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["use_responses_api"] is True
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_use_responses_api_default_unchanged(self, mock_init, monkeypatch):
|
|
"""Empty use_responses_api preserves default behavior (no kwarg set)."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("EVOSCIENTIST_USE_RESPONSES_API", raising=False)
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert "use_responses_api" not in call_kwargs
|
|
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
|
|
|
def test_responses_api_env_ignored_for_host_routed_deepseek(self, monkeypatch):
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.deepseek import EvoChatDeepSeek
|
|
|
|
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "sk-test")
|
|
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://api.deepseek.com")
|
|
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", "true")
|
|
|
|
model = get_chat_model("deepseek-chat", provider="custom-openai")
|
|
|
|
assert isinstance(model, EvoChatDeepSeek)
|
|
assert model.use_responses_api is not True
|
|
assert "messages" in model._get_request_payload([HumanMessage("hi")])
|
|
|
|
@pytest.mark.parametrize("provider", ["deepseek", "custom-openai"])
|
|
def test_deepseek_rejects_explicit_responses_api(self, monkeypatch, provider):
|
|
if provider == "deepseek":
|
|
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-test")
|
|
else:
|
|
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "sk-test")
|
|
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://api.deepseek.com")
|
|
|
|
with pytest.raises(ValueError, match="does not support the OpenAI Responses"):
|
|
get_chat_model(
|
|
"deepseek-chat",
|
|
provider=provider,
|
|
use_responses_api=True,
|
|
)
|
|
|
|
@pytest.mark.parametrize("env_value", ["FALSE", " false ", "False"])
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_use_responses_api_false_normalization(
|
|
self, mock_init, monkeypatch, env_value
|
|
):
|
|
"""Case/whitespace variants of 'false' are normalized correctly."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", env_value)
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["use_responses_api"] is False
|
|
assert "reasoning" not in call_kwargs
|
|
|
|
@pytest.mark.parametrize("env_value", ["TRUE", " true ", "True"])
|
|
@patch("EvoScientist.llm.models.init_chat_model")
|
|
def test_use_responses_api_true_normalization(
|
|
self, mock_init, monkeypatch, env_value
|
|
):
|
|
"""Case/whitespace variants of 'true' are normalized correctly."""
|
|
mock_init.return_value = "mock_model"
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.setenv("EVOSCIENTIST_USE_RESPONSES_API", env_value)
|
|
|
|
get_chat_model("gpt-5-nano", provider="openai")
|
|
|
|
call_kwargs = mock_init.call_args[1]
|
|
assert call_kwargs["use_responses_api"] is True
|
|
|
|
|
|
# =============================================================================
|
|
# Test validate_requesty_key
|
|
# =============================================================================
|
|
|
|
|
|
class TestValidateRequestyKey:
|
|
"""The Requesty key validator probes the router's auth layer.
|
|
|
|
Validation targets a deliberately nonexistent sentinel model
|
|
(``requesty/auth-preflight``) so key checks don't depend on any real
|
|
model staying available: the router resolves auth *before* the model,
|
|
so a valid key returns 404 (model-not-found) while a bad key returns
|
|
401/403. All HTTP calls are mocked — no network in unit tests.
|
|
"""
|
|
|
|
def test_empty_key_skipped(self):
|
|
"""No key provided is skipped, not an error."""
|
|
from EvoScientist.config.onboard.validators import validate_requesty_key
|
|
|
|
is_valid, msg = validate_requesty_key("")
|
|
assert is_valid is True
|
|
assert "Skipped" in msg
|
|
|
|
def test_uses_sentinel_model_not_a_real_one(self):
|
|
"""The probe targets a nonexistent sentinel model, not a real model."""
|
|
from EvoScientist.config.onboard.validators import validate_requesty_key
|
|
|
|
with patch("httpx.post") as mock_post:
|
|
mock_post.return_value.status_code = 404
|
|
validate_requesty_key("rq-key")
|
|
|
|
payload = mock_post.call_args.kwargs["json"]
|
|
assert payload["model"] == "requesty/auth-preflight"
|
|
assert payload["max_tokens"] == 1
|
|
|
|
def test_model_not_found_means_auth_passed(self):
|
|
"""404 (model not found) means auth was accepted → key is valid."""
|
|
from EvoScientist.config.onboard.validators import validate_requesty_key
|
|
|
|
with patch("httpx.post") as mock_post:
|
|
mock_post.return_value.status_code = 404
|
|
is_valid, msg = validate_requesty_key("rq-key")
|
|
|
|
assert is_valid is True
|
|
assert msg == "Valid"
|
|
|
|
def test_success_means_valid(self):
|
|
"""200 (accepted) also means the key is valid."""
|
|
from EvoScientist.config.onboard.validators import validate_requesty_key
|
|
|
|
with patch("httpx.post") as mock_post:
|
|
mock_post.return_value.status_code = 200
|
|
is_valid, msg = validate_requesty_key("rq-key")
|
|
|
|
assert is_valid is True
|
|
assert msg == "Valid"
|
|
|
|
@pytest.mark.parametrize("status", [401, 403])
|
|
def test_auth_rejected_means_invalid(self, status):
|
|
"""401/403 mean the key was rejected by the auth layer."""
|
|
from EvoScientist.config.onboard.validators import validate_requesty_key
|
|
|
|
with patch("httpx.post") as mock_post:
|
|
mock_post.return_value.status_code = status
|
|
is_valid, msg = validate_requesty_key("bad-key")
|
|
|
|
assert is_valid is False
|
|
assert msg == "Invalid API key"
|
|
|
|
@pytest.mark.parametrize("status", [429, 500, 502, 503])
|
|
def test_transient_status_is_inconclusive(self, status):
|
|
"""Rate-limit / 5xx leave key validity unknown, not rejected."""
|
|
from EvoScientist.config.onboard.validators import validate_requesty_key
|
|
|
|
with patch("httpx.post") as mock_post:
|
|
mock_post.return_value.status_code = status
|
|
is_valid, msg = validate_requesty_key("rq-key")
|
|
|
|
assert is_valid is False
|
|
assert "inconclusive" in msg.lower()
|
|
assert str(status) in msg
|