diff --git a/.github/assets/badge-pypi-dark.svg b/.github/assets/badge-pypi-dark.svg
index c6c5f09..9d5db97 100644
--- a/.github/assets/badge-pypi-dark.svg
+++ b/.github/assets/badge-pypi-dark.svg
@@ -5,5 +5,5 @@
v0.2.3
+ font-size="13" font-weight="700" fill="#ffffff">v0.2.4
\ No newline at end of file
diff --git a/.github/assets/badge-pypi-light.svg b/.github/assets/badge-pypi-light.svg
index 0b141e2..49f7df1 100644
--- a/.github/assets/badge-pypi-light.svg
+++ b/.github/assets/badge-pypi-light.svg
@@ -5,5 +5,5 @@
v0.2.3
+ font-size="13" font-weight="700" fill="#ffffff">v0.2.4
\ No newline at end of file
diff --git a/.github/assets/wechat_group.jpeg b/.github/assets/wechat_group.jpeg
index 2f92b22..b41363c 100644
Binary files a/.github/assets/wechat_group.jpeg and b/.github/assets/wechat_group.jpeg differ
diff --git a/EvoScientist/llm/models.py b/EvoScientist/llm/models.py
index f7d7fbd..bdcd4f2 100644
--- a/EvoScientist/llm/models.py
+++ b/EvoScientist/llm/models.py
@@ -28,6 +28,8 @@ from .context_window import apply_known_context_window
from .deepseek import EvoChatDeepSeek
from .patches import (
_is_ccproxy_codex,
+ _patch_anthropic_strip_foreign_reasoning,
+ _patch_anthropic_structured_output,
_patch_ccproxy_system_to_developer,
_patch_openai_compat_content,
_patch_openrouter_strip_responses_reasoning,
@@ -145,6 +147,13 @@ _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset(
{"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"}
)
+
+def _is_mandatory_thinking_kimi(model_id: str) -> bool:
+ """True for Kimi models whose thinking cannot be disabled (K3 family)."""
+ short_id = model_id.split("/")[-1]
+ return short_id.startswith("kimi-k3") or short_id == "kimi-for-coding"
+
+
# Model registry: list of (short_name, model_id, provider)
# Allows same short_name across different providers.
_MODEL_ENTRIES: list[tuple[str, str, str]] = [
@@ -161,6 +170,7 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
("gpt-5-mini", "gpt-5-mini", "custom-openai"),
# Anthropic (current generation)
("claude-fable-5", "claude-fable-5", "anthropic"),
+ ("claude-opus-5", "claude-opus-5", "anthropic"),
("claude-opus-4-8", "claude-opus-4-8", "anthropic"),
("claude-sonnet-5", "claude-sonnet-5", "anthropic"),
("claude-sonnet-4-6", "claude-sonnet-4-6", "anthropic"),
@@ -182,7 +192,9 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
("gpt-5-mini", "gpt-5-mini", "openai"),
("gpt-5-nano", "gpt-5-nano", "openai"),
# Google GenAI
+ ("gemini-3.6-flash", "gemini-3.6-flash", "google-genai"),
("gemini-3.5-flash", "gemini-3.5-flash", "google-genai"),
+ ("gemini-3.5-flash-lite", "gemini-3.5-flash-lite", "google-genai"),
("gemini-3.1-pro", "gemini-3.1-pro-preview", "google-genai"),
(
"gemini-3.1-pro-customtools",
@@ -221,6 +233,8 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
("glm-4.7", "Pro/zai-org/GLM-4.7", "siliconflow"),
# OpenRouter
("claude-fable-5", "anthropic/claude-fable-5", "openrouter"),
+ ("claude-opus-5", "anthropic/claude-opus-5", "openrouter"),
+ ("claude-opus-5-fast", "anthropic/claude-opus-5-fast", "openrouter"),
("claude-opus-4.8", "anthropic/claude-opus-4.8", "openrouter"),
("claude-opus-4.8-fast", "anthropic/claude-opus-4.8-fast", "openrouter"),
("claude-sonnet-5", "anthropic/claude-sonnet-5", "openrouter"),
@@ -232,7 +246,9 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
("gpt-5.5", "openai/gpt-5.5", "openrouter"),
("gpt-5.4", "openai/gpt-5.4", "openrouter"),
("gpt-5.3-codex", "openai/gpt-5.3-codex", "openrouter"),
+ ("gemini-3.6-flash", "google/gemini-3.6-flash", "openrouter"),
("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"),
+ ("gemini-3.5-flash-lite", "google/gemini-3.5-flash-lite", "openrouter"),
("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"),
("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"),
("kimi-k3", "moonshotai/kimi-k3", "openrouter"),
@@ -428,8 +444,15 @@ def _apply_auto_config(
else:
_is_proxy = False
if _is_proxy or (is_third_party and not _supports_thinking):
- pass
- elif "fable" in model_id or model_id.endswith(("4-6", "4-7", "4-8")):
+ # Mandatory-thinking Kimi models (K3 / Kimi For Coding) must declare
+ # thinking so with_structured_output avoids forced tool_choice (400).
+ # max_tokens must exceed budget_tokens (default resolves to 4096).
+ if is_third_party and _is_mandatory_thinking_kimi(model_id):
+ kwargs["thinking"] = {"type": "enabled", "budget_tokens": 10000}
+ kwargs.setdefault("max_tokens", 16000)
+ elif "fable" in model_id or model_id.endswith(
+ ("opus-5", "sonnet-5", "4-6", "4-7", "4-8")
+ ):
kwargs["thinking"] = {"type": "adaptive", "display": "summarized"}
kwargs.setdefault("effort", "max")
else:
@@ -744,11 +767,14 @@ def get_chat_model(
# (SiliconFlow, OpenRouter, custom-openai, etc.) and
# native OpenAI through a proxy, to avoid "sequence expected string" errors.
# Moonshot and Kimi Coding support standard format, no patch needed.
+ # Mandatory-thinking Kimi models on Anthropic-routed endpoints are exempt:
+ # flatten drops thinking blocks, which Kimi requires on tool-call turns.
_no_patch_providers = {"moonshot", "kimi-coding"}
if (
(_is_third_party or _is_openai_proxy)
and _original_provider not in _no_patch_providers
and not _uses_native_deepseek
+ and not (provider == "anthropic" and _is_mandatory_thinking_kimi(model_id))
):
# Anthropic-routed providers accept media in tool results natively;
# only OpenAI-compatible providers need tool-media hoisting.
@@ -761,6 +787,10 @@ def get_chat_model(
if provider == "openrouter":
_enable_openrouter_429_retry(chat_model)
+ if provider == "anthropic":
+ _patch_anthropic_strip_foreign_reasoning()
+ _patch_anthropic_structured_output()
+
apply_known_context_window(chat_model)
return chat_model
diff --git a/EvoScientist/llm/patches.py b/EvoScientist/llm/patches.py
index 73e11eb..dcd30aa 100644
--- a/EvoScientist/llm/patches.py
+++ b/EvoScientist/llm/patches.py
@@ -22,11 +22,19 @@ Utilities:
from __future__ import annotations
+import functools
import os
-from collections.abc import AsyncIterator, Awaitable, Callable, Iterator, Mapping
+from collections.abc import (
+ AsyncIterator,
+ Awaitable,
+ Callable,
+ Iterator,
+ Mapping,
+ Sequence,
+)
from typing import Any
-from langchain_core.messages import BaseMessage
+from langchain_core.messages import AIMessage, BaseMessage
# ---------------------------------------------------------------------------
@@ -825,6 +833,80 @@ def _patch_langgraph_schema_generator_silence_warnings() -> None:
_patch_langgraph_schema_generator_silence_warnings()
+# ---------------------------------------------------------------------------
+# Patch (lazy, Anthropic protocol): strip foreign reasoning blocks from
+# outgoing assistant messages.
+#
+# History produced on OpenAI-compatible providers (Qwen/DashScope, DeepSeek, …)
+# can carry `reasoning_content` content blocks in AI messages. When a session
+# switches to an Anthropic-protocol endpoint (native Anthropic, custom-anthropic,
+# kimi-coding, minimax), langchain-anthropic's `_format_messages` passes unknown
+# block types through verbatim (`else: content.append(block)`), and strict
+# Anthropic-compatible backends (e.g. Moonshot's Kimi For Coding endpoint at
+# api.kimi.com) reject the request with an opaque HTTP 400. Upstream already
+# skips the OpenAI-Responses `reasoning` / `function_call` types the same way;
+# `reasoning_content` is simply missing from that list. These blocks are display
+# metadata — no Anthropic-protocol block type has this name — so stripping is
+# always safe. Anthropic `thinking` blocks are preserved (Kimi backends require
+# them on tool-call turns, even unsigned), but a missing `signature` is
+# defaulted to "": Anthropic-compatible backends stream thinking without a
+# signature_delta, so the aggregated block lacks the key, and strict edges
+# (e.g. OpenRouter's Anthropic schema) reject replay with 400
+# ("signature: expected string, received undefined"). Native Anthropic thinking
+# always carries a signature string, so the default never fires there.
+# ---------------------------------------------------------------------------
+_anthropic_foreign_reasoning_patched = False
+_FOREIGN_REASONING_BLOCK_TYPES = frozenset({"reasoning_content"})
+
+
+def _normalize_anthropic_replay_messages(
+ messages: Sequence[BaseMessage],
+) -> Sequence[BaseMessage]:
+ """Sanitize AI-message list content for Anthropic-protocol replay."""
+ cleaned: list[BaseMessage] = []
+ changed = False
+ for message in messages:
+ if isinstance(message, AIMessage) and isinstance(message.content, list):
+ blocks: list[Any] = []
+ block_changed = False
+ for block in message.content:
+ if isinstance(block, dict):
+ btype = block.get("type")
+ if btype in _FOREIGN_REASONING_BLOCK_TYPES:
+ block_changed = True
+ continue
+ if btype == "thinking" and not isinstance(
+ block.get("signature"), str
+ ):
+ block = {**block, "signature": ""}
+ block_changed = True
+ blocks.append(block)
+ if block_changed:
+ message = message.model_copy(update={"content": blocks})
+ changed = True
+ cleaned.append(message)
+ return cleaned if changed else messages
+
+
+def _patch_anthropic_strip_foreign_reasoning() -> None:
+ global _anthropic_foreign_reasoning_patched
+ if _anthropic_foreign_reasoning_patched:
+ return
+ try:
+ import langchain_anthropic.chat_models as _mod
+
+ _orig = _mod._format_messages
+
+ @functools.wraps(_orig)
+ def _patched(messages: Sequence[BaseMessage]) -> Any:
+ return _orig(_normalize_anthropic_replay_messages(messages))
+
+ _mod._format_messages = _patched
+ _anthropic_foreign_reasoning_patched = True
+ except Exception:
+ pass
+
+
# ---------------------------------------------------------------------------
# Patch (lazy, OpenRouter only): strip OpenAI-Responses encrypted reasoning
# items from outgoing assistant messages.
@@ -925,6 +1007,53 @@ def _patch_openrouter_structured_output() -> None:
pass
+# ---------------------------------------------------------------------------
+# Patch (lazy, Anthropic protocol): default structured output to json_schema
+# for mandatory-thinking Kimi models (K3 / Kimi For Coding).
+#
+# These endpoints run with thinking always on and reject a forced tool choice
+# with HTTP 400 ("tool_choice 'specified' is incompatible with thinking
+# enabled") — ChatAnthropic's function_calling default forces tool_choice, so
+# every structured-output call (LLMToolSelectorMiddleware included) fails.
+# With thinking declared, langchain-anthropic falls back to tool_choice auto,
+# but the model then calls the tool only sometimes (OutputParserException).
+# These endpoints accept output_config.format (verified live via OpenRouter's
+# Anthropic-compatible edge), which needs no tool_choice — route them there.
+# Gated on the instance's model, so copies behave correctly and Claude models
+# keep the function_calling default.
+# ---------------------------------------------------------------------------
+_anthropic_structured_output_patched = False
+
+
+def _patch_anthropic_structured_output() -> None:
+ global _anthropic_structured_output_patched
+ if _anthropic_structured_output_patched:
+ return
+ try:
+ from langchain_anthropic import ChatAnthropic
+
+ _orig = ChatAnthropic.with_structured_output
+
+ def _patched(
+ self: Any,
+ schema: Any = None,
+ *,
+ method: str = "function_calling",
+ **kwargs: Any,
+ ) -> Any:
+ if method == "function_calling":
+ from .models import _is_mandatory_thinking_kimi
+
+ if _is_mandatory_thinking_kimi(self.model):
+ method = "json_schema"
+ return _orig(self, schema, method=method, **kwargs)
+
+ ChatAnthropic.with_structured_output = _patched
+ _anthropic_structured_output_patched = True
+ except Exception:
+ pass
+
+
# ---------------------------------------------------------------------------
# Patch: forward CLI's live (model, model_provider) into deepagents'
# start_async_task / update_async_task tool calls so the deployed graph
diff --git a/README.md b/README.md
index 334be39..dc8ac90 100644
--- a/README.md
+++ b/README.md
@@ -10,7 +10,7 @@
-
+
@@ -151,6 +151,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
📦 Release Highlights — version changelog
+- **[26 Jul 2026]** **[v0.2.4](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.4)** — Claude Opus 5 selectable on Anthropic and OpenRouter (incl. fast), plus Gemini 3.6 Flash and 3.5 Flash Lite on Google and OpenRouter; Kimi K3 now works over Anthropic-protocol channels (Kimi For Coding, custom endpoints), covering structured output, history replay, and multi-turn thinking; fixes for interrupted tool-call history, skill-install path leaks, and OpenRouter SSE streaming (pinned below 0.11).
- **[18 Jul 2026]** **[v0.2.3](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.3)** — Kimi K3 selectable on Moonshot and OpenRouter (1M context); async sub-agent runs no longer get stuck pending thanks to orphaned-run cleanup; Telegram slash commands; provider fixes (DeepSeek native SDK, GPT-5.x via ChatGPT OAuth, OpenAI `reasoning_effort`); quieter tool-selector streaming and smaller checkpoints.
- **[11 Jul 2026]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — New models selectable in onboarding and `/model`: GPT-5.6 (sol, terra, luna) for OpenAI and OpenRouter, plus Grok 4.5 and Tencent Hunyuan HY3 on OpenRouter; tighter config-file permissions and a reworked onboarding OAuth flow for auxiliary models.
- **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5.
@@ -749,9 +750,9 @@ Every contribution brings us one step closer to a future where AI accelerates sc
-
+[](https://www.star-history.com/?repos=EvoScientist%2FEvoScientist&type=date&legend=top-left)
🔝Back to top
diff --git a/README.zh-CN.md b/README.zh-CN.md
index 2a5cf20..5f614b5 100644
--- a/README.zh-CN.md
+++ b/README.zh-CN.md
@@ -15,7 +15,7 @@
-
+
@@ -160,6 +160,9 @@ EvoScientist 超越了传统的人在回路(Human-in-the-Loop)模式,采
📦 版本更新摘要(changelog)
+- **[2026 年 7 月 26 日]** **[v0.2.4](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.4)** — Claude Opus 5 可在 Anthropic 与 OpenRouter 中选用(含 fast),Google 与 OpenRouter 另新增 Gemini 3.6 Flash 与 3.5 Flash Lite;Kimi K3 打通 Anthropic 协议通道(Kimi For Coding、自定义端点),结构化输出、历史回放、多轮 thinking 均可用;修复中断的工具调用历史、技能安装路径泄漏,以及 OpenRouter SSE 流式回归(固定到 0.11 以下)。
+- **[2026 年 7 月 18 日]** **[v0.2.3](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.3)** — Kimi K3 可在 Moonshot 与 OpenRouter 中选用(1M 上下文);清理孤儿运行后,异步子代理不再卡在 pending 状态;Telegram 斜杠命令;provider 修复(DeepSeek 原生 SDK、ChatGPT OAuth 下的 GPT-5.x、OpenAI `reasoning_effort`);tool-selector 流式输出更安静,checkpoint 体积更小。
+- **[2026 年 7 月 11 日]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — onboarding 与 `/model` 新增可选模型:OpenAI 与 OpenRouter 的 GPT-5.6(sol、terra、luna),以及 OpenRouter 上的 Grok 4.5 与腾讯混元 HY3;收紧配置文件权限,并重做了辅助模型的 onboarding OAuth 流程。
- **[2026 年 7 月 5 日]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills:EvoMemory 从自身的观察记录聚类中起草可复用技能,供你用 `/autoskills` 审核;新增面向无头 / SDK 客户端的 `--output-format stream-json`;更丰富的斜杠命令补全;修复 Windows UTF-8 配置读取;TUI 欢迎横幅修复;langchain-openrouter 升级到 0.2.5。
- **[2026 年 6 月 26 日]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — 定时任务:用 `/schedule` 或自然语言设置 cron 风格的重复运行,无人值守并对 shell 访问做门控;记忆自连成图:将观察记录连成知识图谱(互补 / 矛盾 / 取代);新增只读 `GET /api/models` 端点,供 WebUI 模型选择器使用。
- **[2026 年 6 月 23 日]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — 新安装热修复:deepagents 0.6.11 / langchain-quickjs 0.3 将 `task` 保留为 REPL 全局后,首次对话即崩溃(`The subagent `task` tool cannot be exposed via `ptc``)。从 code-interpreter 的 PTC 白名单移除 `task`(`task()` 仍作为 REPL 全局可用,异步分发工具继续保留在 PTC 中),并将 deepagents pin 升级到 `~=0.6.11`。
@@ -752,9 +755,9 @@ Jan Piotrowski, Wiktor Cupiał, Jakub Kaliski, Jakub Filipiuk, Xinhao Yi, Shuyu
-
+[](https://www.star-history.com/?repos=EvoScientist%2FEvoScientist&type=date&legend=top-left)
🔝回到顶部
diff --git a/pyproject.toml b/pyproject.toml
index 6b0300a..0e06e2d 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,6 +1,6 @@
[project]
name = "EvoScientist"
-version = "0.2.3"
+version = "0.2.4"
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
readme = "README.md"
requires-python = ">=3.11"
diff --git a/tests/test_llm.py b/tests/test_llm.py
index 92e0056..4fab005 100644
--- a/tests/test_llm.py
+++ b/tests/test_llm.py
@@ -2500,6 +2500,229 @@ class TestPatchOpenrouterStripResponsesReasoning:
self._restore(patches, mod, orig, orig_flag)
+# =============================================================================
+# Test _patch_anthropic_strip_foreign_reasoning
+# =============================================================================
+
+
+class TestAnthropicStripForeignReasoning:
+ def test_strip_removes_reasoning_content_blocks(self):
+ """reasoning_content blocks are dropped; text and thinking survive."""
+ from langchain_core.messages import AIMessage, HumanMessage
+
+ from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
+
+ messages = [
+ HumanMessage("hello"),
+ AIMessage(
+ content=[
+ {"type": "reasoning_content", "reasoning_content": {"text": "hm"}},
+ {"type": "thinking", "thinking": "hm", "signature": ""},
+ {"type": "text", "text": "hi"},
+ ]
+ ),
+ ]
+
+ result = _normalize_anthropic_replay_messages(messages)
+
+ types = [b["type"] for b in result[1].content]
+ assert types == ["thinking", "text"]
+
+ def test_missing_thinking_signature_defaulted(self):
+ """Streamed thinking blocks without a signature key get signature ''."""
+ from langchain_core.messages import AIMessage
+
+ from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
+
+ messages = [
+ AIMessage(
+ content=[
+ {"type": "thinking", "thinking": "hm", "index": 0},
+ {"type": "text", "text": "hi", "index": 1},
+ ]
+ ),
+ ]
+
+ result = _normalize_anthropic_replay_messages(messages)
+
+ assert result[0].content[0]["signature"] == ""
+ assert "signature" not in result[0].content[1]
+
+ def test_strip_no_change_returns_same_object(self):
+ """Clean histories pass through without copying."""
+ from langchain_core.messages import AIMessage, HumanMessage
+
+ from EvoScientist.llm.patches import _normalize_anthropic_replay_messages
+
+ messages = [
+ HumanMessage("hello"),
+ AIMessage(content=[{"type": "text", "text": "hi"}]),
+ AIMessage(content="plain string content"),
+ ]
+
+ assert _normalize_anthropic_replay_messages(messages) is messages
+
+ def test_kimi_k3_exempt_from_flatten_patch(self, monkeypatch):
+ """K3 on custom-anthropic gets no instance flatten closures; others do."""
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
+
+ kimi = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
+ assert "_generate" not in vars(kimi)
+
+ other = get_chat_model(
+ "claude-sonnet-4-6", provider="custom-anthropic", max_tokens=1024
+ )
+ assert "_generate" in vars(other)
+
+ def test_reasoning_content_stripped_on_the_wire(self, monkeypatch):
+ """End-to-end: foreign reasoning blocks never reach the Anthropic wire."""
+ import json
+
+ import anthropic
+ import httpx
+ from langchain_core.messages import AIMessage, HumanMessage
+
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
+ model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
+
+ captured: dict = {}
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ captured.update(json.loads(request.content.decode()))
+ return httpx.Response(
+ 200,
+ json={
+ "id": "msg_test",
+ "type": "message",
+ "role": "assistant",
+ "content": [{"type": "text", "text": "ok"}],
+ "model": "moonshotai/kimi-k3",
+ "stop_reason": "end_turn",
+ "stop_sequence": None,
+ "usage": {"input_tokens": 1, "output_tokens": 1},
+ },
+ )
+
+ model._client = anthropic.Anthropic(
+ api_key="test-key",
+ base_url="https://compat.example.com",
+ http_client=httpx.Client(transport=httpx.MockTransport(handler)),
+ )
+
+ history = [
+ HumanMessage("hello"),
+ AIMessage(
+ content=[
+ {"type": "reasoning_content", "reasoning_content": {"text": "hm"}},
+ {"type": "thinking", "thinking": "hm", "index": 0},
+ {"type": "text", "text": "hi there"},
+ ]
+ ),
+ HumanMessage("say ok"),
+ ]
+ result = model.invoke(history)
+
+ sent_blocks = [
+ block
+ for message in captured["messages"]
+ for block in (
+ message["content"] if isinstance(message["content"], list) else []
+ )
+ ]
+ sent_types = [block["type"] for block in sent_blocks]
+ assert "reasoning_content" not in sent_types
+ assert "text" in sent_types
+ thinking_blocks = [b for b in sent_blocks if b["type"] == "thinking"]
+ assert thinking_blocks
+ assert thinking_blocks[0]["signature"] == ""
+ assert result.content == "ok"
+
+
+# =============================================================================
+# Test _patch_anthropic_structured_output
+# =============================================================================
+
+
+class TestAnthropicStructuredOutput:
+ @staticmethod
+ def _capture_structured_request(model, response_text):
+ """Invoke a structured-output runnable against a capturing transport."""
+ import json
+
+ import anthropic
+ import httpx
+ from pydantic import BaseModel
+
+ class Pick(BaseModel):
+ answer: str
+
+ captured: dict = {}
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ captured.update(json.loads(request.content.decode()))
+ return httpx.Response(
+ 200,
+ json={
+ "id": "msg_test",
+ "type": "message",
+ "role": "assistant",
+ "content": response_text,
+ "model": "test",
+ "stop_reason": "end_turn",
+ "stop_sequence": None,
+ "usage": {"input_tokens": 1, "output_tokens": 1},
+ },
+ )
+
+ model._client = anthropic.Anthropic(
+ api_key="test-key",
+ base_url="https://compat.example.com",
+ http_client=httpx.Client(transport=httpx.MockTransport(handler)),
+ )
+ result = model.with_structured_output(Pick).invoke("Reply with answer='ok'")
+ return captured, result
+
+ def test_kimi_k3_defaults_to_json_schema(self, monkeypatch):
+ """K3 structured output binds output_config.format, no forced tool_choice."""
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
+ model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic")
+
+ captured, result = self._capture_structured_request(
+ model, [{"type": "text", "text": '{"answer": "ok"}'}]
+ )
+
+ assert captured["output_config"]["format"]["type"] == "json_schema"
+ assert "tool_choice" not in captured
+ assert result.answer == "ok"
+
+ def test_claude_keeps_function_calling(self, monkeypatch):
+ """Claude models keep tool-based structured output (no json_schema flip)."""
+ monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
+ monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key")
+ model = get_chat_model(
+ "claude-haiku-4-5", provider="anthropic", max_tokens=1024
+ )
+
+ captured, result = self._capture_structured_request(
+ model,
+ [
+ {
+ "type": "tool_use",
+ "id": "toolu_1",
+ "name": "Pick",
+ "input": {"answer": "ok"},
+ }
+ ],
+ )
+
+ assert "output_config" not in captured
+ assert [t["name"] for t in captured["tools"]] == ["Pick"]
+ assert result.answer == "ok"
+
+
# =============================================================================
# Test _apply_auto_config
# =============================================================================
@@ -2546,6 +2769,58 @@ class TestAutoConfig:
assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
assert call_kwargs["effort"] == "max"
+ @pytest.mark.parametrize("model", ["claude-opus-5", "claude-sonnet-5"])
+ @patch("EvoScientist.llm.models.init_chat_model")
+ def test_anthropic_5_series_adaptive_thinking(self, mock_init, model, monkeypatch):
+ """Anthropic 5-series models get adaptive thinking (budget_tokens would 400)."""
+ mock_init.return_value = "mock_model"
+ monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False)
+
+ get_chat_model(model, provider="anthropic")
+
+ call_kwargs = mock_init.call_args[1]
+ assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"}
+ assert call_kwargs["effort"] == "max"
+
+ @pytest.mark.parametrize("model", ["moonshotai/kimi-k3", "kimi-k3"])
+ @patch("EvoScientist.llm.models.init_chat_model")
+ def test_custom_anthropic_kimi_k3_declares_thinking(
+ self, mock_init, model, monkeypatch
+ ):
+ """K3 via custom-anthropic declares thinking (else forced tool_choice 400s)."""
+ mock_init.return_value = "mock_model"
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
+
+ get_chat_model(model, provider="custom-anthropic")
+
+ call_kwargs = mock_init.call_args[1]
+ assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
+ assert call_kwargs["max_tokens"] == 16000
+
+ @patch("EvoScientist.llm.models.init_chat_model")
+ def test_kimi_coding_declares_thinking(self, mock_init):
+ """Kimi For Coding plan models declare thinking on the kimi-coding provider."""
+ mock_init.return_value = "mock_model"
+
+ get_chat_model("kimi-for-coding", provider="kimi-coding")
+
+ call_kwargs = mock_init.call_args[1]
+ assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000}
+ assert call_kwargs["max_tokens"] == 16000
+
+ @patch("EvoScientist.llm.models.init_chat_model")
+ def test_custom_anthropic_non_kimi_no_thinking(self, mock_init, monkeypatch):
+ """Non-Kimi models on custom-anthropic still skip thinking injection."""
+ mock_init.return_value = "mock_model"
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com")
+ monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key")
+
+ get_chat_model("glm-4.7", provider="custom-anthropic")
+
+ call_kwargs = mock_init.call_args[1]
+ assert "thinking" not in call_kwargs
+
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_4_6_proxy_no_thinking(self, mock_init, monkeypatch):
"""Anthropic 4-6 models via proxy skip thinking (history round-trip 422)."""
diff --git a/uv.lock b/uv.lock
index 5e9b492..c6e855b 100644
--- a/uv.lock
+++ b/uv.lock
@@ -952,7 +952,7 @@ wheels = [
[[package]]
name = "evoscientist"
-version = "0.2.3"
+version = "0.2.4"
source = { editable = "." }
dependencies = [
{ name = "deepagents", extra = ["quickjs"] },