diff --git a/.github/assets/badge-pypi-dark.svg b/.github/assets/badge-pypi-dark.svg index c6c5f09..9d5db97 100644 --- a/.github/assets/badge-pypi-dark.svg +++ b/.github/assets/badge-pypi-dark.svg @@ -5,5 +5,5 @@ v0.2.3 + font-size="13" font-weight="700" fill="#ffffff">v0.2.4 \ No newline at end of file diff --git a/.github/assets/badge-pypi-light.svg b/.github/assets/badge-pypi-light.svg index 0b141e2..49f7df1 100644 --- a/.github/assets/badge-pypi-light.svg +++ b/.github/assets/badge-pypi-light.svg @@ -5,5 +5,5 @@ v0.2.3 + font-size="13" font-weight="700" fill="#ffffff">v0.2.4 \ No newline at end of file diff --git a/.github/assets/wechat_group.jpeg b/.github/assets/wechat_group.jpeg index 2f92b22..b41363c 100644 Binary files a/.github/assets/wechat_group.jpeg and b/.github/assets/wechat_group.jpeg differ diff --git a/EvoScientist/llm/models.py b/EvoScientist/llm/models.py index f7d7fbd..bdcd4f2 100644 --- a/EvoScientist/llm/models.py +++ b/EvoScientist/llm/models.py @@ -28,6 +28,8 @@ from .context_window import apply_known_context_window from .deepseek import EvoChatDeepSeek from .patches import ( _is_ccproxy_codex, + _patch_anthropic_strip_foreign_reasoning, + _patch_anthropic_structured_output, _patch_ccproxy_system_to_developer, _patch_openai_compat_content, _patch_openrouter_strip_responses_reasoning, @@ -145,6 +147,13 @@ _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset( {"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"} ) + +def _is_mandatory_thinking_kimi(model_id: str) -> bool: + """True for Kimi models whose thinking cannot be disabled (K3 family).""" + short_id = model_id.split("/")[-1] + return short_id.startswith("kimi-k3") or short_id == "kimi-for-coding" + + # Model registry: list of (short_name, model_id, provider) # Allows same short_name across different providers. _MODEL_ENTRIES: list[tuple[str, str, str]] = [ @@ -161,6 +170,7 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [ ("gpt-5-mini", "gpt-5-mini", "custom-openai"), # Anthropic (current generation) ("claude-fable-5", "claude-fable-5", "anthropic"), + ("claude-opus-5", "claude-opus-5", "anthropic"), ("claude-opus-4-8", "claude-opus-4-8", "anthropic"), ("claude-sonnet-5", "claude-sonnet-5", "anthropic"), ("claude-sonnet-4-6", "claude-sonnet-4-6", "anthropic"), @@ -182,7 +192,9 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [ ("gpt-5-mini", "gpt-5-mini", "openai"), ("gpt-5-nano", "gpt-5-nano", "openai"), # Google GenAI + ("gemini-3.6-flash", "gemini-3.6-flash", "google-genai"), ("gemini-3.5-flash", "gemini-3.5-flash", "google-genai"), + ("gemini-3.5-flash-lite", "gemini-3.5-flash-lite", "google-genai"), ("gemini-3.1-pro", "gemini-3.1-pro-preview", "google-genai"), ( "gemini-3.1-pro-customtools", @@ -221,6 +233,8 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [ ("glm-4.7", "Pro/zai-org/GLM-4.7", "siliconflow"), # OpenRouter ("claude-fable-5", "anthropic/claude-fable-5", "openrouter"), + ("claude-opus-5", "anthropic/claude-opus-5", "openrouter"), + ("claude-opus-5-fast", "anthropic/claude-opus-5-fast", "openrouter"), ("claude-opus-4.8", "anthropic/claude-opus-4.8", "openrouter"), ("claude-opus-4.8-fast", "anthropic/claude-opus-4.8-fast", "openrouter"), ("claude-sonnet-5", "anthropic/claude-sonnet-5", "openrouter"), @@ -232,7 +246,9 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [ ("gpt-5.5", "openai/gpt-5.5", "openrouter"), ("gpt-5.4", "openai/gpt-5.4", "openrouter"), ("gpt-5.3-codex", "openai/gpt-5.3-codex", "openrouter"), + ("gemini-3.6-flash", "google/gemini-3.6-flash", "openrouter"), ("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"), + ("gemini-3.5-flash-lite", "google/gemini-3.5-flash-lite", "openrouter"), ("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"), ("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"), ("kimi-k3", "moonshotai/kimi-k3", "openrouter"), @@ -428,8 +444,15 @@ def _apply_auto_config( else: _is_proxy = False if _is_proxy or (is_third_party and not _supports_thinking): - pass - elif "fable" in model_id or model_id.endswith(("4-6", "4-7", "4-8")): + # Mandatory-thinking Kimi models (K3 / Kimi For Coding) must declare + # thinking so with_structured_output avoids forced tool_choice (400). + # max_tokens must exceed budget_tokens (default resolves to 4096). + if is_third_party and _is_mandatory_thinking_kimi(model_id): + kwargs["thinking"] = {"type": "enabled", "budget_tokens": 10000} + kwargs.setdefault("max_tokens", 16000) + elif "fable" in model_id or model_id.endswith( + ("opus-5", "sonnet-5", "4-6", "4-7", "4-8") + ): kwargs["thinking"] = {"type": "adaptive", "display": "summarized"} kwargs.setdefault("effort", "max") else: @@ -744,11 +767,14 @@ def get_chat_model( # (SiliconFlow, OpenRouter, custom-openai, etc.) and # native OpenAI through a proxy, to avoid "sequence expected string" errors. # Moonshot and Kimi Coding support standard format, no patch needed. + # Mandatory-thinking Kimi models on Anthropic-routed endpoints are exempt: + # flatten drops thinking blocks, which Kimi requires on tool-call turns. _no_patch_providers = {"moonshot", "kimi-coding"} if ( (_is_third_party or _is_openai_proxy) and _original_provider not in _no_patch_providers and not _uses_native_deepseek + and not (provider == "anthropic" and _is_mandatory_thinking_kimi(model_id)) ): # Anthropic-routed providers accept media in tool results natively; # only OpenAI-compatible providers need tool-media hoisting. @@ -761,6 +787,10 @@ def get_chat_model( if provider == "openrouter": _enable_openrouter_429_retry(chat_model) + if provider == "anthropic": + _patch_anthropic_strip_foreign_reasoning() + _patch_anthropic_structured_output() + apply_known_context_window(chat_model) return chat_model diff --git a/EvoScientist/llm/patches.py b/EvoScientist/llm/patches.py index 73e11eb..dcd30aa 100644 --- a/EvoScientist/llm/patches.py +++ b/EvoScientist/llm/patches.py @@ -22,11 +22,19 @@ Utilities: from __future__ import annotations +import functools import os -from collections.abc import AsyncIterator, Awaitable, Callable, Iterator, Mapping +from collections.abc import ( + AsyncIterator, + Awaitable, + Callable, + Iterator, + Mapping, + Sequence, +) from typing import Any -from langchain_core.messages import BaseMessage +from langchain_core.messages import AIMessage, BaseMessage # --------------------------------------------------------------------------- @@ -825,6 +833,80 @@ def _patch_langgraph_schema_generator_silence_warnings() -> None: _patch_langgraph_schema_generator_silence_warnings() +# --------------------------------------------------------------------------- +# Patch (lazy, Anthropic protocol): strip foreign reasoning blocks from +# outgoing assistant messages. +# +# History produced on OpenAI-compatible providers (Qwen/DashScope, DeepSeek, …) +# can carry `reasoning_content` content blocks in AI messages. When a session +# switches to an Anthropic-protocol endpoint (native Anthropic, custom-anthropic, +# kimi-coding, minimax), langchain-anthropic's `_format_messages` passes unknown +# block types through verbatim (`else: content.append(block)`), and strict +# Anthropic-compatible backends (e.g. Moonshot's Kimi For Coding endpoint at +# api.kimi.com) reject the request with an opaque HTTP 400. Upstream already +# skips the OpenAI-Responses `reasoning` / `function_call` types the same way; +# `reasoning_content` is simply missing from that list. These blocks are display +# metadata — no Anthropic-protocol block type has this name — so stripping is +# always safe. Anthropic `thinking` blocks are preserved (Kimi backends require +# them on tool-call turns, even unsigned), but a missing `signature` is +# defaulted to "": Anthropic-compatible backends stream thinking without a +# signature_delta, so the aggregated block lacks the key, and strict edges +# (e.g. OpenRouter's Anthropic schema) reject replay with 400 +# ("signature: expected string, received undefined"). Native Anthropic thinking +# always carries a signature string, so the default never fires there. +# --------------------------------------------------------------------------- +_anthropic_foreign_reasoning_patched = False +_FOREIGN_REASONING_BLOCK_TYPES = frozenset({"reasoning_content"}) + + +def _normalize_anthropic_replay_messages( + messages: Sequence[BaseMessage], +) -> Sequence[BaseMessage]: + """Sanitize AI-message list content for Anthropic-protocol replay.""" + cleaned: list[BaseMessage] = [] + changed = False + for message in messages: + if isinstance(message, AIMessage) and isinstance(message.content, list): + blocks: list[Any] = [] + block_changed = False + for block in message.content: + if isinstance(block, dict): + btype = block.get("type") + if btype in _FOREIGN_REASONING_BLOCK_TYPES: + block_changed = True + continue + if btype == "thinking" and not isinstance( + block.get("signature"), str + ): + block = {**block, "signature": ""} + block_changed = True + blocks.append(block) + if block_changed: + message = message.model_copy(update={"content": blocks}) + changed = True + cleaned.append(message) + return cleaned if changed else messages + + +def _patch_anthropic_strip_foreign_reasoning() -> None: + global _anthropic_foreign_reasoning_patched + if _anthropic_foreign_reasoning_patched: + return + try: + import langchain_anthropic.chat_models as _mod + + _orig = _mod._format_messages + + @functools.wraps(_orig) + def _patched(messages: Sequence[BaseMessage]) -> Any: + return _orig(_normalize_anthropic_replay_messages(messages)) + + _mod._format_messages = _patched + _anthropic_foreign_reasoning_patched = True + except Exception: + pass + + # --------------------------------------------------------------------------- # Patch (lazy, OpenRouter only): strip OpenAI-Responses encrypted reasoning # items from outgoing assistant messages. @@ -925,6 +1007,53 @@ def _patch_openrouter_structured_output() -> None: pass +# --------------------------------------------------------------------------- +# Patch (lazy, Anthropic protocol): default structured output to json_schema +# for mandatory-thinking Kimi models (K3 / Kimi For Coding). +# +# These endpoints run with thinking always on and reject a forced tool choice +# with HTTP 400 ("tool_choice 'specified' is incompatible with thinking +# enabled") — ChatAnthropic's function_calling default forces tool_choice, so +# every structured-output call (LLMToolSelectorMiddleware included) fails. +# With thinking declared, langchain-anthropic falls back to tool_choice auto, +# but the model then calls the tool only sometimes (OutputParserException). +# These endpoints accept output_config.format (verified live via OpenRouter's +# Anthropic-compatible edge), which needs no tool_choice — route them there. +# Gated on the instance's model, so copies behave correctly and Claude models +# keep the function_calling default. +# --------------------------------------------------------------------------- +_anthropic_structured_output_patched = False + + +def _patch_anthropic_structured_output() -> None: + global _anthropic_structured_output_patched + if _anthropic_structured_output_patched: + return + try: + from langchain_anthropic import ChatAnthropic + + _orig = ChatAnthropic.with_structured_output + + def _patched( + self: Any, + schema: Any = None, + *, + method: str = "function_calling", + **kwargs: Any, + ) -> Any: + if method == "function_calling": + from .models import _is_mandatory_thinking_kimi + + if _is_mandatory_thinking_kimi(self.model): + method = "json_schema" + return _orig(self, schema, method=method, **kwargs) + + ChatAnthropic.with_structured_output = _patched + _anthropic_structured_output_patched = True + except Exception: + pass + + # --------------------------------------------------------------------------- # Patch: forward CLI's live (model, model_provider) into deepagents' # start_async_task / update_async_task tool calls so the deployed graph diff --git a/README.md b/README.md index 334be39..dc8ac90 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ - PyPI v0.2.3 + PyPI v0.2.4 @@ -151,6 +151,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
📦 Release Highlights — version changelog +- **[26 Jul 2026]** **[v0.2.4](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.4)** — Claude Opus 5 selectable on Anthropic and OpenRouter (incl. fast), plus Gemini 3.6 Flash and 3.5 Flash Lite on Google and OpenRouter; Kimi K3 now works over Anthropic-protocol channels (Kimi For Coding, custom endpoints), covering structured output, history replay, and multi-turn thinking; fixes for interrupted tool-call history, skill-install path leaks, and OpenRouter SSE streaming (pinned below 0.11). - **[18 Jul 2026]** **[v0.2.3](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.3)** — Kimi K3 selectable on Moonshot and OpenRouter (1M context); async sub-agent runs no longer get stuck pending thanks to orphaned-run cleanup; Telegram slash commands; provider fixes (DeepSeek native SDK, GPT-5.x via ChatGPT OAuth, OpenAI `reasoning_effort`); quieter tool-selector streaming and smaller checkpoints. - **[11 Jul 2026]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — New models selectable in onboarding and `/model`: GPT-5.6 (sol, terra, luna) for OpenAI and OpenRouter, plus Grok 4.5 and Tencent Hunyuan HY3 on OpenRouter; tighter config-file permissions and a reworked onboarding OAuth flow for auxiliary models. - **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5. @@ -749,9 +750,9 @@ Every contribution brings us one step closer to a future where AI accelerates sc - +[![Star History Chart](https://api.star-history.com/chart?repos=EvoScientist/EvoScientist&type=date&legend=top-left&sealed_token=-XivKBib6Pb_YTJjMxBwUghZaRWxGqr5HYBKsa5jyiCgVMWfHmmkLyCYbT0uUvJJdUQsza9mRnlk1-QVQzm-s0UExQ_8DIBSrWKIrPQz5WzNlRURsUoHSA)](https://www.star-history.com/?repos=EvoScientist%2FEvoScientist&type=date&legend=top-left)

🔝Back to top

diff --git a/README.zh-CN.md b/README.zh-CN.md index 2a5cf20..5f614b5 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -15,7 +15,7 @@ - PyPI v0.2.1 + PyPI v0.2.4 @@ -160,6 +160,9 @@ EvoScientist 超越了传统的人在回路(Human-in-the-Loop)模式,采
📦 版本更新摘要(changelog) +- **[2026 年 7 月 26 日]** **[v0.2.4](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.4)** — Claude Opus 5 可在 Anthropic 与 OpenRouter 中选用(含 fast),Google 与 OpenRouter 另新增 Gemini 3.6 Flash 与 3.5 Flash Lite;Kimi K3 打通 Anthropic 协议通道(Kimi For Coding、自定义端点),结构化输出、历史回放、多轮 thinking 均可用;修复中断的工具调用历史、技能安装路径泄漏,以及 OpenRouter SSE 流式回归(固定到 0.11 以下)。 +- **[2026 年 7 月 18 日]** **[v0.2.3](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.3)** — Kimi K3 可在 Moonshot 与 OpenRouter 中选用(1M 上下文);清理孤儿运行后,异步子代理不再卡在 pending 状态;Telegram 斜杠命令;provider 修复(DeepSeek 原生 SDK、ChatGPT OAuth 下的 GPT-5.x、OpenAI `reasoning_effort`);tool-selector 流式输出更安静,checkpoint 体积更小。 +- **[2026 年 7 月 11 日]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — onboarding 与 `/model` 新增可选模型:OpenAI 与 OpenRouter 的 GPT-5.6(sol、terra、luna),以及 OpenRouter 上的 Grok 4.5 与腾讯混元 HY3;收紧配置文件权限,并重做了辅助模型的 onboarding OAuth 流程。 - **[2026 年 7 月 5 日]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills:EvoMemory 从自身的观察记录聚类中起草可复用技能,供你用 `/autoskills` 审核;新增面向无头 / SDK 客户端的 `--output-format stream-json`;更丰富的斜杠命令补全;修复 Windows UTF-8 配置读取;TUI 欢迎横幅修复;langchain-openrouter 升级到 0.2.5。 - **[2026 年 6 月 26 日]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — 定时任务:用 `/schedule` 或自然语言设置 cron 风格的重复运行,无人值守并对 shell 访问做门控;记忆自连成图:将观察记录连成知识图谱(互补 / 矛盾 / 取代);新增只读 `GET /api/models` 端点,供 WebUI 模型选择器使用。 - **[2026 年 6 月 23 日]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — 新安装热修复:deepagents 0.6.11 / langchain-quickjs 0.3 将 `task` 保留为 REPL 全局后,首次对话即崩溃(`The subagent `task` tool cannot be exposed via `ptc``)。从 code-interpreter 的 PTC 白名单移除 `task`(`task()` 仍作为 REPL 全局可用,异步分发工具继续保留在 PTC 中),并将 deepagents pin 升级到 `~=0.6.11`。 @@ -752,9 +755,9 @@ Jan Piotrowski, Wiktor Cupiał, Jakub Kaliski, Jakub Filipiuk, Xinhao Yi, Shuyu - +[![Star History Chart](https://api.star-history.com/chart?repos=EvoScientist/EvoScientist&type=date&legend=top-left&sealed_token=-XivKBib6Pb_YTJjMxBwUghZaRWxGqr5HYBKsa5jyiCgVMWfHmmkLyCYbT0uUvJJdUQsza9mRnlk1-QVQzm-s0UExQ_8DIBSrWKIrPQz5WzNlRURsUoHSA)](https://www.star-history.com/?repos=EvoScientist%2FEvoScientist&type=date&legend=top-left)

🔝回到顶部

diff --git a/pyproject.toml b/pyproject.toml index 6b0300a..0e06e2d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "EvoScientist" -version = "0.2.3" +version = "0.2.4" description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery" readme = "README.md" requires-python = ">=3.11" diff --git a/tests/test_llm.py b/tests/test_llm.py index 92e0056..4fab005 100644 --- a/tests/test_llm.py +++ b/tests/test_llm.py @@ -2500,6 +2500,229 @@ class TestPatchOpenrouterStripResponsesReasoning: self._restore(patches, mod, orig, orig_flag) +# ============================================================================= +# Test _patch_anthropic_strip_foreign_reasoning +# ============================================================================= + + +class TestAnthropicStripForeignReasoning: + def test_strip_removes_reasoning_content_blocks(self): + """reasoning_content blocks are dropped; text and thinking survive.""" + from langchain_core.messages import AIMessage, HumanMessage + + from EvoScientist.llm.patches import _normalize_anthropic_replay_messages + + messages = [ + HumanMessage("hello"), + AIMessage( + content=[ + {"type": "reasoning_content", "reasoning_content": {"text": "hm"}}, + {"type": "thinking", "thinking": "hm", "signature": ""}, + {"type": "text", "text": "hi"}, + ] + ), + ] + + result = _normalize_anthropic_replay_messages(messages) + + types = [b["type"] for b in result[1].content] + assert types == ["thinking", "text"] + + def test_missing_thinking_signature_defaulted(self): + """Streamed thinking blocks without a signature key get signature ''.""" + from langchain_core.messages import AIMessage + + from EvoScientist.llm.patches import _normalize_anthropic_replay_messages + + messages = [ + AIMessage( + content=[ + {"type": "thinking", "thinking": "hm", "index": 0}, + {"type": "text", "text": "hi", "index": 1}, + ] + ), + ] + + result = _normalize_anthropic_replay_messages(messages) + + assert result[0].content[0]["signature"] == "" + assert "signature" not in result[0].content[1] + + def test_strip_no_change_returns_same_object(self): + """Clean histories pass through without copying.""" + from langchain_core.messages import AIMessage, HumanMessage + + from EvoScientist.llm.patches import _normalize_anthropic_replay_messages + + messages = [ + HumanMessage("hello"), + AIMessage(content=[{"type": "text", "text": "hi"}]), + AIMessage(content="plain string content"), + ] + + assert _normalize_anthropic_replay_messages(messages) is messages + + def test_kimi_k3_exempt_from_flatten_patch(self, monkeypatch): + """K3 on custom-anthropic gets no instance flatten closures; others do.""" + monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com") + monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key") + + kimi = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic") + assert "_generate" not in vars(kimi) + + other = get_chat_model( + "claude-sonnet-4-6", provider="custom-anthropic", max_tokens=1024 + ) + assert "_generate" in vars(other) + + def test_reasoning_content_stripped_on_the_wire(self, monkeypatch): + """End-to-end: foreign reasoning blocks never reach the Anthropic wire.""" + import json + + import anthropic + import httpx + from langchain_core.messages import AIMessage, HumanMessage + + monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com") + monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key") + model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic") + + captured: dict = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured.update(json.loads(request.content.decode())) + return httpx.Response( + 200, + json={ + "id": "msg_test", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "ok"}], + "model": "moonshotai/kimi-k3", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 1, "output_tokens": 1}, + }, + ) + + model._client = anthropic.Anthropic( + api_key="test-key", + base_url="https://compat.example.com", + http_client=httpx.Client(transport=httpx.MockTransport(handler)), + ) + + history = [ + HumanMessage("hello"), + AIMessage( + content=[ + {"type": "reasoning_content", "reasoning_content": {"text": "hm"}}, + {"type": "thinking", "thinking": "hm", "index": 0}, + {"type": "text", "text": "hi there"}, + ] + ), + HumanMessage("say ok"), + ] + result = model.invoke(history) + + sent_blocks = [ + block + for message in captured["messages"] + for block in ( + message["content"] if isinstance(message["content"], list) else [] + ) + ] + sent_types = [block["type"] for block in sent_blocks] + assert "reasoning_content" not in sent_types + assert "text" in sent_types + thinking_blocks = [b for b in sent_blocks if b["type"] == "thinking"] + assert thinking_blocks + assert thinking_blocks[0]["signature"] == "" + assert result.content == "ok" + + +# ============================================================================= +# Test _patch_anthropic_structured_output +# ============================================================================= + + +class TestAnthropicStructuredOutput: + @staticmethod + def _capture_structured_request(model, response_text): + """Invoke a structured-output runnable against a capturing transport.""" + import json + + import anthropic + import httpx + from pydantic import BaseModel + + class Pick(BaseModel): + answer: str + + captured: dict = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured.update(json.loads(request.content.decode())) + return httpx.Response( + 200, + json={ + "id": "msg_test", + "type": "message", + "role": "assistant", + "content": response_text, + "model": "test", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 1, "output_tokens": 1}, + }, + ) + + model._client = anthropic.Anthropic( + api_key="test-key", + base_url="https://compat.example.com", + http_client=httpx.Client(transport=httpx.MockTransport(handler)), + ) + result = model.with_structured_output(Pick).invoke("Reply with answer='ok'") + return captured, result + + def test_kimi_k3_defaults_to_json_schema(self, monkeypatch): + """K3 structured output binds output_config.format, no forced tool_choice.""" + monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com") + monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key") + model = get_chat_model("moonshotai/kimi-k3", provider="custom-anthropic") + + captured, result = self._capture_structured_request( + model, [{"type": "text", "text": '{"answer": "ok"}'}] + ) + + assert captured["output_config"]["format"]["type"] == "json_schema" + assert "tool_choice" not in captured + assert result.answer == "ok" + + def test_claude_keeps_function_calling(self, monkeypatch): + """Claude models keep tool-based structured output (no json_schema flip).""" + monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False) + monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key") + model = get_chat_model( + "claude-haiku-4-5", provider="anthropic", max_tokens=1024 + ) + + captured, result = self._capture_structured_request( + model, + [ + { + "type": "tool_use", + "id": "toolu_1", + "name": "Pick", + "input": {"answer": "ok"}, + } + ], + ) + + assert "output_config" not in captured + assert [t["name"] for t in captured["tools"]] == ["Pick"] + assert result.answer == "ok" + + # ============================================================================= # Test _apply_auto_config # ============================================================================= @@ -2546,6 +2769,58 @@ class TestAutoConfig: assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"} assert call_kwargs["effort"] == "max" + @pytest.mark.parametrize("model", ["claude-opus-5", "claude-sonnet-5"]) + @patch("EvoScientist.llm.models.init_chat_model") + def test_anthropic_5_series_adaptive_thinking(self, mock_init, model, monkeypatch): + """Anthropic 5-series models get adaptive thinking (budget_tokens would 400).""" + mock_init.return_value = "mock_model" + monkeypatch.delenv("ANTHROPIC_BASE_URL", raising=False) + + get_chat_model(model, provider="anthropic") + + call_kwargs = mock_init.call_args[1] + assert call_kwargs["thinking"] == {"type": "adaptive", "display": "summarized"} + assert call_kwargs["effort"] == "max" + + @pytest.mark.parametrize("model", ["moonshotai/kimi-k3", "kimi-k3"]) + @patch("EvoScientist.llm.models.init_chat_model") + def test_custom_anthropic_kimi_k3_declares_thinking( + self, mock_init, model, monkeypatch + ): + """K3 via custom-anthropic declares thinking (else forced tool_choice 400s).""" + mock_init.return_value = "mock_model" + monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com") + monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key") + + get_chat_model(model, provider="custom-anthropic") + + call_kwargs = mock_init.call_args[1] + assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000} + assert call_kwargs["max_tokens"] == 16000 + + @patch("EvoScientist.llm.models.init_chat_model") + def test_kimi_coding_declares_thinking(self, mock_init): + """Kimi For Coding plan models declare thinking on the kimi-coding provider.""" + mock_init.return_value = "mock_model" + + get_chat_model("kimi-for-coding", provider="kimi-coding") + + call_kwargs = mock_init.call_args[1] + assert call_kwargs["thinking"] == {"type": "enabled", "budget_tokens": 10000} + assert call_kwargs["max_tokens"] == 16000 + + @patch("EvoScientist.llm.models.init_chat_model") + def test_custom_anthropic_non_kimi_no_thinking(self, mock_init, monkeypatch): + """Non-Kimi models on custom-anthropic still skip thinking injection.""" + mock_init.return_value = "mock_model" + monkeypatch.setenv("CUSTOM_ANTHROPIC_BASE_URL", "https://compat.example.com") + monkeypatch.setenv("CUSTOM_ANTHROPIC_API_KEY", "test-key") + + get_chat_model("glm-4.7", provider="custom-anthropic") + + call_kwargs = mock_init.call_args[1] + assert "thinking" not in call_kwargs + @patch("EvoScientist.llm.models.init_chat_model") def test_anthropic_4_6_proxy_no_thinking(self, mock_init, monkeypatch): """Anthropic 4-6 models via proxy skip thinking (history round-trip 422).""" diff --git a/uv.lock b/uv.lock index 5e9b492..c6e855b 100644 --- a/uv.lock +++ b/uv.lock @@ -952,7 +952,7 @@ wheels = [ [[package]] name = "evoscientist" -version = "0.2.3" +version = "0.2.4" source = { editable = "." } dependencies = [ { name = "deepagents", extra = ["quickjs"] },