feat(llm): add Kimi K3 support (#367)
* feat(context-window): add Kimi K3 model with 1M context window * feat(openrouter): implement structured output for Kimi K3 and add 429 retry handling * Refactor code structure for improved readability and maintainability
This commit is contained in:
@@ -48,6 +48,9 @@ _KNOWN_MODEL_FAMILIES: list[tuple[str, int]] = [
|
||||
("gpt-5.5", 1_050_000),
|
||||
# Google Gemini 3.x family — flash, flash-lite, pro (1.05M). Excludes 2.5.
|
||||
("gemini-3", 1_050_000),
|
||||
# Moonshot Kimi K3 — 1M context; covers bare ``kimi-k3`` (native Moonshot),
|
||||
# OpenRouter ``moonshotai/kimi-k3``, and dated slugs like ``kimi-k3-20260715``.
|
||||
("kimi-k3", 1_048_576),
|
||||
# Moonshot Kimi K2 family — k2.5, k2.6, k2-thinking, k2-thinking-turbo
|
||||
("kimi-k2", 262_000),
|
||||
# Zhipu GLM-5 family — base, 5.1, 5-turbo, 5v-turbo, etc.
|
||||
|
||||
@@ -31,6 +31,7 @@ from .patches import (
|
||||
_patch_ccproxy_system_to_developer,
|
||||
_patch_openai_compat_content,
|
||||
_patch_openrouter_strip_responses_reasoning,
|
||||
_patch_openrouter_structured_output,
|
||||
)
|
||||
|
||||
_MINIMAX_ANTHROPIC_BASE_URL = "https://api.minimaxi.com/anthropic"
|
||||
@@ -137,6 +138,13 @@ _FALSEY_ENV_VALUES = {"0", "false", "no", "off"}
|
||||
# capped to this many below. https://openrouter.ai/docs/app-attribution
|
||||
_OPENROUTER_MAX_CATEGORIES_PER_REQUEST = 2
|
||||
|
||||
# Moonshot rejects a forced tool choice while thinking is enabled, and kimi-k3
|
||||
# cannot disable thinking — structured output must use json_schema there.
|
||||
# Moonshot-specific: do NOT widen to other mandatory-reasoning models.
|
||||
_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset(
|
||||
{"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"}
|
||||
)
|
||||
|
||||
# Model registry: list of (short_name, model_id, provider)
|
||||
# Allows same short_name across different providers.
|
||||
_MODEL_ENTRIES: list[tuple[str, str, str]] = [
|
||||
@@ -227,6 +235,7 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
|
||||
("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"),
|
||||
("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"),
|
||||
("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"),
|
||||
("kimi-k3", "moonshotai/kimi-k3", "openrouter"),
|
||||
("kimi-k2.6", "moonshotai/kimi-k2.6", "openrouter"),
|
||||
("glm-5.2", "z-ai/glm-5.2", "openrouter"),
|
||||
("glm-5v-turbo", "z-ai/glm-5v-turbo", "openrouter"),
|
||||
@@ -291,6 +300,7 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
|
||||
("deepseek-r1", "deepseek-reasoner", "deepseek"),
|
||||
("deepseek-v3", "deepseek-chat", "deepseek"),
|
||||
# Moonshot (OpenAI-compatible)
|
||||
("kimi-k3", "kimi-k3", "moonshot"),
|
||||
("kimi-k2.6", "kimi-k2.6", "moonshot"),
|
||||
("kimi-k2.5", "kimi-k2.5", "moonshot"),
|
||||
("kimi-k2-thinking", "kimi-k2-thinking", "moonshot"),
|
||||
@@ -378,6 +388,22 @@ def _apply_openrouter_anthropic_prompt_cache(
|
||||
kwargs.setdefault("model_kwargs", {})["cache_control"] = {"type": "ephemeral"}
|
||||
|
||||
|
||||
def _enable_openrouter_429_retry(chat_model: Any) -> None:
|
||||
"""Add 429 to the OpenRouter SDK's retryable status codes (default ["5XX"]).
|
||||
|
||||
Upstream rate limits ("temporarily rate-limited upstream", whose
|
||||
Retry-After the SDK backoff already honors) otherwise fail the run outright.
|
||||
"""
|
||||
sdk_config = getattr(getattr(chat_model, "client", None), "sdk_configuration", None)
|
||||
retry_config: Any = getattr(sdk_config, "retry_config", None)
|
||||
# Skip the UNSET sentinel (max_retries=0) and explicit caller overrides.
|
||||
if not hasattr(retry_config, "status_codes_override"):
|
||||
return
|
||||
if retry_config.status_codes_override:
|
||||
return
|
||||
retry_config.status_codes_override = ["429", "5XX"]
|
||||
|
||||
|
||||
def _apply_auto_config(
|
||||
provider: str,
|
||||
model_id: str,
|
||||
@@ -566,10 +592,12 @@ def get_chat_model(
|
||||
# from history, causing error 20015 on multi-turn requests.
|
||||
if provider == "siliconflow":
|
||||
kwargs.setdefault("extra_body", {})["enable_thinking"] = False
|
||||
# Moonshot: disable thinking for all models to prevent LangChain from dropping
|
||||
# reasoning_content, which causes multi-turn conversation errors (error 20015).
|
||||
# Even native thinking models like kimi-k2-thinking operate in non-thinking mode.
|
||||
if provider == "moonshot":
|
||||
# Moonshot: disable thinking for pre-K3 models to prevent LangChain from
|
||||
# dropping reasoning_content, which causes multi-turn conversation errors
|
||||
# (error 20015). Even native thinking models like kimi-k2-thinking operate
|
||||
# in non-thinking mode. kimi-k3+ is exempt: always-thinking, and
|
||||
# Moonshot's K3 guide forbids the K2.x `thinking` parameter for it.
|
||||
if provider == "moonshot" and not model_id.startswith("kimi-k3"):
|
||||
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
|
||||
provider = "openai"
|
||||
|
||||
@@ -585,6 +613,9 @@ def get_chat_model(
|
||||
# passback (OpenRouter's `/responses` beta is stateless, store=false —
|
||||
# "Item with id 'rs_...' not found"); the patch strips them on passback,
|
||||
# so enabling `summary` is safe. See langchain-ai/langchain#37777.
|
||||
# Note: mandatory-reasoning endpoints (kimi-k3, grok-4.5, …) reject
|
||||
# effort "none" with HTTP 400 — that error is surfaced to the user
|
||||
# as-is; pick a real effort (low+) for those models.
|
||||
effort = _resolve_reasoning_effort("high")
|
||||
kwargs.setdefault("reasoning", {"effort": effort, "summary": "auto"})
|
||||
# App attribution (issue #339): identify EvoScientist to OpenRouter so
|
||||
@@ -634,6 +665,7 @@ def get_chat_model(
|
||||
if _app_categories:
|
||||
kwargs.setdefault("app_categories", _app_categories)
|
||||
_patch_openrouter_strip_responses_reasoning()
|
||||
_patch_openrouter_structured_output()
|
||||
|
||||
# Anthropic-routed providers → route through Anthropic provider with base_url
|
||||
elif provider in _ANTHROPIC_ROUTED_PROVIDERS:
|
||||
@@ -719,6 +751,9 @@ def get_chat_model(
|
||||
if _is_openai_proxy:
|
||||
_patch_ccproxy_system_to_developer(chat_model)
|
||||
|
||||
if provider == "openrouter":
|
||||
_enable_openrouter_429_retry(chat_model)
|
||||
|
||||
apply_known_context_window(chat_model)
|
||||
|
||||
return chat_model
|
||||
|
||||
@@ -879,6 +879,52 @@ def _patch_openrouter_strip_responses_reasoning() -> None:
|
||||
pass
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Patch (lazy, OpenRouter only): default structured output to json_schema for
|
||||
# Moonshot's always-thinking models (kimi-k3).
|
||||
#
|
||||
# Moonshot rejects a forced tool choice while thinking is enabled with HTTP 400:
|
||||
# "tool_choice 'specified' is incompatible with thinking enabled"
|
||||
# and kimi-k3's thinking cannot be disabled, so the function_calling default of
|
||||
# ChatOpenRouter.with_structured_output fails every structured-output call
|
||||
# (LLMToolSelectorMiddleware included). These endpoints support
|
||||
# response_format json_schema, which needs no tool_choice — route them there.
|
||||
# Moonshot-specific: other mandatory-reasoning models keep function_calling.
|
||||
# Gated on the instance's model_name, so copies behave correctly and other
|
||||
# models keep the function_calling default.
|
||||
# ---------------------------------------------------------------------------
|
||||
_openrouter_structured_output_patched = False
|
||||
|
||||
|
||||
def _patch_openrouter_structured_output() -> None:
|
||||
global _openrouter_structured_output_patched
|
||||
if _openrouter_structured_output_patched:
|
||||
return
|
||||
try:
|
||||
from langchain_openrouter import ChatOpenRouter
|
||||
|
||||
_orig = ChatOpenRouter.with_structured_output
|
||||
|
||||
def _patched(
|
||||
self: Any,
|
||||
schema: Any = None,
|
||||
*,
|
||||
method: str = "function_calling",
|
||||
**kwargs: Any,
|
||||
) -> Any:
|
||||
if method == "function_calling":
|
||||
from .models import _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS
|
||||
|
||||
if self.model_name in _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS:
|
||||
method = "json_schema"
|
||||
return _orig(self, schema, method=method, **kwargs)
|
||||
|
||||
ChatOpenRouter.with_structured_output = _patched
|
||||
_openrouter_structured_output_patched = True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Patch: forward CLI's live (model, model_provider) into deepagents'
|
||||
# start_async_task / update_async_task tool calls so the deployed graph
|
||||
|
||||
@@ -450,6 +450,212 @@ class TestThirdPartyRouting:
|
||||
call_kwargs = mock_init.call_args[1]
|
||||
assert call_kwargs["reasoning"] == {"effort": "medium", "summary": "auto"}
|
||||
|
||||
@patch("EvoScientist.llm.models.init_chat_model")
|
||||
def test_moonshot_thinking_disable_exempts_kimi_k3(self, mock_init, monkeypatch):
|
||||
"""Native Moonshot: K3 must not receive the K2.x thinking-disable field.
|
||||
|
||||
Moonshot's K3 guide forbids the K2.x `thinking` parameter (K3 is
|
||||
always-thinking); other Moonshot models keep the disable that guards
|
||||
against multi-turn error 20015.
|
||||
"""
|
||||
mock_init.return_value = "mock_model"
|
||||
monkeypatch.setenv("MOONSHOT_API_KEY", "ms-key")
|
||||
|
||||
get_chat_model("kimi-k3", provider="moonshot")
|
||||
extra_body = mock_init.call_args[1].get("extra_body") or {}
|
||||
assert "thinking" not in extra_body
|
||||
|
||||
get_chat_model("kimi-k2.6", provider="moonshot")
|
||||
extra_body = mock_init.call_args[1]["extra_body"]
|
||||
assert extra_body["thinking"] == {"type": "disabled"}
|
||||
|
||||
# --- OpenRouter upstream 429 retry ---
|
||||
|
||||
def test_openrouter_429_added_to_retryable_status_codes(self, monkeypatch):
|
||||
"""Upstream 429s must become retryable on the real SDK client.
|
||||
|
||||
The openrouter SDK hardcodes per-operation retryable statuses to
|
||||
["5XX"], so a launch-day "temporarily rate-limited upstream" 429
|
||||
(Retry-After: 1) fails the run outright instead of being retried.
|
||||
"""
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
||||
|
||||
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
|
||||
|
||||
retry_config = model.client.sdk_configuration.retry_config
|
||||
assert retry_config.status_codes_override == ["429", "5XX"]
|
||||
|
||||
def test_openrouter_429_override_not_injected_when_retries_disabled(
|
||||
self, monkeypatch
|
||||
):
|
||||
"""max_retries=0 leaves the SDK retry config UNSET — no 429 override.
|
||||
|
||||
Note this only asserts our override is absent; the SDK still applies
|
||||
its own per-operation default (backoff on 5XX) when the config is
|
||||
UNSET, so retries as such are not fully disabled at the SDK level.
|
||||
"""
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
||||
|
||||
model = get_chat_model("x-ai/grok-4.3", provider="openrouter", max_retries=0)
|
||||
|
||||
retry_config = model.client.sdk_configuration.retry_config
|
||||
assert getattr(retry_config, "status_codes_override", None) is None
|
||||
|
||||
def test_openrouter_429_retried_on_the_wire(self, monkeypatch):
|
||||
"""End-to-end: a 429 with Retry-After is retried and the retry succeeds."""
|
||||
import httpx
|
||||
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
||||
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
calls["n"] += 1
|
||||
if calls["n"] == 1:
|
||||
return httpx.Response(
|
||||
429,
|
||||
headers={"Retry-After": "1"},
|
||||
json={"error": {"message": "Provider returned error", "code": 429}},
|
||||
)
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "gen-1",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "moonshotai/kimi-k3",
|
||||
"system_fingerprint": "fp-test",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "ok"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
},
|
||||
)
|
||||
|
||||
model.client.sdk_configuration.client = httpx.Client(
|
||||
transport=httpx.MockTransport(handler)
|
||||
)
|
||||
|
||||
result = model.invoke("hi")
|
||||
|
||||
assert calls["n"] == 2
|
||||
assert result.content == "ok"
|
||||
|
||||
# --- OpenRouter structured output vs mandatory reasoning ---
|
||||
|
||||
@staticmethod
|
||||
def _capture_structured_request(model, structured, response_message):
|
||||
"""Invoke a structured-output runnable against a capturing transport."""
|
||||
import json
|
||||
|
||||
import httpx
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
captured.update(json.loads(request.content.decode()))
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "gen-1",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "m",
|
||||
"system_fingerprint": "fp-test",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": response_message,
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
},
|
||||
)
|
||||
|
||||
model.client.sdk_configuration.client = httpx.Client(
|
||||
transport=httpx.MockTransport(handler)
|
||||
)
|
||||
result = structured.invoke("pick tools")
|
||||
return captured, result
|
||||
|
||||
def test_openrouter_structured_output_json_schema_for_mandatory_model(
|
||||
self, monkeypatch
|
||||
):
|
||||
"""with_structured_output must not force tool_choice on kimi-k3.
|
||||
|
||||
Moonshot rejects a forced tool choice with HTTP 400 "tool_choice
|
||||
'specified' is incompatible with thinking enabled", and kimi-k3's
|
||||
thinking cannot be disabled — so the default function_calling method
|
||||
400s every structured-output call (LLMToolSelectorMiddleware included).
|
||||
The json_schema method (response_format) is supported and needs none.
|
||||
"""
|
||||
from pydantic import BaseModel
|
||||
|
||||
class ToolSelection(BaseModel):
|
||||
tools: list[str]
|
||||
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
||||
|
||||
# The dated canonical_slug is routable too and must be equally covered.
|
||||
for model_name in ("moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"):
|
||||
model = get_chat_model(model_name, provider="openrouter")
|
||||
structured = model.with_structured_output(ToolSelection)
|
||||
|
||||
captured, result = self._capture_structured_request(
|
||||
model,
|
||||
structured,
|
||||
{"role": "assistant", "content": '{"tools": ["tavily_search"]}'},
|
||||
)
|
||||
|
||||
assert "tool_choice" not in captured, model_name
|
||||
assert captured["response_format"]["type"] == "json_schema", model_name
|
||||
assert result == ToolSelection(tools=["tavily_search"])
|
||||
|
||||
def test_openrouter_structured_output_default_for_other_models(self, monkeypatch):
|
||||
"""Non-Moonshot models keep the function_calling default.
|
||||
|
||||
Includes always-thinking models like grok-4.5 — the forced tool_choice
|
||||
restriction is Moonshot-specific, so the json_schema rerouting must
|
||||
stay limited to _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS.
|
||||
"""
|
||||
from pydantic import BaseModel
|
||||
|
||||
class ToolSelection(BaseModel):
|
||||
tools: list[str]
|
||||
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
||||
|
||||
for model_name in ("x-ai/grok-4.3", "x-ai/grok-4.5"):
|
||||
model = get_chat_model(model_name, provider="openrouter")
|
||||
structured = model.with_structured_output(ToolSelection)
|
||||
|
||||
captured, result = self._capture_structured_request(
|
||||
model,
|
||||
structured,
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "ToolSelection",
|
||||
"arguments": '{"tools": ["tavily_search"]}',
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
)
|
||||
|
||||
assert captured.get("tool_choice"), model_name
|
||||
assert "response_format" not in captured, model_name
|
||||
assert result == ToolSelection(tools=["tavily_search"])
|
||||
|
||||
# --- OpenRouter app attribution (issue #339) ---
|
||||
|
||||
_APP_ATTR_ENV = (
|
||||
|
||||
Reference in New Issue
Block a user