feat(llm): add Kimi K3 support (#367)

* feat(context-window): add Kimi K3 model with 1M context window

* feat(openrouter): implement structured output for Kimi K3 and add 429 retry handling

* Refactor code structure for improved readability and maintainability
This commit is contained in:
Xi Zhang
2026-07-18 00:12:08 +01:00
committed by GitHub
parent 06a9511bdd
commit 042da63d54
5 changed files with 2237 additions and 1771 deletions
+3
View File
@@ -48,6 +48,9 @@ _KNOWN_MODEL_FAMILIES: list[tuple[str, int]] = [
("gpt-5.5", 1_050_000),
# Google Gemini 3.x family — flash, flash-lite, pro (1.05M). Excludes 2.5.
("gemini-3", 1_050_000),
# Moonshot Kimi K3 — 1M context; covers bare ``kimi-k3`` (native Moonshot),
# OpenRouter ``moonshotai/kimi-k3``, and dated slugs like ``kimi-k3-20260715``.
("kimi-k3", 1_048_576),
# Moonshot Kimi K2 family — k2.5, k2.6, k2-thinking, k2-thinking-turbo
("kimi-k2", 262_000),
# Zhipu GLM-5 family — base, 5.1, 5-turbo, 5v-turbo, etc.
+39 -4
View File
@@ -31,6 +31,7 @@ from .patches import (
_patch_ccproxy_system_to_developer,
_patch_openai_compat_content,
_patch_openrouter_strip_responses_reasoning,
_patch_openrouter_structured_output,
)
_MINIMAX_ANTHROPIC_BASE_URL = "https://api.minimaxi.com/anthropic"
@@ -137,6 +138,13 @@ _FALSEY_ENV_VALUES = {"0", "false", "no", "off"}
# capped to this many below. https://openrouter.ai/docs/app-attribution
_OPENROUTER_MAX_CATEGORIES_PER_REQUEST = 2
# Moonshot rejects a forced tool choice while thinking is enabled, and kimi-k3
# cannot disable thinking — structured output must use json_schema there.
# Moonshot-specific: do NOT widen to other mandatory-reasoning models.
_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset(
{"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"}
)
# Model registry: list of (short_name, model_id, provider)
# Allows same short_name across different providers.
_MODEL_ENTRIES: list[tuple[str, str, str]] = [
@@ -227,6 +235,7 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"),
("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"),
("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"),
("kimi-k3", "moonshotai/kimi-k3", "openrouter"),
("kimi-k2.6", "moonshotai/kimi-k2.6", "openrouter"),
("glm-5.2", "z-ai/glm-5.2", "openrouter"),
("glm-5v-turbo", "z-ai/glm-5v-turbo", "openrouter"),
@@ -291,6 +300,7 @@ _MODEL_ENTRIES: list[tuple[str, str, str]] = [
("deepseek-r1", "deepseek-reasoner", "deepseek"),
("deepseek-v3", "deepseek-chat", "deepseek"),
# Moonshot (OpenAI-compatible)
("kimi-k3", "kimi-k3", "moonshot"),
("kimi-k2.6", "kimi-k2.6", "moonshot"),
("kimi-k2.5", "kimi-k2.5", "moonshot"),
("kimi-k2-thinking", "kimi-k2-thinking", "moonshot"),
@@ -378,6 +388,22 @@ def _apply_openrouter_anthropic_prompt_cache(
kwargs.setdefault("model_kwargs", {})["cache_control"] = {"type": "ephemeral"}
def _enable_openrouter_429_retry(chat_model: Any) -> None:
"""Add 429 to the OpenRouter SDK's retryable status codes (default ["5XX"]).
Upstream rate limits ("temporarily rate-limited upstream", whose
Retry-After the SDK backoff already honors) otherwise fail the run outright.
"""
sdk_config = getattr(getattr(chat_model, "client", None), "sdk_configuration", None)
retry_config: Any = getattr(sdk_config, "retry_config", None)
# Skip the UNSET sentinel (max_retries=0) and explicit caller overrides.
if not hasattr(retry_config, "status_codes_override"):
return
if retry_config.status_codes_override:
return
retry_config.status_codes_override = ["429", "5XX"]
def _apply_auto_config(
provider: str,
model_id: str,
@@ -566,10 +592,12 @@ def get_chat_model(
# from history, causing error 20015 on multi-turn requests.
if provider == "siliconflow":
kwargs.setdefault("extra_body", {})["enable_thinking"] = False
# Moonshot: disable thinking for all models to prevent LangChain from dropping
# reasoning_content, which causes multi-turn conversation errors (error 20015).
# Even native thinking models like kimi-k2-thinking operate in non-thinking mode.
if provider == "moonshot":
# Moonshot: disable thinking for pre-K3 models to prevent LangChain from
# dropping reasoning_content, which causes multi-turn conversation errors
# (error 20015). Even native thinking models like kimi-k2-thinking operate
# in non-thinking mode. kimi-k3+ is exempt: always-thinking, and
# Moonshot's K3 guide forbids the K2.x `thinking` parameter for it.
if provider == "moonshot" and not model_id.startswith("kimi-k3"):
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
provider = "openai"
@@ -585,6 +613,9 @@ def get_chat_model(
# passback (OpenRouter's `/responses` beta is stateless, store=false —
# "Item with id 'rs_...' not found"); the patch strips them on passback,
# so enabling `summary` is safe. See langchain-ai/langchain#37777.
# Note: mandatory-reasoning endpoints (kimi-k3, grok-4.5, …) reject
# effort "none" with HTTP 400 — that error is surfaced to the user
# as-is; pick a real effort (low+) for those models.
effort = _resolve_reasoning_effort("high")
kwargs.setdefault("reasoning", {"effort": effort, "summary": "auto"})
# App attribution (issue #339): identify EvoScientist to OpenRouter so
@@ -634,6 +665,7 @@ def get_chat_model(
if _app_categories:
kwargs.setdefault("app_categories", _app_categories)
_patch_openrouter_strip_responses_reasoning()
_patch_openrouter_structured_output()
# Anthropic-routed providers → route through Anthropic provider with base_url
elif provider in _ANTHROPIC_ROUTED_PROVIDERS:
@@ -719,6 +751,9 @@ def get_chat_model(
if _is_openai_proxy:
_patch_ccproxy_system_to_developer(chat_model)
if provider == "openrouter":
_enable_openrouter_429_retry(chat_model)
apply_known_context_window(chat_model)
return chat_model
+46
View File
@@ -879,6 +879,52 @@ def _patch_openrouter_strip_responses_reasoning() -> None:
pass
# ---------------------------------------------------------------------------
# Patch (lazy, OpenRouter only): default structured output to json_schema for
# Moonshot's always-thinking models (kimi-k3).
#
# Moonshot rejects a forced tool choice while thinking is enabled with HTTP 400:
# "tool_choice 'specified' is incompatible with thinking enabled"
# and kimi-k3's thinking cannot be disabled, so the function_calling default of
# ChatOpenRouter.with_structured_output fails every structured-output call
# (LLMToolSelectorMiddleware included). These endpoints support
# response_format json_schema, which needs no tool_choice — route them there.
# Moonshot-specific: other mandatory-reasoning models keep function_calling.
# Gated on the instance's model_name, so copies behave correctly and other
# models keep the function_calling default.
# ---------------------------------------------------------------------------
_openrouter_structured_output_patched = False
def _patch_openrouter_structured_output() -> None:
global _openrouter_structured_output_patched
if _openrouter_structured_output_patched:
return
try:
from langchain_openrouter import ChatOpenRouter
_orig = ChatOpenRouter.with_structured_output
def _patched(
self: Any,
schema: Any = None,
*,
method: str = "function_calling",
**kwargs: Any,
) -> Any:
if method == "function_calling":
from .models import _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS
if self.model_name in _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS:
method = "json_schema"
return _orig(self, schema, method=method, **kwargs)
ChatOpenRouter.with_structured_output = _patched
_openrouter_structured_output_patched = True
except Exception:
pass
# ---------------------------------------------------------------------------
# Patch: forward CLI's live (model, model_provider) into deepagents'
# start_async_task / update_async_task tool calls so the deployed graph
+206
View File
@@ -450,6 +450,212 @@ class TestThirdPartyRouting:
call_kwargs = mock_init.call_args[1]
assert call_kwargs["reasoning"] == {"effort": "medium", "summary": "auto"}
@patch("EvoScientist.llm.models.init_chat_model")
def test_moonshot_thinking_disable_exempts_kimi_k3(self, mock_init, monkeypatch):
"""Native Moonshot: K3 must not receive the K2.x thinking-disable field.
Moonshot's K3 guide forbids the K2.x `thinking` parameter (K3 is
always-thinking); other Moonshot models keep the disable that guards
against multi-turn error 20015.
"""
mock_init.return_value = "mock_model"
monkeypatch.setenv("MOONSHOT_API_KEY", "ms-key")
get_chat_model("kimi-k3", provider="moonshot")
extra_body = mock_init.call_args[1].get("extra_body") or {}
assert "thinking" not in extra_body
get_chat_model("kimi-k2.6", provider="moonshot")
extra_body = mock_init.call_args[1]["extra_body"]
assert extra_body["thinking"] == {"type": "disabled"}
# --- OpenRouter upstream 429 retry ---
def test_openrouter_429_added_to_retryable_status_codes(self, monkeypatch):
"""Upstream 429s must become retryable on the real SDK client.
The openrouter SDK hardcodes per-operation retryable statuses to
["5XX"], so a launch-day "temporarily rate-limited upstream" 429
(Retry-After: 1) fails the run outright instead of being retried.
"""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
retry_config = model.client.sdk_configuration.retry_config
assert retry_config.status_codes_override == ["429", "5XX"]
def test_openrouter_429_override_not_injected_when_retries_disabled(
self, monkeypatch
):
"""max_retries=0 leaves the SDK retry config UNSET — no 429 override.
Note this only asserts our override is absent; the SDK still applies
its own per-operation default (backoff on 5XX) when the config is
UNSET, so retries as such are not fully disabled at the SDK level.
"""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
model = get_chat_model("x-ai/grok-4.3", provider="openrouter", max_retries=0)
retry_config = model.client.sdk_configuration.retry_config
assert getattr(retry_config, "status_codes_override", None) is None
def test_openrouter_429_retried_on_the_wire(self, monkeypatch):
"""End-to-end: a 429 with Retry-After is retried and the retry succeeds."""
import httpx
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
model = get_chat_model("moonshotai/kimi-k3", provider="openrouter")
calls = {"n": 0}
def handler(request: httpx.Request) -> httpx.Response:
calls["n"] += 1
if calls["n"] == 1:
return httpx.Response(
429,
headers={"Retry-After": "1"},
json={"error": {"message": "Provider returned error", "code": 429}},
)
return httpx.Response(
200,
json={
"id": "gen-1",
"object": "chat.completion",
"created": 1,
"model": "moonshotai/kimi-k3",
"system_fingerprint": "fp-test",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok"},
"finish_reason": "stop",
}
],
},
)
model.client.sdk_configuration.client = httpx.Client(
transport=httpx.MockTransport(handler)
)
result = model.invoke("hi")
assert calls["n"] == 2
assert result.content == "ok"
# --- OpenRouter structured output vs mandatory reasoning ---
@staticmethod
def _capture_structured_request(model, structured, response_message):
"""Invoke a structured-output runnable against a capturing transport."""
import json
import httpx
captured: dict = {}
def handler(request: httpx.Request) -> httpx.Response:
captured.update(json.loads(request.content.decode()))
return httpx.Response(
200,
json={
"id": "gen-1",
"object": "chat.completion",
"created": 1,
"model": "m",
"system_fingerprint": "fp-test",
"choices": [
{
"index": 0,
"message": response_message,
"finish_reason": "stop",
}
],
},
)
model.client.sdk_configuration.client = httpx.Client(
transport=httpx.MockTransport(handler)
)
result = structured.invoke("pick tools")
return captured, result
def test_openrouter_structured_output_json_schema_for_mandatory_model(
self, monkeypatch
):
"""with_structured_output must not force tool_choice on kimi-k3.
Moonshot rejects a forced tool choice with HTTP 400 "tool_choice
'specified' is incompatible with thinking enabled", and kimi-k3's
thinking cannot be disabled — so the default function_calling method
400s every structured-output call (LLMToolSelectorMiddleware included).
The json_schema method (response_format) is supported and needs none.
"""
from pydantic import BaseModel
class ToolSelection(BaseModel):
tools: list[str]
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
# The dated canonical_slug is routable too and must be equally covered.
for model_name in ("moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"):
model = get_chat_model(model_name, provider="openrouter")
structured = model.with_structured_output(ToolSelection)
captured, result = self._capture_structured_request(
model,
structured,
{"role": "assistant", "content": '{"tools": ["tavily_search"]}'},
)
assert "tool_choice" not in captured, model_name
assert captured["response_format"]["type"] == "json_schema", model_name
assert result == ToolSelection(tools=["tavily_search"])
def test_openrouter_structured_output_default_for_other_models(self, monkeypatch):
"""Non-Moonshot models keep the function_calling default.
Includes always-thinking models like grok-4.5 — the forced tool_choice
restriction is Moonshot-specific, so the json_schema rerouting must
stay limited to _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS.
"""
from pydantic import BaseModel
class ToolSelection(BaseModel):
tools: list[str]
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
for model_name in ("x-ai/grok-4.3", "x-ai/grok-4.5"):
model = get_chat_model(model_name, provider="openrouter")
structured = model.with_structured_output(ToolSelection)
captured, result = self._capture_structured_request(
model,
structured,
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {
"name": "ToolSelection",
"arguments": '{"tools": ["tavily_search"]}',
},
}
],
},
)
assert captured.get("tool_choice"), model_name
assert "response_format" not in captured, model_name
assert result == ToolSelection(tools=["tavily_search"])
# --- OpenRouter app attribution (issue #339) ---
_APP_ATTR_ENV = (
Generated
+1943 -1767
View File
File diff suppressed because it is too large Load Diff