c0106e50e7
The Kimi team noticed that traffic from Hermes Coding Plan users
identifies itself as Claude (User-Agent: claude-code/0.1.0) rather
than the actual client. They asked us to update the UA so they can
properly attribute traffic and understand how their services are
accessed — especially important as they open up to more third-party
agents.
Three code paths were sending wrong/attribution-less headers to Kimi:
1. run_agent.py — _apply_client_headers_for_base_url sent
{"User-Agent": "claude-code/0.1.0"} for api.kimi.com. Now sends
the same _AI_GATEWAY_HEADERS set used for Vercel AI Gateway:
HTTP-Referer + X-Title + HermesAgent/{version} User-Agent.
2. agent/anthropic_adapter.py — the Anthropic Messages path for
api.kimi.com/coding sent 'claude-code/0.1.0'. Now sends the same
three-header attribution set.
3. plugins/model-providers/kimi-coding/__init__.py — both kimi and
kimi_cn profiles sent a static 'hermes-agent/1.0' with no
HTTP-Referer or X-Title. Now sends the full three-header set with
a dynamic version, matching the pattern used by the gmi, fireworks,
xai, and ai-gateway provider profiles.
The attribution header set (HTTP-Referer + X-Title + User-Agent) is
the canonical Hermes pattern used for OpenRouter, Vercel AI Gateway,
Fireworks, and other providers that read these headers for traffic
attribution.
131 lines
4.6 KiB
Python
131 lines
4.6 KiB
Python
"""Kimi / Moonshot provider profiles.
|
|
|
|
Kimi has dual endpoints:
|
|
- sk-kimi-* keys → api.kimi.com/coding (Anthropic Messages API)
|
|
- legacy keys → api.moonshot.ai/v1 (OpenAI chat completions)
|
|
|
|
This module covers the chat_completions path (/v1 endpoint).
|
|
"""
|
|
|
|
from typing import Any
|
|
from urllib.parse import urlparse
|
|
|
|
from hermes_cli import __version__ as _HERMES_VERSION
|
|
from providers import register_provider
|
|
from providers.base import OMIT_TEMPERATURE, ProviderProfile
|
|
|
|
|
|
def _is_confirmed_kimi_coding_url(base_url: str) -> bool:
|
|
"""Return True only for Kimi Code's canonical HTTPS API surfaces."""
|
|
try:
|
|
parsed = urlparse(base_url)
|
|
port = parsed.port
|
|
except ValueError:
|
|
return False
|
|
return (
|
|
parsed.scheme.lower() == "https"
|
|
and (parsed.hostname or "").lower() == "api.kimi.com"
|
|
and port in (None, 443)
|
|
and parsed.username is None
|
|
and parsed.password is None
|
|
and parsed.path.rstrip("/") in {"/coding", "/coding/v1"}
|
|
and not parsed.query
|
|
and not parsed.fragment
|
|
)
|
|
|
|
|
|
class KimiProfile(ProviderProfile):
|
|
"""Kimi/Moonshot — temperature omitted, thinking xor reasoning_effort."""
|
|
|
|
def fetch_models(
|
|
self,
|
|
*,
|
|
api_key: str | None = None,
|
|
base_url: str | None = None,
|
|
timeout: float = 8.0,
|
|
) -> list[str] | None:
|
|
"""Use Kimi Code's OpenAI-compatible surface for model discovery."""
|
|
effective_base = (base_url or self.base_url or "").rstrip("/")
|
|
confirmed_coding_endpoint = _is_confirmed_kimi_coding_url(effective_base)
|
|
if confirmed_coding_endpoint and urlparse(effective_base).path.rstrip("/") == "/coding":
|
|
effective_base += "/v1"
|
|
models = super().fetch_models(
|
|
api_key=api_key,
|
|
base_url=effective_base or None,
|
|
timeout=timeout,
|
|
)
|
|
if models is None or confirmed_coding_endpoint:
|
|
return models
|
|
return [model for model in models if model.strip().lower() != "k3"]
|
|
|
|
def build_api_kwargs_extras(
|
|
self, *, reasoning_config: dict | None = None, **context
|
|
) -> tuple[dict[str, Any], dict[str, Any]]:
|
|
"""Kimi reasoning controls.
|
|
|
|
Moonshot's wire shape treats ``extra_body.thinking`` (a binary toggle)
|
|
and a top-level ``reasoning_effort`` as mutually exclusive — sending
|
|
both is at best redundant and risks "cannot specify both 'thinking' and
|
|
'reasoning_effort'" (HTTP 400). This mirrors the kimi-k2 handling on the
|
|
opencode-go relay: send effort when one is requested, otherwise fall
|
|
back to ``extra_body.thinking`` — never both.
|
|
"""
|
|
extra_body = {}
|
|
top_level = {}
|
|
|
|
if not reasoning_config or not isinstance(reasoning_config, dict):
|
|
# No config → thinking enabled, let the server pick the depth.
|
|
# (Previously also sent reasoning_effort="medium", which paired
|
|
# thinking + effort on every default call.)
|
|
extra_body["thinking"] = {"type": "enabled"}
|
|
return extra_body, top_level
|
|
|
|
enabled = reasoning_config.get("enabled", True)
|
|
if enabled is False:
|
|
extra_body["thinking"] = {"type": "disabled"}
|
|
return extra_body, top_level
|
|
|
|
# Enabled: prefer an explicit effort; only fall back to extra_body
|
|
# thinking when no recognized effort is requested.
|
|
effort = (reasoning_config.get("effort") or "").strip().lower()
|
|
if effort in {"low", "medium", "high"}:
|
|
top_level["reasoning_effort"] = effort
|
|
else:
|
|
extra_body["thinking"] = {"type": "enabled"}
|
|
|
|
return extra_body, top_level
|
|
|
|
|
|
kimi = KimiProfile(
|
|
name="kimi-coding",
|
|
aliases=("kimi", "moonshot", "kimi-for-coding"),
|
|
env_vars=("KIMI_API_KEY", "KIMI_CODING_API_KEY"),
|
|
base_url="https://api.moonshot.ai/v1",
|
|
fixed_temperature=OMIT_TEMPERATURE,
|
|
default_max_tokens=32000,
|
|
default_headers={
|
|
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
|
"X-Title": "Hermes Agent",
|
|
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
|
},
|
|
default_aux_model="kimi-k2-turbo-preview",
|
|
)
|
|
|
|
kimi_cn = KimiProfile(
|
|
name="kimi-coding-cn",
|
|
aliases=("kimi-cn", "moonshot-cn"),
|
|
env_vars=("KIMI_CN_API_KEY",),
|
|
base_url="https://api.moonshot.cn/v1",
|
|
fixed_temperature=OMIT_TEMPERATURE,
|
|
default_max_tokens=32000,
|
|
default_headers={
|
|
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
|
"X-Title": "Hermes Agent",
|
|
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
|
},
|
|
default_aux_model="kimi-k2-turbo-preview",
|
|
)
|
|
|
|
register_provider(kimi)
|
|
register_provider(kimi_cn)
|