Files
hermes-agent/hermes_cli/auth_zai_kimi.py
T

229 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Kimi Code and Z.AI endpoint auto-detection, LM Studio base-URL normalization.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
"""
from __future__ import annotations
import logging
import hashlib
from typing import Dict, Optional
from hermes_cli.auth_constants import httpx
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
# Kimi Code (kimi.com/code) issues keys prefixed "sk-kimi-" that only work
# on api.kimi.com/coding. Legacy keys from platform.moonshot.ai work on
# api.moonshot.ai/v1 (the old default). Auto-detect when user hasn't set
# KIMI_BASE_URL explicitly.
#
# Note: the base URL intentionally has NO /v1 suffix. The /coding endpoint
# speaks the Anthropic Messages protocol, and the anthropic SDK appends
# "/v1/messages" internally — so "/coding" + SDK suffix → "/coding/v1/messages"
# (the correct target). Using "/coding/v1" here would produce
# "/coding/v1/v1/messages" (a 404).
KIMI_CODE_BASE_URL = "https://api.kimi.com/coding"
def _resolve_kimi_base_url(api_key: str, default_url: str, env_override: str) -> str:
"""Return the correct Kimi base URL based on the API key prefix.
If the user has explicitly set KIMI_BASE_URL, that always wins. Otherwise, sk-kimi- prefixed
keys route to api.kimi.com/coding/v1.
"""
if env_override:
return env_override
# No key → nothing to infer from. Return default without inspecting.
if not api_key:
return default_url
if api_key.startswith("sk-kimi-"):
return KIMI_CODE_BASE_URL
return default_url
# Z.AI has separate billing for general vs coding plans, and global vs China
# endpoints. A key that works on one may return "Insufficient balance" on
# another. We probe at setup time and store the working endpoint.
# Each entry lists candidate models to try in order — newer coding plan accounts
# may only have access to recent models (glm-5.1, glm-5v-turbo) while older
# ones still use glm-4.7.
ZAI_ENDPOINTS = [
# (id, base_url, probe_models, label)
("global", "https://api.z.ai/api/paas/v4", ["glm-5"], "Global"),
("cn", "https://open.bigmodel.cn/api/paas/v4", ["glm-5"], "China"),
("coding-global", "https://api.z.ai/api/coding/paas/v4", ["glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"], "Global (Coding Plan)"),
("coding-cn", "https://open.bigmodel.cn/api/coding/paas/v4", ["glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"], "China (Coding Plan)"),
]
def _probe_single_zai_endpoint(
api_key: str, endpoint: tuple, timeout: float,
) -> Optional[Dict[str, str]]:
"""Probe a single Z.AI endpoint. Returns endpoint info dict or None.
Preserves the per-endpoint candidate-model loop: endpoints carry a ``probe_models`` LIST and
each model is tried in order until one succeeds (some plans only accept newer/older GLM slugs).
"""
ep_id, base_url, probe_models, label = endpoint
for model in probe_models:
try:
resp = httpx.post(
f"{base_url}/chat/completions",
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
},
json={
"model": model,
"stream": False,
"max_tokens": 1,
"messages": [{"role": "user", "content": "ping"}],
},
timeout=timeout,
)
if resp.status_code == 200:
logger.debug("Z.AI endpoint probe: %s (%s) model=%s OK", ep_id, base_url, model)
return {
"id": ep_id,
"base_url": base_url,
"model": model,
"label": label,
}
logger.debug("Z.AI endpoint probe: %s model=%s returned %s", ep_id, model, resp.status_code)
except Exception as exc:
logger.debug("Z.AI endpoint probe: %s model=%s failed: %s", ep_id, model, exc)
return None
def detect_zai_endpoint(api_key: str, timeout: float = 8.0) -> Optional[Dict[str, str]]:
"""Probe z.ai endpoints in parallel to find one that accepts this API key.
Returns {"id": ..., "base_url": ..., "model": ..., "label": ...} for the first working endpoint
(in ZAI_ENDPOINTS priority order), or None if all fail. For endpoints with multiple candidate
models, each worker tries its endpoint's models in order and returns the first that succeeds.
"""
from concurrent.futures import ThreadPoolExecutor, as_completed
# No `with` block: a context manager would join ALL probe threads on
# exit, defeating the early return below. shutdown(wait=False) lets the
# surviving daemon-style probes drain in the background instead of
# blocking the caller on slow/unreachable endpoints.
pool = ThreadPoolExecutor(max_workers=len(ZAI_ENDPOINTS))
try:
futures = {
pool.submit(_probe_single_zai_endpoint, api_key, ep, timeout): ep[0]
for ep in ZAI_ENDPOINTS
}
by_id = {ep_id: f for f, ep_id in futures.items()}
results: Dict[str, Dict[str, str]] = {}
for future in as_completed(futures):
ep_id = futures[future]
try:
result = future.result()
if result is not None:
results[ep_id] = result
except Exception:
pass
# Early exit in PRIORITY order: walk endpoints highest-priority
# first; if one has succeeded and every higher-priority probe
# has already finished (without success), no later completion
# can win — return now instead of waiting out slow endpoints
# (main's sequential loop also stopped at first success).
for ep in ZAI_ENDPOINTS:
if not by_id[ep[0]].done():
break # a higher-priority probe is still in flight
if ep[0] in results:
return results[ep[0]]
# All probes finished: first match in priority order, if any.
for ep in ZAI_ENDPOINTS:
if ep[0] in results:
return results[ep[0]]
return None
finally:
pool.shutdown(wait=False)
def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) -> str:
"""Return the correct Z.AI base URL by probing endpoints.
If the user has explicitly set GLM_BASE_URL, that always wins. Otherwise, probe the candidate
endpoints to find one that accepts the key. The detected endpoint is cached in provider state
(auth.json) keyed on a hash of the API key so subsequent starts skip the probe.
"""
from hermes_cli.auth import _auth_store_lock, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state, detect_zai_endpoint
if env_override:
return env_override
# No API key set → don't probe (would fire N×M HTTPS requests with an
# empty Bearer token, all returning 401). This path is hit during
# auxiliary-client auto-detection when the user has no Z.AI credentials
# at all — the caller discards the result immediately, so the probe is
# pure latency for every AIAgent construction.
if not api_key:
return default_url
# Check provider-state cache for a previously-detected endpoint.
auth_store = _load_auth_store()
state = _load_provider_state(auth_store, "zai") or {}
cached = state.get("detected_endpoint")
if isinstance(cached, dict) and cached.get("base_url"):
key_hash = cached.get("key_hash", "")
if key_hash == hashlib.sha256(api_key.encode()).hexdigest()[:16]:
logger.debug("Z.AI: using cached endpoint %s", cached["base_url"])
return cached["base_url"]
# Probe — may take up to ~8s per endpoint.
detected = detect_zai_endpoint(api_key)
if detected and detected.get("base_url"):
# Persist the detection result keyed on the API key hash.
key_hash = hashlib.sha256(api_key.encode()).hexdigest()[:16]
detected_endpoint = {
"base_url": detected["base_url"],
"endpoint_id": detected.get("id", ""),
"model": detected.get("model", ""),
"label": detected.get("label", ""),
"key_hash": key_hash,
}
# Persist failure (disk full, permissions, lock timeout) must not
# break resolution — detection already succeeded; worst case the
# next start re-probes.
try:
with _auth_store_lock():
# Reload auth_store under lock to avoid overwriting concurrent changes
auth_store = _load_auth_store()
state_under_lock = _load_provider_state(auth_store, "zai") or {}
state_under_lock["detected_endpoint"] = detected_endpoint
# set_active=False: this runs from credential-pool env seeding
# (agent/credential_pool.py) for ANY user with a Z.AI key in env,
# and caching a probe result must not flip their active provider.
_store_provider_state(auth_store, "zai", state_under_lock, set_active=False)
_save_auth_store(auth_store)
except Exception as exc:
logger.warning("Z.AI: could not persist detected endpoint (%s); will re-probe next start", exc)
logger.info("Z.AI: auto-detected endpoint %s (%s)", detected["label"], detected["base_url"])
return detected["base_url"]
logger.debug("Z.AI: probe failed, falling back to default %s", default_url)
return default_url
def _normalize_lmstudio_runtime_base_url(base_url: str) -> str:
"""Return the OpenAI-compatible LM Studio runtime base URL.
LM Studio's native management API lives under ``/api/v1`` while its OpenAI-compatible chat
endpoint lives under ``/v1``. Users often paste either form into ``LM_BASE_URL`` or
``model.base_url``; normalize before the OpenAI SDK appends ``/chat/completions``.
"""
root = str(base_url or "").strip().rstrip("/")
for suffix in ("/api/v1", "/api", "/v1"):
if root.endswith(suffix):
root = root[: -len(suffix)].rstrip("/")
break
return (root or "http://127.0.0.1:1234") + "/v1"