refactor(plugins/model-providers): reuse agent.reasoning_effort clamps, per-model dict tables, compact profiles
This commit is contained in:
@@ -14,10 +14,11 @@ from providers.base import ProviderProfile, _profile_user_agent
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_ACTUAL_BASE_URL = "https://api.actual.inc/v1"
|
||||
DEFAULT_ACTUAL_LOCAL_BASE_URL = "http://127.0.0.1:8080/v1"
|
||||
_LOCAL_HOSTS = {"localhost", "127.0.0.1", "::1", "0.0.0.0"}
|
||||
|
||||
|
||||
def _normalize_actual_base_url(base_url: str) -> str:
|
||||
"""Append /v1 to a bare hosted or local host; pass anything else through."""
|
||||
url = str(base_url or "").strip().rstrip("/")
|
||||
if not url:
|
||||
return DEFAULT_ACTUAL_BASE_URL
|
||||
@@ -27,42 +28,28 @@ def _normalize_actual_base_url(base_url: str) -> str:
|
||||
path = parsed.path.rstrip("/")
|
||||
except Exception:
|
||||
return url
|
||||
if host == "api.actual.inc" and path in {"", "/"}:
|
||||
return url + "/v1"
|
||||
if host in {"localhost", "127.0.0.1", "::1", "0.0.0.0"} and path in {"", "/"}:
|
||||
if (host == "api.actual.inc" or host in _LOCAL_HOSTS) and path in {"", "/"}:
|
||||
return url + "/v1"
|
||||
return url
|
||||
|
||||
|
||||
class ActualProfile(ProviderProfile):
|
||||
"""Actual Computer provider.
|
||||
|
||||
Hosted inference defaults to api.actual.inc. Local inference is exposed by
|
||||
the Actual client only when it runs in offline mode, so users opt into it by
|
||||
setting ACTUAL_BASE_URL to the local API URL.
|
||||
"""
|
||||
"""Actual Computer: hosted at api.actual.inc; local (offline-mode client)
|
||||
inference opted into via ACTUAL_BASE_URL."""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
from hermes_cli.urllib_security import open_credentialed_url
|
||||
|
||||
base_url = _normalize_actual_base_url(
|
||||
os.getenv("ACTUAL_BASE_URL", "").strip() or base_url or self.base_url
|
||||
)
|
||||
if not base_url:
|
||||
return None
|
||||
|
||||
req = urllib.request.Request(base_url + "/models")
|
||||
if api_key:
|
||||
req.add_header("Authorization", f"Bearer {api_key}")
|
||||
req.add_header("Accept", "application/json")
|
||||
req.add_header("User-Agent", _profile_user_agent())
|
||||
|
||||
from hermes_cli.urllib_security import open_credentialed_url
|
||||
|
||||
try:
|
||||
with open_credentialed_url(req, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read().decode())
|
||||
|
||||
@@ -1,8 +1,4 @@
|
||||
"""Vercel AI Gateway provider profile.
|
||||
|
||||
AI Gateway routes to multiple backends. Hermes sends attribution
|
||||
headers and full reasoning config passthrough.
|
||||
"""
|
||||
"""Vercel AI Gateway provider profile: attribution headers + reasoning passthrough."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
@@ -14,18 +10,12 @@ class VercelAIGatewayProfile(ProviderProfile):
|
||||
"""Vercel AI Gateway — attribution headers + reasoning passthrough."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self,
|
||||
*,
|
||||
reasoning_config: dict | None = None,
|
||||
supports_reasoning: bool = True,
|
||||
**ctx: Any,
|
||||
self, *, reasoning_config: dict | None = None, supports_reasoning: bool = True, **ctx: Any
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
if supports_reasoning and reasoning_config is not None:
|
||||
extra_body["reasoning"] = dict(reasoning_config)
|
||||
elif supports_reasoning:
|
||||
extra_body["reasoning"] = {"enabled": True, "effort": "medium"}
|
||||
return extra_body, {}
|
||||
if not supports_reasoning:
|
||||
return {}, {}
|
||||
reasoning = dict(reasoning_config) if reasoning_config is not None else {"enabled": True, "effort": "medium"}
|
||||
return {"reasoning": reasoning}, {}
|
||||
|
||||
|
||||
vercel = VercelAIGatewayProfile(
|
||||
|
||||
@@ -1,18 +1,8 @@
|
||||
"""Alibaba Cloud Coding Plan provider profiles.
|
||||
"""Alibaba Cloud Coding Plan provider profiles (intl + CN): a dedicated endpoint
|
||||
and key tier separate from ``alibaba``. Names match models.dev catalog keys.
|
||||
|
||||
Separate from the standard `alibaba` profile because it hits a different
|
||||
endpoint (coding-intl.dashscope.aliyuncs.com) with a dedicated API key tier.
|
||||
|
||||
Region split, mirroring the base DashScope pair (#73265):
|
||||
- ``alibaba-coding-plan`` → coding-intl.dashscope.aliyuncs.com (international)
|
||||
- ``alibaba-coding-plan-cn`` → coding.dashscope.aliyuncs.com (mainland China)
|
||||
|
||||
Profile names match the models.dev catalog keys exactly so model metadata
|
||||
lines up and ``model.provider: alibaba-coding-plan-cn`` resolves at runtime.
|
||||
|
||||
The CN profile checks its own ``ALIBABA_CODING_PLAN_CN_API_KEY`` first (#101122,
|
||||
mirroring kimi-coding-cn) and keeps the shared vars as ordered fallbacks so
|
||||
existing CN users configured with the shared key keep working.
|
||||
The CN profile checks its own key first and keeps the shared vars as ordered
|
||||
fallbacks so existing CN users configured with the shared key keep working.
|
||||
"""
|
||||
|
||||
from providers import register_provider
|
||||
|
||||
@@ -1,19 +1,8 @@
|
||||
"""Alibaba Cloud DashScope provider profiles.
|
||||
"""Alibaba Cloud DashScope provider profiles (intl + CN, plus the Model Studio
|
||||
Token Plan flat-token tier with its own key/endpoints — one module per vendor).
|
||||
|
||||
DashScope has region-split endpoints with the same key type:
|
||||
- ``alibaba`` → dashscope-intl.aliyuncs.com (international)
|
||||
- ``alibaba-cn`` → dashscope.aliyuncs.com (mainland China)
|
||||
|
||||
The Model Studio Token Plan (flat-token tier of the SAME vendor/service,
|
||||
same OpenAI-compatible protocol, its own key + endpoints) registers here
|
||||
too rather than as a new plugin directory — one module per vendor, matching
|
||||
how the kimi module carries both of its endpoint variants:
|
||||
- ``alibaba-token-plan`` → token-plan.ap-southeast-1.maas.aliyuncs.com
|
||||
- ``alibaba-token-plan-cn`` → token-plan.cn-beijing.maas.aliyuncs.com
|
||||
|
||||
Profile names match the models.dev catalog keys exactly
|
||||
(``alibaba`` / ``alibaba-cn``) so model metadata lines up and
|
||||
``model.provider: alibaba-cn`` resolves at runtime (#73265).
|
||||
Profile names match models.dev catalog keys exactly so model metadata lines up
|
||||
and ``model.provider: alibaba-cn`` resolves at runtime.
|
||||
"""
|
||||
|
||||
from providers import register_provider
|
||||
|
||||
@@ -15,11 +15,7 @@ class AnthropicProfile(ProviderProfile):
|
||||
"""Native Anthropic — uses x-api-key header, not Bearer."""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Anthropic uses x-api-key header and anthropic-version."""
|
||||
if not api_key:
|
||||
@@ -31,11 +27,7 @@ class AnthropicProfile(ProviderProfile):
|
||||
req.add_header("Accept", "application/json")
|
||||
with open_credentialed_url(req, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read().decode())
|
||||
return [
|
||||
m["id"]
|
||||
for m in data.get("data", [])
|
||||
if isinstance(m, dict) and "id" in m
|
||||
]
|
||||
return [m["id"] for m in data.get("data", []) if isinstance(m, dict) and "id" in m]
|
||||
except Exception as exc:
|
||||
logger.debug("fetch_models(anthropic): %s", exc)
|
||||
return None
|
||||
|
||||
@@ -1,8 +1,5 @@
|
||||
"""Microsoft Foundry provider profile.
|
||||
|
||||
Azure Foundry exposes an OpenAI-compatible endpoint; users supply their own
|
||||
base URL at setup since endpoints are per-resource.
|
||||
"""
|
||||
"""Microsoft Foundry provider profile: OpenAI-compatible, per-resource base URL
|
||||
supplied by the user at setup."""
|
||||
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
@@ -8,11 +8,7 @@ class BedrockProfile(ProviderProfile):
|
||||
"""AWS Bedrock — no REST /v1/models endpoint; uses AWS SDK."""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Bedrock model listing requires AWS SDK, not a REST call."""
|
||||
return None
|
||||
|
||||
@@ -1,28 +1,6 @@
|
||||
"""CommandCode provider profile.
|
||||
|
||||
CommandCode provides a unified API that fronts 20+ models from DeepSeek, Qwen,
|
||||
Kimi, GLM, MiniMax, StepFun, Xiaomi Mimo, Google Gemini, and OpenAI GPT — all
|
||||
accessible through either OpenAI-compatible chat completions or Anthropic
|
||||
Messages endpoints from a single base URL and API key.
|
||||
|
||||
Two provider profiles are registered:
|
||||
|
||||
``commandcode``
|
||||
``api_mode=chat_completions`` — standard OpenAI-compatible endpoint.
|
||||
Model prefix: ``deepseek/deepseek-v4-pro``, ``Qwen/Qwen3.7-Max``, etc.
|
||||
|
||||
``commandcode-anthropic``
|
||||
``api_mode=anthropic_messages`` — Anthropic Messages API-compatible.
|
||||
Model names: ``claude-sonnet-4-6``, ``claude-opus-4-7``,
|
||||
``claude-haiku-4-5-20251001``.
|
||||
|
||||
Both use the same ``COMMANDCODE_API_KEY`` env var and
|
||||
``https://api.commandcode.ai/provider/v1`` base URL. The
|
||||
``commandcode-anthropic`` profile relies on ``agent/anthropic_adapter.py``
|
||||
recognizing the ``api.commandcode.ai`` hostname for Bearer auth (the
|
||||
CommandCode /anthropic endpoint uses ``Authorization: Bearer``, not
|
||||
Anthropic's native ``x-api-key`` header).
|
||||
"""
|
||||
"""CommandCode provider profiles: ``commandcode`` (chat_completions) and
|
||||
``commandcode-anthropic`` (anthropic_messages, Bearer auth — see
|
||||
``agent/anthropic_adapter.py``). Same key and base URL for both."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -35,74 +13,63 @@ from providers.base import ProviderProfile, _profile_user_agent
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ── Shared constants ──────────────────────────────────────────────────────────
|
||||
_COMMANDCODE_BASE = "https://api.commandcode.ai/provider/v1"
|
||||
_COMMANDCODE_MODELS_URL = f"{_COMMANDCODE_BASE}/models"
|
||||
# Both profiles authenticate with the same key; each carries its own base-URL
|
||||
# override var so each renders its own card on the desktop Keys tab (rows are
|
||||
# keyed by env var, and the shared API key attributes to the first profile).
|
||||
_COMMANDCODE_ENV = ("COMMANDCODE_API_KEY", "COMMANDCODE_BASE_URL")
|
||||
_COMMANDCODE_ANTHROPIC_ENV = ("COMMANDCODE_API_KEY", "COMMANDCODE_ANTHROPIC_BASE_URL")
|
||||
|
||||
|
||||
def _fetch_commandcode_models(
|
||||
timeout: float = 10.0,
|
||||
base_url: str | None = None,
|
||||
) -> list[str] | None:
|
||||
"""Fetch the live model list from the CommandCode /models endpoint.
|
||||
"""Fetch model IDs from the public (unauthenticated) /models endpoint.
|
||||
|
||||
Returns a flat list of model IDs or None on failure.
|
||||
No auth required — the public models endpoint is open.
|
||||
|
||||
``base_url`` overrides the endpoint only when the caller passed a URL
|
||||
that differs from the default ``_COMMANDCODE_BASE`` (a user-configured
|
||||
``model.base_url`` / ``COMMANDCODE_BASE_URL`` pointing at a proxy or
|
||||
custom deployment). The picker passes base_url unconditionally, falling
|
||||
back to the profile default — equality means "not customised".
|
||||
The picker passes base_url unconditionally, so only a value differing from
|
||||
the default counts as a customised endpoint.
|
||||
"""
|
||||
caller_base = (base_url or "").strip()
|
||||
if caller_base and caller_base.rstrip("/") != _COMMANDCODE_BASE.rstrip("/"):
|
||||
models_url = caller_base.rstrip("/") + "/models"
|
||||
else:
|
||||
models_url = _COMMANDCODE_MODELS_URL
|
||||
caller_base = (base_url or "").strip().rstrip("/")
|
||||
custom = caller_base and caller_base != _COMMANDCODE_BASE
|
||||
models_url = caller_base + "/models" if custom else _COMMANDCODE_MODELS_URL
|
||||
try:
|
||||
req = urllib.request.Request(models_url)
|
||||
req.add_header("Accept", "application/json")
|
||||
req.add_header("User-Agent", _profile_user_agent())
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read().decode())
|
||||
# Response shape: {"object": "list", "data": [{"id": "..."}, ...]}
|
||||
return [
|
||||
m["id"]
|
||||
for m in data.get("data", [])
|
||||
if isinstance(m, dict) and "id" in m
|
||||
]
|
||||
return [m["id"] for m in data.get("data", []) if isinstance(m, dict) and "id" in m]
|
||||
except Exception as exc:
|
||||
logger.debug("fetch_models(commandcode): %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
# ── Chat Completions profile ──────────────────────────────────────────────────
|
||||
|
||||
class CommandCodeProfile(ProviderProfile):
|
||||
"""CommandCode — OpenAI-compatible chat completions endpoint."""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Fetch from the public CommandCode /models endpoint."""
|
||||
return _fetch_commandcode_models(timeout=timeout, base_url=base_url)
|
||||
|
||||
|
||||
class CommandCodeAnthropicProfile(ProviderProfile):
|
||||
"""CommandCode — Anthropic Messages API-compatible endpoint."""
|
||||
|
||||
def fetch_models(
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Public /models endpoint, filtered to Anthropic-family models."""
|
||||
all_models = _fetch_commandcode_models(timeout=timeout, base_url=base_url)
|
||||
if all_models is None:
|
||||
return None
|
||||
return [m for m in all_models if m.startswith("claude-")]
|
||||
|
||||
|
||||
commandcode = CommandCodeProfile(
|
||||
name="commandcode",
|
||||
aliases=("commandcode-chat",),
|
||||
api_mode="chat_completions",
|
||||
env_vars=_COMMANDCODE_ENV,
|
||||
# Same key as the anthropic profile; distinct base-URL override vars so each
|
||||
# profile renders its own card on the desktop Keys tab (rows keyed by env var).
|
||||
env_vars=("COMMANDCODE_API_KEY", "COMMANDCODE_BASE_URL"),
|
||||
display_name="CommandCode",
|
||||
description="CommandCode — 20+ models via OpenAI-compatible API",
|
||||
signup_url="https://commandcode.ai/",
|
||||
@@ -124,53 +91,19 @@ commandcode = CommandCodeProfile(
|
||||
default_aux_model="deepseek/deepseek-v4-flash",
|
||||
)
|
||||
|
||||
|
||||
# ── Anthropic Messages profile ────────────────────────────────────────────────
|
||||
|
||||
class CommandCodeAnthropicProfile(ProviderProfile):
|
||||
"""CommandCode — Anthropic Messages API-compatible endpoint.
|
||||
|
||||
Uses Bearer auth (same API key), not Anthropic's native x-api-key header.
|
||||
``agent/anthropic_adapter.py`` must recognize ``api.commandcode.ai``
|
||||
as a Bearer-auth domain for this to work.
|
||||
"""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
) -> list[str] | None:
|
||||
"""Fetch from the public CommandCode /models endpoint.
|
||||
|
||||
Filter to Anthropic-family models only (claude-*).
|
||||
"""
|
||||
all_models = _fetch_commandcode_models(timeout=timeout, base_url=base_url)
|
||||
if all_models is None:
|
||||
return None
|
||||
return [m for m in all_models if m.startswith("claude-")]
|
||||
|
||||
|
||||
commandcode_anthropic = CommandCodeAnthropicProfile(
|
||||
name="commandcode-anthropic",
|
||||
aliases=("commandcode-claude",),
|
||||
api_mode="anthropic_messages",
|
||||
env_vars=_COMMANDCODE_ANTHROPIC_ENV,
|
||||
env_vars=("COMMANDCODE_API_KEY", "COMMANDCODE_ANTHROPIC_BASE_URL"),
|
||||
display_name="CommandCode (Anthropic)",
|
||||
description="CommandCode — Claude models via Anthropic Messages API",
|
||||
signup_url="https://commandcode.ai/",
|
||||
base_url=_COMMANDCODE_BASE,
|
||||
models_url=_COMMANDCODE_MODELS_URL,
|
||||
fallback_models=(
|
||||
"claude-sonnet-4-6",
|
||||
"claude-opus-4-7",
|
||||
"claude-haiku-4-5-20251001",
|
||||
),
|
||||
fallback_models=("claude-sonnet-4-6", "claude-opus-4-7", "claude-haiku-4-5-20251001"),
|
||||
default_aux_model="claude-haiku-4-5-20251001",
|
||||
)
|
||||
|
||||
|
||||
# ── Registration ──────────────────────────────────────────────────────────────
|
||||
register_provider(commandcode)
|
||||
register_provider(commandcode_anthropic)
|
||||
|
||||
@@ -1,11 +1,9 @@
|
||||
"""GitHub Copilot ACP provider profile.
|
||||
|
||||
copilot-acp does not speak OpenAI-over-HTTP: it drives an external ACP
|
||||
subprocess over stdio. The profile therefore supplies its own client through
|
||||
:meth:`ProviderProfile.create_client` instead of letting the core build an
|
||||
``openai.OpenAI``. That hook is the registration seam — this profile is its
|
||||
in-tree consumer, and an out-of-tree ACP provider registered from
|
||||
``~/.hermes/plugins/model-providers/`` or a pip entry point uses the exact same
|
||||
subprocess over stdio, so the profile supplies its own client via
|
||||
:meth:`ProviderProfile.create_client`. An out-of-tree ACP provider registered
|
||||
from ``~/.hermes/plugins/model-providers/`` or a pip entry point uses the same
|
||||
three lines without touching core.
|
||||
"""
|
||||
|
||||
@@ -25,11 +23,7 @@ class CopilotACPProfile(ProviderProfile):
|
||||
return CopilotACPClient(**client_kwargs)
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Model listing is handled by the ACP subprocess."""
|
||||
return None
|
||||
|
||||
@@ -1,13 +1,8 @@
|
||||
"""Copilot / GitHub Models provider profile.
|
||||
|
||||
Copilot uses per-model api_mode routing:
|
||||
- GPT-5+ / Codex models → codex_responses
|
||||
- Claude models → anthropic_messages
|
||||
- Everything else → chat_completions (this profile covers that subset)
|
||||
|
||||
Key quirks for the chat_completions subset:
|
||||
- Editor attribution headers (via copilot_default_headers())
|
||||
- GitHub Models reasoning extra_body (model-catalog gated)
|
||||
Core routes GPT-5+/Codex -> codex_responses and Claude -> anthropic_messages;
|
||||
this profile covers the chat_completions remainder: editor attribution headers
|
||||
(copilot_default_headers()) and catalog-gated GitHub Models reasoning.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
@@ -27,46 +22,28 @@ class CopilotProfile(ProviderProfile):
|
||||
supports_reasoning: bool = False,
|
||||
**ctx,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
if supports_reasoning and model:
|
||||
try:
|
||||
from hermes_cli.models import github_model_reasoning_efforts
|
||||
if not (supports_reasoning and model):
|
||||
return {}, {}
|
||||
try:
|
||||
from hermes_cli.models import clamp_reasoning_effort_to_supported, github_model_reasoning_efforts
|
||||
|
||||
supported_efforts = github_model_reasoning_efforts(model)
|
||||
if supported_efforts and reasoning_config:
|
||||
effort = reasoning_config.get("effort", "medium")
|
||||
# Honor the requested level when the live Copilot catalog
|
||||
# lists it as supported: gpt-5.5/gpt-5.4 DO support
|
||||
# ``xhigh``. Otherwise clamp to the nearest WEAKER
|
||||
# supported level via the shared ladder helper — the old
|
||||
# ad-hoc rules dropped everything unrecognized to
|
||||
# ``medium``, which inverted the ladder: ``ultra`` (the
|
||||
# strongest ask) resolved weaker than an explicit
|
||||
# ``high`` (#74295).
|
||||
if effort not in supported_efforts:
|
||||
from hermes_cli.models import (
|
||||
clamp_reasoning_effort_to_supported,
|
||||
)
|
||||
|
||||
effort = clamp_reasoning_effort_to_supported(
|
||||
effort, list(supported_efforts)
|
||||
)
|
||||
if effort not in supported_efforts:
|
||||
# Unrecognized/bespoke level the ladder can't
|
||||
# place — fall back to medium, then to the
|
||||
# catalog's first entry.
|
||||
effort = (
|
||||
"medium"
|
||||
if "medium" in supported_efforts
|
||||
else supported_efforts[0]
|
||||
)
|
||||
if effort in supported_efforts:
|
||||
extra_body["reasoning"] = {"effort": effort}
|
||||
elif supported_efforts:
|
||||
extra_body["reasoning"] = {"effort": "medium"}
|
||||
except Exception:
|
||||
pass
|
||||
return extra_body, {}
|
||||
supported = github_model_reasoning_efforts(model)
|
||||
if not supported:
|
||||
return {}, {}
|
||||
if not reasoning_config:
|
||||
return {"reasoning": {"effort": "medium"}}, {}
|
||||
effort = reasoning_config.get("effort", "medium")
|
||||
# Honor a level the live catalog lists; otherwise clamp to the nearest
|
||||
# WEAKER supported level (never drop straight to medium, which inverted
|
||||
# the ladder: ultra < high). Bespoke levels the ladder can't place fall
|
||||
# to medium (or the first supported level).
|
||||
if effort not in supported:
|
||||
effort = clamp_reasoning_effort_to_supported(effort, list(supported))
|
||||
if effort not in supported:
|
||||
effort = "medium" if "medium" in supported else supported[0]
|
||||
return {"reasoning": {"effort": effort}}, {}
|
||||
except Exception:
|
||||
return {}, {}
|
||||
|
||||
|
||||
copilot = CopilotProfile(
|
||||
|
||||
@@ -1,127 +1,65 @@
|
||||
"""Custom / Ollama (local) provider profile.
|
||||
|
||||
Covers any endpoint registered as provider="custom", including local
|
||||
Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on
|
||||
Volcengine ARK, vLLM, llama.cpp). Key quirks:
|
||||
- ollama_num_ctx → extra_body.options.num_ctx (local context window)
|
||||
- reasoning_config disabled → top-level reasoning_effort="none"
|
||||
(Ollama /v1/chat/completions ignores think=False — ollama#14820)
|
||||
+ extra_body.think = False only on Ollama URLs (/api/chat and proxies)
|
||||
- reasoning_config enabled + effort → top-level reasoning_effort
|
||||
(the native OpenAI-compatible format GLM/ARK expect; unset omits it
|
||||
so the endpoint's server default applies)
|
||||
"""
|
||||
"""Custom / Ollama (local) provider profile: any endpoint registered as
|
||||
provider="custom" (Ollama, vLLM, llama.cpp, GLM-5.2 on ARK, …)."""
|
||||
|
||||
from typing import Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
def _looks_like_ollama_endpoint(base_url: str | None) -> bool:
|
||||
"""True when ``base_url`` is an Ollama host, not a generic OpenAI-compat relay.
|
||||
"""True only for explicit Ollama signatures (port 11434 or an ``ollama`` host label).
|
||||
|
||||
``think`` is an Ollama-native extra_body field. Strict hosts (Mistral
|
||||
``extra=forbid``, Groq, …) reject it with HTTP 422. Match only explicit
|
||||
Ollama signatures — default port 11434, or ``ollama`` as a hostname
|
||||
label — not arbitrary localhost (llama.cpp / vLLM / LM Studio).
|
||||
``think`` is Ollama-native; strict hosts (Mistral, Groq) 422 on it, and
|
||||
arbitrary localhost may be llama.cpp / vLLM / LM Studio.
|
||||
"""
|
||||
raw = (base_url or "").strip()
|
||||
if not raw:
|
||||
return False
|
||||
parsed = urlparse(raw if "://" in raw else f"//{raw}")
|
||||
# urlparse raises ValueError for non-integer / out-of-range ports
|
||||
# ("http://host:99999/v1" parses fine in the OpenAI client, so the URL
|
||||
# is reachable here). Treat a malformed port as "not Ollama" instead of
|
||||
# killing the whole kwargs build — same try/except shape the 11434
|
||||
# check in hermes_cli/models.py uses, not the same detection logic.
|
||||
# urlparse raises ValueError on malformed ports ("host:99999"); treat as not-Ollama.
|
||||
try:
|
||||
if parsed.port == 11434:
|
||||
return True
|
||||
except ValueError:
|
||||
return False
|
||||
host = (parsed.hostname or "").lower().rstrip(".")
|
||||
if not host:
|
||||
return False
|
||||
if host == "ollama.com" or host.endswith(".ollama.com"):
|
||||
return True
|
||||
return "ollama" in host.split(".")
|
||||
return bool(host) and (host == "ollama.com" or host.endswith(".ollama.com") or "ollama" in host.split("."))
|
||||
|
||||
|
||||
class CustomProfile(ProviderProfile):
|
||||
"""Custom/Ollama local provider — think=false and num_ctx support."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self,
|
||||
*,
|
||||
reasoning_config: dict | None = None,
|
||||
ollama_num_ctx: int | None = None,
|
||||
**ctx: Any,
|
||||
self, *, reasoning_config: dict | None = None, ollama_num_ctx: int | None = None, **ctx: Any
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
top_level: dict[str, Any] = {}
|
||||
|
||||
# Ollama context window
|
||||
if ollama_num_ctx:
|
||||
options = extra_body.get("options", {})
|
||||
options["num_ctx"] = ollama_num_ctx
|
||||
extra_body["options"] = options
|
||||
extra_body["options"] = {"num_ctx": ollama_num_ctx}
|
||||
|
||||
# Reasoning / thinking control for custom OpenAI-compatible endpoints
|
||||
# (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …).
|
||||
#
|
||||
# - disabled → top-level reasoning_effort="none"; extra_body.think
|
||||
# = False only on Ollama URLs (Ollama's thinking-off flag)
|
||||
# - enabled + effort set → TOP-LEVEL reasoning_effort string, the
|
||||
# format GLM-5.2/ARK and other OpenAI-compatible reasoning APIs
|
||||
# expect (GLM documents "high" and "max"; "max" is its default).
|
||||
# - enabled + no effort → omit both, so the endpoint applies its own
|
||||
# server-side default (do NOT force a level the user didn't pick).
|
||||
#
|
||||
# We deliberately do NOT emit ``think=True`` on enable: it is an
|
||||
# Ollama-only flag and thinking is already server-default-on for these
|
||||
# backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't
|
||||
# recognize it. Mirrors the DeepSeek/Zai profile precedent. The same
|
||||
# constraint applies to ``think=False`` on disable — Mistral/Groq
|
||||
# reject unknown fields (HTTP 422 extra_forbidden) rather than ignoring
|
||||
# them, so that flag stays Ollama-URL-gated.
|
||||
# disabled -> top-level reasoning_effort="none" (Ollama's /v1 ignores
|
||||
# extra_body.think) plus think=False only on Ollama URLs; enabled+effort ->
|
||||
# top-level reasoning_effort clamped to the OpenAI-compat wire (GLM/ARK,
|
||||
# vLLM and SGLang all top out at "max"; "ultra" verbatim 400s); enabled
|
||||
# without effort -> omit so the server default applies. Never emit
|
||||
# think=True (Ollama-only flag).
|
||||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
_effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
_enabled = reasoning_config.get("enabled", True)
|
||||
if _effort == "none" or _enabled is False:
|
||||
# Ollama's /v1/chat/completions silently ignores
|
||||
# extra_body.think (only /api/chat honours it — ollama#14820)
|
||||
# but respects the top-level reasoning_effort field (#25758).
|
||||
# Always emit reasoning_effort="none"; only add think=False
|
||||
# when the URL is actually Ollama.
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if effort == "none" or reasoning_config.get("enabled", True) is False:
|
||||
top_level["reasoning_effort"] = "none"
|
||||
if _looks_like_ollama_endpoint(ctx.get("base_url")):
|
||||
extra_body["think"] = False
|
||||
elif _effort:
|
||||
# Clamp the internal ladder onto the widest OpenAI-compatible
|
||||
# wire vocabulary (shared policy in agent.reasoning_effort) —
|
||||
# GLM/ARK, vLLM and SGLang all top out at "max"; forwarding
|
||||
# "ultra" verbatim is a guaranteed 400 (#89503).
|
||||
from agent.reasoning_effort import (
|
||||
OPENAI_COMPAT_WIRE_EFFORTS,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
top_level["reasoning_effort"] = clamp_effort(
|
||||
_effort, OPENAI_COMPAT_WIRE_EFFORTS
|
||||
)
|
||||
|
||||
elif effort:
|
||||
top_level["reasoning_effort"] = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS)
|
||||
return extra_body, top_level
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Custom/Ollama: base_url is user-configured; fetch if set."""
|
||||
"""base_url is user-configured; fetch only if set."""
|
||||
if not (base_url or self.base_url):
|
||||
return None
|
||||
return super().fetch_models(api_key=api_key, base_url=base_url, timeout=timeout)
|
||||
@@ -129,20 +67,11 @@ class CustomProfile(ProviderProfile):
|
||||
|
||||
custom = CustomProfile(
|
||||
name="custom",
|
||||
aliases=(
|
||||
"ollama",
|
||||
"local",
|
||||
"vllm",
|
||||
"llamacpp",
|
||||
"llama.cpp",
|
||||
"llama-cpp",
|
||||
),
|
||||
aliases=("ollama", "local", "vllm", "llamacpp", "llama.cpp", "llama-cpp"),
|
||||
env_vars=(), # No fixed key — custom endpoint
|
||||
base_url="", # User-configured
|
||||
# Without this, no max_tokens is sent and Ollama falls back to its internal
|
||||
# num_predict=128, truncating responses after a few tokens (#39281). This is
|
||||
# only a floor used when the user hasn't set model.max_tokens — they can
|
||||
# override per-model — so we set it generously rather than lowballing it.
|
||||
# Floor only (user model.max_tokens overrides); without it Ollama falls
|
||||
# back to num_predict=128 and truncates.
|
||||
default_max_tokens=65536,
|
||||
)
|
||||
|
||||
|
||||
@@ -1,33 +1,20 @@
|
||||
"""DeepInfra provider profile.
|
||||
|
||||
DeepInfra is an OpenAI-compatible inference gateway that hosts 100+ open
|
||||
models (Step, GLM, Kimi, DeepSeek, MiniMax, Nemotron, Mistral, Qwen, …) as
|
||||
well as image-gen / TTS / STT / embedding endpoints. The chat surface is
|
||||
wired in through this profile; non-chat surfaces are wired in through
|
||||
their respective plugin subsystems (``plugins/image_gen/deepinfra`` and
|
||||
the TTS/STT dispatchers in ``tools/``).
|
||||
"""
|
||||
"""DeepInfra provider profile (chat surface; image-gen/TTS/STT are wired via
|
||||
their own plugin subsystems)."""
|
||||
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
class _DeepInfraProfile(ProviderProfile):
|
||||
"""DeepInfra profile with live vision-default discovery.
|
||||
|
||||
Owns its own vision default so shared vision resolution in
|
||||
``agent/auxiliary_client.py`` stays provider-agnostic (a
|
||||
``default_vision_model()`` hook call instead of an ``if provider ==
|
||||
"deepinfra"`` branch reaching into the catalog helpers).
|
||||
"""
|
||||
"""DeepInfra profile with live vision-default discovery, so shared vision
|
||||
resolution in ``agent/auxiliary_client.py`` stays provider-agnostic."""
|
||||
|
||||
def default_vision_model(self): # type: ignore[override]
|
||||
"""First vision-capable *chat* model from the live catalog, or None.
|
||||
|
||||
Key-gated so a box without ``DEEPINFRA_API_KEY`` never pays the
|
||||
catalog round-trip. Requires the ``chat`` surface tag (not just the
|
||||
``vision`` capability) so an image-gen/edit model that merely carries
|
||||
a ``vision`` tag can't be picked as a chat-completions vision backend.
|
||||
Key-gated so a box without DEEPINFRA_API_KEY never pays the round-trip.
|
||||
Requires the ``chat`` surface tag so an image-gen model carrying a
|
||||
``vision`` tag can't be picked as a chat-completions vision backend.
|
||||
"""
|
||||
from agent.secret_scope import get_secret
|
||||
|
||||
@@ -41,10 +28,8 @@ class _DeepInfraProfile(ProviderProfile):
|
||||
for item in items or []:
|
||||
metadata = item.get("metadata") or {}
|
||||
tags = metadata.get("tags") if isinstance(metadata, dict) else None
|
||||
if isinstance(tags, list) and "vision" in tags:
|
||||
model_id = item.get("id")
|
||||
if model_id:
|
||||
return model_id
|
||||
if isinstance(tags, list) and "vision" in tags and item.get("id"):
|
||||
return item["id"]
|
||||
return None
|
||||
|
||||
|
||||
@@ -57,24 +42,13 @@ deepinfra = _DeepInfraProfile(
|
||||
env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"),
|
||||
base_url="https://api.deepinfra.com/v1/openai",
|
||||
auth_type="api_key",
|
||||
# The catalog spans models with different output limits. Omitting a
|
||||
# provider-wide default lets DeepInfra apply its documented per-model cap;
|
||||
# an explicit user ``agent.max_tokens`` still passes through normally.
|
||||
# No provider-wide cap: DeepInfra applies its documented per-model limit.
|
||||
default_max_tokens=None,
|
||||
# Auxiliary model — cheap/fast chat model the same provider uses for
|
||||
# side tasks (context compression, session search, web extract,
|
||||
# vision). This is the *only* hardcoded DeepInfra model in the
|
||||
# integration: aux resolution is synchronous (no time for a catalog
|
||||
# round-trip on every agent turn), so we need one explicit choice.
|
||||
# Every other surface (chat picker, image-gen, tts, stt, pricing)
|
||||
# discovers models live from
|
||||
# ``api.deepinfra.com/v1/openai/models?filter=true&sort_by=hermes``.
|
||||
# The only hardcoded DeepInfra model: aux resolution is synchronous, so it
|
||||
# can't wait on a catalog round-trip. Everything else is discovered live.
|
||||
default_aux_model="deepseek-ai/DeepSeek-V4-Flash",
|
||||
# ``fallback_models`` deliberately empty — the live catalog at
|
||||
# ``hermes_cli/models.py::_fetch_deepinfra_models`` is the source of
|
||||
# truth. When the live fetch fails (network/DNS), the picker shows
|
||||
# no options, which is preferable to silently routing the user to a
|
||||
# model that may have been retired upstream.
|
||||
# Empty on purpose: the live catalog is the source of truth; an empty picker
|
||||
# beats silently routing to a retired model.
|
||||
fallback_models=(),
|
||||
)
|
||||
|
||||
|
||||
@@ -1,96 +1,45 @@
|
||||
"""DeepSeek provider profile.
|
||||
|
||||
DeepSeek's V4 family defaults to thinking-mode ON when ``extra_body.thinking``
|
||||
is unset. The API then returns ``reasoning_content`` and starts enforcing
|
||||
the contract that subsequent turns echo it back; combined with how Hermes
|
||||
replays history this lands on the notorious HTTP 400
|
||||
``reasoning_content must be passed back`` error after the first tool call
|
||||
(#15700, #17212, #17825).
|
||||
|
||||
This profile overrides :meth:`build_api_kwargs_extras` to mirror the Kimi /
|
||||
Moonshot wire shape that DeepSeek's OpenAI-compat endpoint expects:
|
||||
|
||||
{"reasoning_effort": "<low|medium|high|max>",
|
||||
"extra_body": {"thinking": {"type": "enabled" | "disabled"}}}
|
||||
|
||||
Non-thinking models (``deepseek-v3-*`` variants) are left as no-ops so we
|
||||
don't perturb the V3 wire format.
|
||||
|
||||
The legacy aliases ``deepseek-chat`` / ``deepseek-reasoner`` were retired on
|
||||
2026-07-24. Use ``deepseek-v4-flash`` or ``deepseek-v4-pro``; Hermes remaps
|
||||
the retired IDs in ``hermes_cli.model_normalize``.
|
||||
V4 defaults to thinking ON when ``extra_body.thinking`` is unset, and then
|
||||
requires ``reasoning_content`` to be echoed back on later turns (HTTP 400 after
|
||||
the first tool call otherwise). This profile sets ``thinking`` explicitly and
|
||||
maps effort onto DeepSeek's ``reasoning_effort``; V3 models are left untouched.
|
||||
Retired ``deepseek-chat``/``deepseek-reasoner`` IDs are remapped in
|
||||
``hermes_cli.model_normalize`` before reaching here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, clamp_effort
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
def _model_supports_thinking(model: str | None) -> bool:
|
||||
"""DeepSeek thinking-capable model families.
|
||||
|
||||
Currently covers the V4 family (``deepseek-v4-pro``, ``deepseek-v4-flash``,
|
||||
and any future ``deepseek-v4-*`` variants). Retired aliases are remapped
|
||||
before requests leave Hermes, so they are not listed here.
|
||||
"""
|
||||
m = (model or "").strip().lower()
|
||||
if not m:
|
||||
return False
|
||||
if m.startswith("deepseek-v") and not m.startswith("deepseek-v3"):
|
||||
# deepseek-v4-*, deepseek-v5-*, etc. — every V4+ generation has
|
||||
# thinking. v3 explicitly excluded.
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
class DeepSeekProfile(ProviderProfile):
|
||||
"""DeepSeek — extra_body.thinking + top-level reasoning_effort."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
m = (model or "").strip().lower()
|
||||
# deepseek-v4-* and every later generation; v3 explicitly excluded.
|
||||
if not m.startswith("deepseek-v") or m.startswith("deepseek-v3"):
|
||||
return {}, {}
|
||||
rc = reasoning_config if isinstance(reasoning_config, dict) else None
|
||||
# Always set explicitly (default enabled, matching the API default) to
|
||||
# avoid the reasoning_content echo trap on subsequent turns.
|
||||
if rc is not None and rc.get("enabled") is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
top_level: dict[str, Any] = {}
|
||||
|
||||
if not _model_supports_thinking(model):
|
||||
# V3 / unknown — leave wire format untouched, current behavior.
|
||||
return extra_body, top_level
|
||||
|
||||
# Determine enabled/disabled. Default is enabled to match DeepSeek's
|
||||
# API default; the API requires this to be set explicitly to avoid the
|
||||
# reasoning_content echo trap on subsequent turns.
|
||||
enabled = True
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
enabled = False
|
||||
|
||||
extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"}
|
||||
|
||||
if not enabled:
|
||||
return extra_body, top_level
|
||||
|
||||
# Effort mapping via the shared vocabulary in agent.reasoning_effort
|
||||
# (DeepSeek V4: low/medium/high/max, xhigh rounds up to max). When no
|
||||
# effort is set we omit reasoning_effort so DeepSeek applies its
|
||||
# server default (currently high).
|
||||
if isinstance(reasoning_config, dict):
|
||||
from agent.reasoning_effort import (
|
||||
DEEPSEEK_V4_EFFORTS,
|
||||
DEEPSEEK_V4_OVERRIDES,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if effort and effort != "none":
|
||||
clamped = clamp_effort(
|
||||
effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES
|
||||
)
|
||||
if clamped in DEEPSEEK_V4_EFFORTS:
|
||||
top_level["reasoning_effort"] = clamped
|
||||
|
||||
return extra_body, top_level
|
||||
# No effort -> omit reasoning_effort so DeepSeek applies its server default.
|
||||
effort = (rc.get("effort") or "").strip().lower() if rc is not None else ""
|
||||
if effort and effort != "none":
|
||||
clamped = clamp_effort(effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES)
|
||||
if clamped in DEEPSEEK_V4_EFFORTS:
|
||||
top_level["reasoning_effort"] = clamped
|
||||
return {"thinking": {"type": "enabled"}}, top_level
|
||||
|
||||
|
||||
deepseek = DeepSeekProfile(
|
||||
@@ -100,10 +49,7 @@ deepseek = DeepSeekProfile(
|
||||
display_name="DeepSeek",
|
||||
description="DeepSeek — native DeepSeek API",
|
||||
signup_url="https://platform.deepseek.com/",
|
||||
fallback_models=(
|
||||
"deepseek-v4-pro",
|
||||
"deepseek-v4-flash",
|
||||
),
|
||||
fallback_models=("deepseek-v4-pro", "deepseek-v4-flash"),
|
||||
base_url="https://api.deepseek.com/v1",
|
||||
default_aux_model="deepseek-v4-flash",
|
||||
)
|
||||
|
||||
@@ -1,13 +1,5 @@
|
||||
"""Fireworks AI provider profile.
|
||||
|
||||
Fireworks AI serves fast, production-grade inference for open and proprietary
|
||||
models through an OpenAI-compatible chat-completions endpoint.
|
||||
|
||||
Address models directly by their catalog ID, e.g.
|
||||
``accounts/fireworks/models/kimi-k2p6`` or ``accounts/fireworks/models/glm-5p2``.
|
||||
Model IDs here track the canonical Fireworks catalog (fw-ai/fireconnect
|
||||
``setup-cli``).
|
||||
"""
|
||||
"""Fireworks AI provider profile. Models are addressed by full catalog ID
|
||||
(``accounts/fireworks/models/<slug>``), tracking fw-ai/fireconnect ``setup-cli``."""
|
||||
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
@@ -23,19 +15,15 @@ fireworks = ProviderProfile(
|
||||
env_vars=("FIREWORKS_API_KEY",),
|
||||
base_url="https://api.fireworks.ai/inference/v1",
|
||||
auth_type="api_key",
|
||||
# Attribution headers sent on every Fireworks request. Values match the
|
||||
# canonical Hermes set in agent/auxiliary_client.py. Applied through the
|
||||
# generic profile.default_headers path, so they survive switch_model and
|
||||
# credential rotation.
|
||||
# Attribution headers (canonical Hermes set); via default_headers so they
|
||||
# survive switch_model and credential rotation.
|
||||
default_headers={
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
},
|
||||
# Auxiliary model for cheap tasks (compaction, title generation, vision).
|
||||
# A standard pay-as-you-go catalog ``/models/`` ID.
|
||||
default_aux_model="accounts/fireworks/models/glm-5p2",
|
||||
# Curated safety net shown in the picker when the live catalog fetch fails.
|
||||
# Picker safety net when the live catalog fetch fails.
|
||||
fallback_models=(
|
||||
"accounts/fireworks/models/kimi-k2p6",
|
||||
"accounts/fireworks/models/glm-5p2",
|
||||
|
||||
@@ -1,12 +1,7 @@
|
||||
"""Google Gemini provider profiles.
|
||||
"""Google Gemini (AI Studio) provider profile.
|
||||
|
||||
gemini: Google AI Studio (API key) — uses GeminiNativeClient
|
||||
|
||||
Reports api_mode="chat_completions" but uses a custom native client
|
||||
that bypasses the standard OpenAI transport. The profile captures auth
|
||||
and endpoint metadata for auth.py / runtime_provider.py migration, and
|
||||
carries the thinking_config translation hook so the transport's profile
|
||||
path produces the same extra_body shape the legacy flag path did.
|
||||
Reports api_mode="chat_completions" but runs on GeminiNativeClient; this
|
||||
profile carries auth/endpoint metadata and the thinking_config translation hook.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
@@ -18,34 +13,22 @@ from providers.base import ProviderProfile
|
||||
class GeminiProfile(ProviderProfile):
|
||||
"""Gemini — translate reasoning_config to thinking_config in extra_body."""
|
||||
|
||||
def build_extra_body(
|
||||
self, *, session_id: str | None = None, **context: Any
|
||||
) -> dict[str, Any]:
|
||||
"""Emit extra_body.thinking_config (native) or extra_body.extra_body.google.thinking_config
|
||||
(OpenAI-compat /openai subpath), mirroring the legacy path's behavior.
|
||||
"""
|
||||
def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]:
|
||||
"""Native: ``thinking_config``; OpenAI-compat /openai subpath:
|
||||
``extra_body.google.thinking_config`` (snake_case)."""
|
||||
from agent.transports.chat_completions import (
|
||||
_build_gemini_thinking_config,
|
||||
_is_gemini_openai_compat_base_url,
|
||||
_snake_case_gemini_thinking_config,
|
||||
)
|
||||
|
||||
model = context.get("model") or ""
|
||||
reasoning_config = context.get("reasoning_config")
|
||||
base_url = context.get("base_url") or self.base_url
|
||||
|
||||
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
|
||||
if not raw_thinking_config:
|
||||
raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config"))
|
||||
if not raw:
|
||||
return {}
|
||||
|
||||
body: dict[str, Any] = {}
|
||||
if self.name == "gemini" and _is_gemini_openai_compat_base_url(base_url):
|
||||
thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config)
|
||||
if thinking_config:
|
||||
body["extra_body"] = {"google": {"thinking_config": thinking_config}}
|
||||
else:
|
||||
body["thinking_config"] = raw_thinking_config
|
||||
return body
|
||||
if self.name == "gemini" and _is_gemini_openai_compat_base_url(context.get("base_url") or self.base_url):
|
||||
thinking_config = _snake_case_gemini_thinking_config(raw)
|
||||
return {"extra_body": {"google": {"thinking_config": thinking_config}}} if thinking_config else {}
|
||||
return {"thinking_config": raw}
|
||||
|
||||
|
||||
gemini = GeminiProfile(
|
||||
|
||||
@@ -13,9 +13,7 @@ gmi = ProviderProfile(
|
||||
env_vars=("GMI_API_KEY", "GMI_BASE_URL"),
|
||||
base_url="https://api.gmi-serving.com/v1",
|
||||
auth_type="api_key",
|
||||
# Attribution so GMI can identify traffic from Hermes Agent.
|
||||
# The generic profile.default_headers fallback in run_agent.py and
|
||||
# agent/auxiliary_client.py picks this up at client construction time.
|
||||
# Attribution so GMI can identify Hermes Agent traffic.
|
||||
default_headers={"User-Agent": f"HermesAgent/{_HERMES_VERSION}"},
|
||||
default_aux_model="google/gemini-3.1-flash-lite-preview",
|
||||
fallback_models=(
|
||||
|
||||
@@ -10,10 +10,7 @@ huggingface = ProviderProfile(
|
||||
display_name="HuggingFace",
|
||||
description="HuggingFace Inference API",
|
||||
signup_url="https://huggingface.co/settings/tokens",
|
||||
fallback_models=(
|
||||
"Qwen/Qwen3.5-72B-Instruct",
|
||||
"deepseek-ai/DeepSeek-V3.2",
|
||||
),
|
||||
fallback_models=("Qwen/Qwen3.5-72B-Instruct", "deepseek-ai/DeepSeek-V3.2"),
|
||||
base_url="https://router.huggingface.co/v1",
|
||||
)
|
||||
|
||||
|
||||
@@ -1,36 +1,32 @@
|
||||
"""Kimi / Moonshot provider profiles.
|
||||
|
||||
Kimi has dual endpoints:
|
||||
- sk-kimi-* keys → api.kimi.com/coding (Anthropic Messages API)
|
||||
- legacy keys → api.moonshot.ai/v1 (OpenAI chat completions)
|
||||
|
||||
This module covers the chat_completions path (/v1 endpoint).
|
||||
"""
|
||||
"""Kimi / Moonshot provider profiles (chat_completions path; sk-kimi-* keys are
|
||||
redirected to api.kimi.com/coding by core)."""
|
||||
|
||||
from typing import Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from agent.reasoning_effort import KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, clamp_effort, requested_effort
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import OMIT_TEMPERATURE, ProviderProfile
|
||||
|
||||
_HEADERS = {
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
}
|
||||
|
||||
|
||||
def _is_confirmed_kimi_coding_url(base_url: str) -> bool:
|
||||
"""Return True only for Kimi Code's canonical HTTPS API surfaces."""
|
||||
try:
|
||||
parsed = urlparse(base_url)
|
||||
port = parsed.port
|
||||
p = urlparse(base_url)
|
||||
port = p.port
|
||||
except ValueError:
|
||||
return False
|
||||
return (
|
||||
parsed.scheme.lower() == "https"
|
||||
and (parsed.hostname or "").lower() == "api.kimi.com"
|
||||
and port in (None, 443)
|
||||
and parsed.username is None
|
||||
and parsed.password is None
|
||||
and parsed.path.rstrip("/") in {"/coding", "/coding/v1"}
|
||||
and not parsed.query
|
||||
and not parsed.fragment
|
||||
p.scheme.lower() == "https" and (p.hostname or "").lower() == "api.kimi.com" and port in (None, 443)
|
||||
and p.username is None and p.password is None
|
||||
and p.path.rstrip("/") in {"/coding", "/coding/v1"} and not p.query and not p.fragment
|
||||
)
|
||||
|
||||
|
||||
@@ -38,22 +34,15 @@ class KimiProfile(ProviderProfile):
|
||||
"""Kimi/Moonshot — temperature omitted, thinking xor reasoning_effort."""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Use Kimi Code's OpenAI-compatible surface for model discovery."""
|
||||
"""Use Kimi Code's OpenAI-compatible surface for model discovery; the bare
|
||||
``k3`` slug is only served there, so it is filtered off other endpoints."""
|
||||
effective_base = (base_url or self.base_url or "").rstrip("/")
|
||||
confirmed_coding_endpoint = _is_confirmed_kimi_coding_url(effective_base)
|
||||
if confirmed_coding_endpoint and urlparse(effective_base).path.rstrip("/") == "/coding":
|
||||
effective_base += "/v1"
|
||||
models = super().fetch_models(
|
||||
api_key=api_key,
|
||||
base_url=effective_base or None,
|
||||
timeout=timeout,
|
||||
)
|
||||
models = super().fetch_models(api_key=api_key, base_url=effective_base or None, timeout=timeout)
|
||||
if models is None or confirmed_coding_endpoint:
|
||||
return models
|
||||
return [model for model in models if model.strip().lower() != "k3"]
|
||||
@@ -61,53 +50,15 @@ class KimiProfile(ProviderProfile):
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Kimi reasoning controls.
|
||||
|
||||
Moonshot's wire shape treats ``extra_body.thinking`` (a binary toggle)
|
||||
and a top-level ``reasoning_effort`` as mutually exclusive — sending
|
||||
both is at best redundant and risks "cannot specify both 'thinking' and
|
||||
'reasoning_effort'" (HTTP 400). This mirrors the kimi-k2 handling on the
|
||||
opencode-go relay: send effort when one is requested, otherwise fall
|
||||
back to ``extra_body.thinking`` — never both.
|
||||
"""
|
||||
extra_body = {}
|
||||
top_level = {}
|
||||
|
||||
if not reasoning_config or not isinstance(reasoning_config, dict):
|
||||
# No config → thinking enabled, let the server pick the depth.
|
||||
# (Previously also sent reasoning_effort="medium", which paired
|
||||
# thinking + effort on every default call.)
|
||||
extra_body["thinking"] = {"type": "enabled"}
|
||||
return extra_body, top_level
|
||||
|
||||
enabled = reasoning_config.get("enabled", True)
|
||||
if enabled is False:
|
||||
extra_body["thinking"] = {"type": "disabled"}
|
||||
return extra_body, top_level
|
||||
|
||||
# Enabled: prefer an explicit effort; only fall back to extra_body
|
||||
# thinking when no recognized effort is requested.
|
||||
# K3's vocabulary (low/high/max, default high) and its documented
|
||||
# rounding (medium→high, xhigh→max) are declared in
|
||||
# agent.reasoning_effort — shared with the chat-completions
|
||||
# transport's Kimi path so both stay in sync.
|
||||
from agent.reasoning_effort import (
|
||||
KIMI_K3_EFFORTS,
|
||||
KIMI_K3_OVERRIDES,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if effort and effort != "none":
|
||||
k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES)
|
||||
else:
|
||||
k3_effort = None
|
||||
"""Moonshot treats extra_body.thinking and reasoning_effort as mutually
|
||||
exclusive (400 on both): send effort when requested, else the toggle."""
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled", True) is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
effort = requested_effort(reasoning_config)
|
||||
k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES) if effort != "none" else None
|
||||
if k3_effort in KIMI_K3_EFFORTS:
|
||||
top_level["reasoning_effort"] = k3_effort
|
||||
else:
|
||||
extra_body["thinking"] = {"type": "enabled"}
|
||||
|
||||
return extra_body, top_level
|
||||
return {}, {"reasoning_effort": k3_effort}
|
||||
return {"thinking": {"type": "enabled"}}, {}
|
||||
|
||||
|
||||
kimi = KimiProfile(
|
||||
@@ -117,11 +68,7 @@ kimi = KimiProfile(
|
||||
base_url="https://api.moonshot.ai/v1",
|
||||
fixed_temperature=OMIT_TEMPERATURE,
|
||||
default_max_tokens=32000,
|
||||
default_headers={
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
},
|
||||
default_headers=dict(_HEADERS),
|
||||
default_aux_model="kimi-k2-turbo-preview",
|
||||
)
|
||||
|
||||
@@ -132,11 +79,7 @@ kimi_cn = KimiProfile(
|
||||
base_url="https://api.moonshot.cn/v1",
|
||||
fixed_temperature=OMIT_TEMPERATURE,
|
||||
default_max_tokens=32000,
|
||||
default_headers={
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
},
|
||||
default_headers=dict(_HEADERS),
|
||||
default_aux_model="kimi-k2-turbo-preview",
|
||||
)
|
||||
|
||||
|
||||
@@ -1,29 +1,9 @@
|
||||
"""Meta Model API (Muse Spark) provider plugin for Hermes Agent.
|
||||
"""Meta Model API (Muse Spark) provider profile — https://api.meta.ai/v1.
|
||||
|
||||
Provider profile for Meta Superintelligence Labs' Muse Spark family, served
|
||||
via the OpenAI-compatible Meta Model API at ``https://api.meta.ai/v1``.
|
||||
|
||||
Bundled from https://github.com/albertodepaola/hermes-meta-provider. Hermes'
|
||||
provider discovery (``providers/__init__.py``) imports it on first
|
||||
``get_provider_profile()`` / ``list_providers()`` call, and the module-level
|
||||
``register_provider()`` below wires it into the registry.
|
||||
|
||||
Design notes
|
||||
------------
|
||||
* **Zero core edits.** Everything rides on ``ProviderProfile`` hooks. No changes
|
||||
to hermes' ``model_metadata.py`` / ``models.py`` / ``run_agent.py`` are needed:
|
||||
- Context window (1M), reasoning and vision capabilities already resolve from
|
||||
models.dev for the muse-spark family, so no static ctx table entry is required.
|
||||
- The reasoning dial is emitted as a **top-level ``reasoning_effort``** kwarg
|
||||
(returned in the ``top_level`` slot of ``build_api_kwargs_extras``), which the
|
||||
chat-completions transport merges unconditionally. This deliberately avoids
|
||||
the ``extra_body.reasoning`` path, whose emission is gated by a hardcoded
|
||||
host allowlist in core (``AIAgent._supports_reasoning_extra_body``) that a
|
||||
third-party plugin must not edit.
|
||||
* **Meta 400 on ``reasoning_effort: "none"``.** Muse rejects ``none``; disabling
|
||||
reasoning maps to ``"minimal"`` instead.
|
||||
* **``default_max_tokens=16384``.** Muse spends completion budget on hidden
|
||||
reasoning tokens first; small caps can finish with empty content.
|
||||
Bundled from albertodepaola/hermes-meta-provider; rides entirely on
|
||||
ProviderProfile hooks (zero core edits). The reasoning dial is emitted as a
|
||||
top-level ``reasoning_effort`` kwarg — not ``extra_body.reasoning``, whose
|
||||
emission is gated by a core host allowlist a third-party plugin must not edit.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -31,30 +11,11 @@ from __future__ import annotations
|
||||
import os
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import META_AI_EFFORTS, clamp_effort
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
def _resolve_effort(reasoning_config: dict | None) -> str:
|
||||
"""Map Hermes' reasoning_config to a Meta-safe ``reasoning_effort`` value.
|
||||
|
||||
Meta's vocabulary (minimal..xhigh; rejects ``none``) is declared in
|
||||
agent.reasoning_effort. Disabled/"none" maps to ``minimal`` (the closest
|
||||
Meta has to off); unset/bespoke levels fall to ``medium``.
|
||||
"""
|
||||
rc = reasoning_config or {}
|
||||
if rc.get("enabled") is False:
|
||||
return "minimal"
|
||||
effort = str(rc.get("effort") or "").strip().lower()
|
||||
if effort in {"", "none"}:
|
||||
return "minimal" if effort == "none" else "medium"
|
||||
|
||||
from agent.reasoning_effort import META_AI_EFFORTS, clamp_effort
|
||||
|
||||
clamped = clamp_effort(effort, META_AI_EFFORTS)
|
||||
return clamped if clamped in META_AI_EFFORTS else "medium"
|
||||
|
||||
|
||||
class MetaAIProfile(ProviderProfile):
|
||||
"""Meta Model API — top-level reasoning_effort, self-contained."""
|
||||
|
||||
@@ -65,19 +26,20 @@ class MetaAIProfile(ProviderProfile):
|
||||
supports_reasoning: bool = False, # noqa: ARG002 — we self-gate below
|
||||
**context: Any,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Emit ``reasoning_effort`` as a top-level api kwarg.
|
||||
"""Ignores the core ``supports_reasoning`` gate (host-allowlist driven);
|
||||
Muse Spark always accepts ``reasoning_effort``.
|
||||
|
||||
We ignore the core ``supports_reasoning`` gate on purpose: that flag is
|
||||
driven by a host allowlist in core we cannot (and should not) edit from
|
||||
an out-of-tree plugin. Muse Spark always accepts ``reasoning_effort``,
|
||||
so we resolve it from ``reasoning_config`` directly.
|
||||
Muse 400s on ``none``: disabled/"none" -> ``minimal`` (closest to off);
|
||||
unset/bespoke levels -> ``medium``.
|
||||
"""
|
||||
return {}, {"reasoning_effort": _resolve_effort(reasoning_config)}
|
||||
|
||||
|
||||
def _base_url() -> str:
|
||||
"""Allow a base-URL override via ``META_BASE_URL`` without editing config."""
|
||||
return os.getenv("META_BASE_URL", "").strip() or "https://api.meta.ai/v1"
|
||||
rc = reasoning_config or {}
|
||||
effort = str(rc.get("effort") or "").strip().lower()
|
||||
if rc.get("enabled") is False or effort == "none":
|
||||
mapped = "minimal"
|
||||
else:
|
||||
clamped = clamp_effort(effort, META_AI_EFFORTS)
|
||||
mapped = clamped if clamped in META_AI_EFFORTS else "medium"
|
||||
return {}, {"reasoning_effort": mapped}
|
||||
|
||||
|
||||
meta_ai = MetaAIProfile(
|
||||
@@ -88,30 +50,18 @@ meta_ai = MetaAIProfile(
|
||||
signup_url="https://developer.meta.com/ai/",
|
||||
# MODEL_API_KEY is Meta's documented env var; the aliases are conveniences.
|
||||
env_vars=("MODEL_API_KEY", "META_API_KEY", "META_MODEL_API_KEY", "META_BASE_URL"),
|
||||
base_url=_base_url(),
|
||||
base_url=os.getenv("META_BASE_URL", "").strip() or "https://api.meta.ai/v1",
|
||||
auth_type="api_key",
|
||||
# Responses API is the wire that engages Muse prompt caching: measured
|
||||
# 0 cached tokens on /v1/chat/completions vs 93-99% cache hits on
|
||||
# /v1/responses with prompt_cache_retention (see host_mandated_api_mode
|
||||
# in hermes_cli/providers.py and the retention hint in
|
||||
# agent/transports/codex.py). The MetaAIProfile chat-completions hook
|
||||
# above still covers custom OpenAI-compatible endpoints configured with
|
||||
# a non-api.meta.ai base URL, which fall through to chat_completions.
|
||||
# Responses API engages Muse prompt caching (0 cached tokens on
|
||||
# chat/completions vs 93-99% hits on /v1/responses); the chat-completions
|
||||
# hook above still covers custom non-api.meta.ai base URLs.
|
||||
api_mode="codex_responses",
|
||||
# Muse Spark is natively multimodal (image/video/pdf/audio in, text out).
|
||||
supports_vision=True,
|
||||
# Cheap contributor tier is a good default for auxiliary tasks
|
||||
# (compaction, title generation, vision) when this is the main provider.
|
||||
default_aux_model="muse-spark-1.2-contributor",
|
||||
# Muse spends completion budget on hidden reasoning tokens first; a low cap
|
||||
# can finish with empty content. 16k is a safe floor.
|
||||
# Muse spends completion budget on hidden reasoning first; low caps can
|
||||
# finish with empty content.
|
||||
default_max_tokens=16384,
|
||||
# Curated safety net shown in the picker when the live /v1/models fetch
|
||||
# fails or no credentials are configured yet.
|
||||
fallback_models=(
|
||||
"muse-spark-1.2",
|
||||
"muse-spark-1.2-contributor",
|
||||
),
|
||||
fallback_models=("muse-spark-1.2", "muse-spark-1.2-contributor"),
|
||||
)
|
||||
|
||||
register_provider(meta_ai)
|
||||
|
||||
@@ -1,9 +1,8 @@
|
||||
"""MiniMax provider profiles (international + China).
|
||||
"""MiniMax provider profiles (international, China, OAuth).
|
||||
|
||||
The default API-key routes use anthropic_messages because their base URLs end
|
||||
with /anthropic. Users can opt MiniMax-M3 into the OpenAI-compatible endpoint
|
||||
with base_url=https://api.minimax.io/v1; that route needs MiniMax-specific
|
||||
reasoning controls in extra_body.
|
||||
Default routes use anthropic_messages (base URLs end in /anthropic). Users can
|
||||
opt MiniMax-M3 into the OpenAI-compatible https://api.minimax.io/v1 route,
|
||||
which needs MiniMax-specific reasoning controls in extra_body.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
@@ -15,15 +14,7 @@ from providers.base import ProviderProfile
|
||||
|
||||
def _is_minimax_global_openai_base_url(base_url: str | None) -> bool:
|
||||
parsed = urlparse(str(base_url or "").strip())
|
||||
if (parsed.hostname or "").lower() != "api.minimax.io":
|
||||
return False
|
||||
path = parsed.path.rstrip("/").lower()
|
||||
return path == "/v1"
|
||||
|
||||
|
||||
def _is_minimax_m3(model: str | None) -> bool:
|
||||
normalized = str(model or "").strip().lower()
|
||||
return normalized in {"minimax-m3", "minimax/minimax-m3"}
|
||||
return (parsed.hostname or "").lower() == "api.minimax.io" and parsed.path.rstrip("/").lower() == "/v1"
|
||||
|
||||
|
||||
class MiniMaxProfile(ProviderProfile):
|
||||
@@ -37,25 +28,16 @@ class MiniMaxProfile(ProviderProfile):
|
||||
base_url: str | None = None,
|
||||
**context: Any,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Emit M3 reasoning controls for api.minimax.io/v1.
|
||||
|
||||
MiniMax-M3's OpenAI-compatible endpoint keeps thinking inline unless
|
||||
``reasoning_split`` is sent, so always request the split format on that
|
||||
route. ``thinking`` controls the M3 mode; Hermes' effort levels are not
|
||||
a MiniMax depth knob here, so they only select adaptive vs disabled.
|
||||
"""
|
||||
if not _is_minimax_global_openai_base_url(base_url) or not _is_minimax_m3(model):
|
||||
"""M3 on api.minimax.io/v1 keeps thinking inline unless ``reasoning_split``
|
||||
is sent; effort levels only select adaptive vs disabled ``thinking``."""
|
||||
is_m3 = str(model or "").strip().lower() in {"minimax-m3", "minimax/minimax-m3"}
|
||||
if not _is_minimax_global_openai_base_url(base_url) or not is_m3:
|
||||
return {}, {}
|
||||
|
||||
extra_body: dict[str, Any] = {"reasoning_split": True}
|
||||
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
extra_body["thinking"] = {"type": "disabled"}
|
||||
return extra_body, {}
|
||||
|
||||
if reasoning_config is not None:
|
||||
elif reasoning_config is not None:
|
||||
extra_body["thinking"] = {"type": "adaptive"}
|
||||
|
||||
return extra_body, {}
|
||||
|
||||
|
||||
|
||||
@@ -4,33 +4,21 @@ from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
def _flat_model_name(model: str | None) -> str:
|
||||
"""Return a lowercase model id, tolerating vendor-prefixed IDs."""
|
||||
return (model or "").strip().rsplit("/", 1)[-1].lower()
|
||||
|
||||
|
||||
def _model_supports_reasoning_effort(model: str | None) -> bool:
|
||||
"""Conservative allowlist for Nebius models that expose reasoning effort."""
|
||||
model_name = _flat_model_name(model)
|
||||
if not model_name:
|
||||
return False
|
||||
return any(
|
||||
marker in model_name
|
||||
for marker in (
|
||||
"deepseek-r1",
|
||||
"deepseek-v4",
|
||||
"deepseek-reasoner",
|
||||
"gpt-oss",
|
||||
"glm-5",
|
||||
"kimi-k2",
|
||||
"minimax-m2",
|
||||
"qwen3",
|
||||
)
|
||||
)
|
||||
# Conservative allowlist of model families that expose reasoning effort.
|
||||
_REASONING_MARKERS = (
|
||||
"deepseek-r1",
|
||||
"deepseek-v4",
|
||||
"deepseek-reasoner",
|
||||
"gpt-oss",
|
||||
"glm-5",
|
||||
"kimi-k2",
|
||||
"minimax-m2",
|
||||
"qwen3",
|
||||
)
|
||||
|
||||
|
||||
class NebiusTokenFactoryProfile(ProviderProfile):
|
||||
@@ -44,46 +32,25 @@ class NebiusTokenFactoryProfile(ProviderProfile):
|
||||
supports_reasoning: bool = False,
|
||||
**context: Any,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
if not supports_reasoning and not _model_supports_reasoning_effort(model):
|
||||
model_name = (model or "").strip().rsplit("/", 1)[-1].lower()
|
||||
if not supports_reasoning and not any(marker in model_name for marker in _REASONING_MARKERS):
|
||||
return {}, {}
|
||||
|
||||
if isinstance(reasoning_config, dict):
|
||||
enabled = reasoning_config.get("enabled", True)
|
||||
raw_effort = reasoning_config.get("effort", "medium")
|
||||
else:
|
||||
enabled = True
|
||||
raw_effort = "medium"
|
||||
|
||||
effort = str(raw_effort or "medium").strip().lower()
|
||||
if enabled is False or effort in {"none", "off", "disabled"}:
|
||||
rc = reasoning_config if isinstance(reasoning_config, dict) else {}
|
||||
# Unset/blank effort defaults to medium (reasoning ON).
|
||||
effort = str(rc.get("effort", "medium") or "medium").strip().lower()
|
||||
if rc.get("enabled", True) is False or effort in {"none", "off", "disabled"}:
|
||||
return {}, {}
|
||||
# Canonical clamp (nearest weaker supported level, never escalate,
|
||||
# monotonic) — the hand-rolled map this replaces inverted the ladder:
|
||||
# ultra fell through to medium while xhigh mapped to high.
|
||||
from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort
|
||||
|
||||
effort = clamp_effort(effort, NEBIUS_EFFORTS) or "medium"
|
||||
|
||||
return {}, {"reasoning_effort": effort}
|
||||
# Canonical clamp: nearest weaker supported level, never escalate.
|
||||
return {}, {"reasoning_effort": clamp_effort(effort, NEBIUS_EFFORTS) or "medium"}
|
||||
|
||||
|
||||
nebius_token_factory = NebiusTokenFactoryProfile(
|
||||
name="nebius-token-factory",
|
||||
aliases=(
|
||||
"nebius",
|
||||
"nebius-tokenfactory",
|
||||
"nebius-tf",
|
||||
"token-factory",
|
||||
"tokenfactory",
|
||||
),
|
||||
aliases=("nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory"),
|
||||
display_name="Nebius Token Factory",
|
||||
description="Nebius Token Factory — OpenAI-compatible inference",
|
||||
signup_url="https://tokenfactory.nebius.com/",
|
||||
env_vars=(
|
||||
"NEBIUS_API_KEY",
|
||||
"NEBIUS_TOKEN_FACTORY_API_KEY",
|
||||
"NEBIUS_BASE_URL",
|
||||
),
|
||||
env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY", "NEBIUS_BASE_URL"),
|
||||
base_url="https://api.tokenfactory.nebius.com/v1",
|
||||
models_url="https://api.tokenfactory.nebius.com/v1/models?verbose=true",
|
||||
auth_type="api_key",
|
||||
|
||||
@@ -16,14 +16,7 @@ class NousProfile(ProviderProfile):
|
||||
"""Nous Portal — product tags, reasoning with Nous-specific omission."""
|
||||
|
||||
def resolve_aux_model(self, *, vision: bool = False) -> str:
|
||||
"""Ask the Portal which cheap model it currently recommends.
|
||||
|
||||
``/api/nous/recommended-models`` is the authoritative, tier-aware
|
||||
source (free vs paid), so the auxiliary fast tier tracks the live
|
||||
catalog instead of a hardcoded id that 404s the day Nous retires it.
|
||||
The underlying fetch is memory- and disk-cached with a last-known-good
|
||||
fallback, so this is cheap to call and safe offline.
|
||||
"""
|
||||
"""Portal's tier-aware ``/api/nous/recommended-models`` pick (cached, offline-safe)."""
|
||||
try:
|
||||
from hermes_cli.models import get_nous_recommended_aux_model
|
||||
|
||||
@@ -31,37 +24,12 @@ class NousProfile(ProviderProfile):
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
def build_extra_body(
|
||||
self, *, session_id: str | None = None, **context
|
||||
) -> dict[str, Any]:
|
||||
def build_extra_body(self, *, session_id: str | None = None, **context) -> dict[str, Any]:
|
||||
body: dict[str, Any] = {"tags": nous_portal_tags(session_id=session_id)}
|
||||
# Top-level session_id → provider sticky routing key. Pins every
|
||||
# turn of a session to the same upstream endpoint so explicit
|
||||
# Anthropic cache_control breakpoints stay warm instead of
|
||||
# cold-writing a fresh cache on each reroute (Anthropic/Vertex/
|
||||
# Bedrock caches are instance-local). Mirrors the OpenRouter
|
||||
# profile; without it the portal falls back to hashing the opening
|
||||
# messages, which breaks pinning whenever those shift.
|
||||
#
|
||||
# Resolve it exactly like ``nous_portal_tags`` resolves the
|
||||
# ``conversation=`` tag: ambient context first (the lineage ROOT id
|
||||
# published by the agent loop), explicit argument as fallback.
|
||||
#
|
||||
# The gap this closes is the auxiliary call sites — compression,
|
||||
# title generation, vision, web_extract, session_search, MoA slots.
|
||||
# They funnel through ``agent.auxiliary_client`` which has no session
|
||||
# handle, so they never pass ``session_id``: they carried the
|
||||
# ``conversation=`` tag but NO sticky key at all, and each one routed
|
||||
# independently of the conversation it belongs to. Reading the same
|
||||
# ambient contextvar the tag already uses fixes that with zero
|
||||
# per-call-site plumbing; a host-declared routing scope (#96811) wins
|
||||
# over it when one was published for this turn.
|
||||
#
|
||||
# For the main loop the two agree anyway under the default
|
||||
# ``compression.in_place: true`` (#38763), where compaction keeps the
|
||||
# session id; the ambient root additionally keeps the key stable for
|
||||
# installs that opt back into rotating compaction, and across
|
||||
# delegate-subagent trees.
|
||||
# Top-level session_id = sticky routing key, so Anthropic-style cache
|
||||
# breakpoints stay warm on one upstream instance. Resolved like the
|
||||
# ``conversation=`` tag: declared scope, then the ambient lineage ROOT
|
||||
# (covers aux call sites that pass no session_id), then the explicit argument.
|
||||
sticky_key = _cache_scope_from_session_id(
|
||||
get_affinity_scope() or get_conversation_context() or session_id
|
||||
)
|
||||
@@ -74,23 +42,13 @@ class NousProfile(ProviderProfile):
|
||||
|
||||
@staticmethod
|
||||
def _cannot_disable_reasoning(model: str | None) -> bool:
|
||||
"""True when a disable can't safely be sent for *model*.
|
||||
"""True when ``reasoning: {enabled: false}`` would 400 on *model*.
|
||||
|
||||
Reasoning-mandatory routes answer ``reasoning: {enabled: false}``
|
||||
with HTTP 400 ("Reasoning is mandatory for this model"), so the
|
||||
catalog decides. Cache-only, and an unknown model (catalog cold,
|
||||
unlisted, or unreachable) also answers True: a cold first turn errs
|
||||
toward the old omit-everything behavior rather than risking a 400.
|
||||
|
||||
A route the catalog says takes no reasoning parameter at all is
|
||||
treated the same way — sending it a disable is sending a parameter
|
||||
the Portal has told us it doesn't accept.
|
||||
Cache-only catalog lookup; unknown/cold (warmer kicked) and
|
||||
no-reasoning-parameter routes both answer True (omit rather than risk a 400).
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.models import (
|
||||
nous_model_reasoning_capabilities,
|
||||
warm_nous_reasoning_caps_async,
|
||||
)
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async
|
||||
|
||||
caps = nous_model_reasoning_capabilities(model)
|
||||
if caps is None:
|
||||
@@ -98,9 +56,7 @@ class NousProfile(ProviderProfile):
|
||||
return True
|
||||
except Exception:
|
||||
return True
|
||||
if not caps.get("supports_reasoning"):
|
||||
return True
|
||||
return bool(caps.get("mandatory"))
|
||||
return not caps.get("supports_reasoning") or bool(caps.get("mandatory"))
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self,
|
||||
@@ -110,25 +66,16 @@ class NousProfile(ProviderProfile):
|
||||
model: str | None = None,
|
||||
**context,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Nous: passes the full reasoning_config, disable included.
|
||||
|
||||
The Portal honors ``reasoning: {enabled: false}`` — it is the only
|
||||
wire shape that does. Sending nothing means the *upstream* default,
|
||||
which for a thinking-first model like ``deepseek/deepseek-v4-pro``
|
||||
(catalog: ``default_effort: high``) is thinking ON, so omitting a
|
||||
disable silently ignored the user's "thinking off".
|
||||
"""
|
||||
extra_body = {}
|
||||
if supports_reasoning:
|
||||
if reasoning_config is not None:
|
||||
rc = dict(reasoning_config)
|
||||
if rc.get("enabled") is False and self._cannot_disable_reasoning(model):
|
||||
pass # route rejects a disable — let the model think
|
||||
else:
|
||||
extra_body["reasoning"] = rc
|
||||
else:
|
||||
extra_body["reasoning"] = {"enabled": True, "effort": "medium"}
|
||||
return extra_body, {}
|
||||
"""Pass the full reasoning_config, disable included (the Portal honors it;
|
||||
omitting it means the upstream default, thinking ON for V4-class models)."""
|
||||
if not supports_reasoning:
|
||||
return {}, {}
|
||||
if reasoning_config is None:
|
||||
return {"reasoning": {"enabled": True, "effort": "medium"}}, {}
|
||||
rc = dict(reasoning_config)
|
||||
if rc.get("enabled") is False and self._cannot_disable_reasoning(model):
|
||||
return {}, {}
|
||||
return {"reasoning": rc}, {}
|
||||
|
||||
|
||||
nous = NousProfile(
|
||||
@@ -138,10 +85,7 @@ nous = NousProfile(
|
||||
display_name="Nous Research",
|
||||
description="Nous Research — Hermes model family",
|
||||
signup_url="https://nousresearch.com/",
|
||||
fallback_models=(
|
||||
"hermes-3-405b",
|
||||
"hermes-3-70b",
|
||||
),
|
||||
fallback_models=("hermes-3-405b", "hermes-3-70b"),
|
||||
base_url="https://inference-api.nousresearch.com/v1",
|
||||
auth_type="oauth_device_code",
|
||||
)
|
||||
|
||||
@@ -9,36 +9,23 @@ from providers.base import ProviderProfile
|
||||
class NvidiaProviderProfile(ProviderProfile):
|
||||
"""NVIDIA NIM accepts a stricter ToolMessage schema than most OpenAI-compatible APIs."""
|
||||
|
||||
def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
needs_sanitize = any(
|
||||
@staticmethod
|
||||
def _needs_strip(msg: Any) -> bool:
|
||||
return (
|
||||
isinstance(msg, dict)
|
||||
and msg.get("role") == "tool"
|
||||
and ("name" in msg or "tool_name" in msg)
|
||||
for msg in messages
|
||||
)
|
||||
if not needs_sanitize:
|
||||
return messages
|
||||
|
||||
# Copy-on-write: shallow outer-list copy, then a shallow dict copy
|
||||
# only for the role:"tool" messages that actually need a field
|
||||
# dropped. Avoids recursively deep-copying every message's content
|
||||
# (including large tool outputs and attachments) for a turn that
|
||||
# only ever needs to touch two top-level keys on a handful of
|
||||
# messages. Matches the pattern already used by the shared
|
||||
# sanitizer in agent/transports/chat_completions.py and by
|
||||
# QwenProfile.prepare_messages().
|
||||
sanitized = list(messages)
|
||||
for idx, msg in enumerate(messages):
|
||||
if (
|
||||
isinstance(msg, dict)
|
||||
and msg.get("role") == "tool"
|
||||
and ("name" in msg or "tool_name" in msg)
|
||||
):
|
||||
msg_copy = dict(msg)
|
||||
msg_copy.pop("name", None)
|
||||
msg_copy.pop("tool_name", None)
|
||||
sanitized[idx] = msg_copy
|
||||
return sanitized
|
||||
def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""Copy-on-write: only tool messages that lose a field are copied
|
||||
(no deep copy of large tool outputs); untouched input returned as-is."""
|
||||
if not any(self._needs_strip(msg) for msg in messages):
|
||||
return messages
|
||||
return [
|
||||
{k: v for k, v in msg.items() if k not in ("name", "tool_name")} if self._needs_strip(msg) else msg
|
||||
for msg in messages
|
||||
]
|
||||
|
||||
|
||||
nvidia = NvidiaProviderProfile(
|
||||
@@ -48,10 +35,7 @@ nvidia = NvidiaProviderProfile(
|
||||
display_name="NVIDIA NIM",
|
||||
description="NVIDIA NIM — accelerated inference",
|
||||
signup_url="https://build.nvidia.com/",
|
||||
fallback_models=(
|
||||
"nvidia/llama-3.1-nemotron-70b-instruct",
|
||||
"nvidia/llama-3.3-70b-instruct",
|
||||
),
|
||||
fallback_models=("nvidia/llama-3.1-nemotron-70b-instruct", "nvidia/llama-3.3-70b-instruct"),
|
||||
base_url="https://integrate.api.nvidia.com/v1",
|
||||
default_max_tokens=16384,
|
||||
)
|
||||
|
||||
@@ -1,24 +1,15 @@
|
||||
"""Ollama Cloud provider profile.
|
||||
|
||||
Ollama Cloud's OpenAI-compatible ``/v1/chat/completions`` endpoint
|
||||
supports top-level ``reasoning_effort`` with values ``none``, ``low``,
|
||||
``medium``, ``high``, and ``max`` (the last being undocumented but
|
||||
empirically confirmed for DeepSeek V4 — ``max`` produces ~2.5× more
|
||||
thinking tokens than ``high``).
|
||||
|
||||
This profile maps Hermes's ``xhigh`` → ``max`` to unlock DeepSeek V4's
|
||||
"Max thinking" tier through Ollama Cloud. ``low`` / ``medium`` / ``high``
|
||||
pass through unchanged.
|
||||
|
||||
When reasoning is explicitly disabled (``enabled: false`` or
|
||||
``effort: "none"``), ``reasoning_effort`` is omitted entirely so the
|
||||
model runs in non-thinking mode.
|
||||
Top-level ``reasoning_effort`` on /v1/chat/completions accepts none|low|medium|
|
||||
high|max (``max`` is undocumented but real — ~2.5x more thinking tokens on
|
||||
DeepSeek V4); Hermes' ``xhigh`` maps to ``max``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES, clamp_effort
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
@@ -27,61 +18,23 @@ class OllamaCloudProfile(ProviderProfile):
|
||||
"""Ollama Cloud — maps xhigh→max via top-level reasoning_effort."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self,
|
||||
*,
|
||||
reasoning_config: dict | None = None,
|
||||
supports_reasoning: bool = False,
|
||||
**ctx: Any,
|
||||
self, *, reasoning_config: dict | None = None, supports_reasoning: bool = False, **ctx: Any
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Emit top-level ``reasoning_effort`` for Ollama Cloud thinking models.
|
||||
|
||||
Gated on ``supports_reasoning``, which the transport resolves from the
|
||||
model's native ``/api/show`` ``capabilities`` (``thinking``). Models
|
||||
without the thinking capability (e.g. ``gemma3``, ``qwen3-coder``) get
|
||||
no ``reasoning_effort`` at all — emitting it there is a no-op the API
|
||||
ignores, and gating avoids sending a meaningless field.
|
||||
"""
|
||||
top_level: dict[str, Any] = {}
|
||||
|
||||
if not supports_reasoning:
|
||||
"""Gated on ``supports_reasoning`` (resolved from the model's /api/show
|
||||
``thinking`` capability) so non-thinking models get no meaningless field."""
|
||||
if not supports_reasoning or not reasoning_config or not isinstance(reasoning_config, dict):
|
||||
return {}, {}
|
||||
|
||||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
enabled = reasoning_config.get("enabled", True)
|
||||
if enabled is False:
|
||||
# Ollama Cloud defaults to thinking ON, and ignores the
|
||||
# extra_body.thinking:{type:disabled} shape (verified live).
|
||||
# The ONLY way to actually suppress thinking on its
|
||||
# /v1/chat/completions endpoint is top-level
|
||||
# reasoning_effort:"none" — omitting the field leaves
|
||||
# thinking on.
|
||||
return {}, {"reasoning_effort": "none"}
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if not effort:
|
||||
# No explicit effort requested — let the model decide
|
||||
# (Ollama Cloud's server default is thinking ON).
|
||||
return {}, {}
|
||||
if effort == "none":
|
||||
return {}, {"reasoning_effort": "none"} # explicit off switch
|
||||
# Accepted set {none, low, medium, high, max} is declared in
|
||||
# agent.reasoning_effort ("minimal" is rejected with HTTP 400 →
|
||||
# clamps to low; xhigh rounds up to max). Bespoke levels outside
|
||||
# the ladder are omitted so the model applies its own default
|
||||
# rather than triggering a hard 400.
|
||||
from agent.reasoning_effort import (
|
||||
OLLAMA_CLOUD_EFFORTS,
|
||||
OLLAMA_CLOUD_OVERRIDES,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
clamped = clamp_effort(
|
||||
effort, OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES
|
||||
)
|
||||
if clamped in OLLAMA_CLOUD_EFFORTS:
|
||||
top_level["reasoning_effort"] = clamped
|
||||
|
||||
return {}, top_level
|
||||
# Ollama Cloud defaults to thinking ON and ignores extra_body.thinking
|
||||
# (verified live); top-level reasoning_effort:"none" is the ONLY off switch.
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if reasoning_config.get("enabled", True) is False or effort == "none":
|
||||
return {}, {"reasoning_effort": "none"}
|
||||
if not effort:
|
||||
return {}, {} # let the server default (thinking ON) apply
|
||||
# "minimal" 400s -> clamps to low; xhigh rounds up to max. Bespoke
|
||||
# levels outside the ladder are omitted rather than risking a 400.
|
||||
clamped = clamp_effort(effort, OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES)
|
||||
return {}, {"reasoning_effort": clamped} if clamped in OLLAMA_CLOUD_EFFORTS else {}
|
||||
|
||||
|
||||
ollama_cloud = OllamaCloudProfile(
|
||||
|
||||
@@ -1,12 +1,9 @@
|
||||
"""OpenCode Free provider profile.
|
||||
"""OpenCode Free provider profile: the free tier on the Zen relay (https://opencode.ai/zen/v1).
|
||||
|
||||
OpenCode's free model tier on the Zen relay (https://opencode.ai/zen/v1).
|
||||
KEYLESS: the relay serves free-tier models anonymously and rejects any
|
||||
Authorization bearer it doesn't recognize with 401 — so this provider
|
||||
never sends a credential at all (the runtime resolver pins the keyless
|
||||
placeholder and an empty Authorization header; see
|
||||
hermes_cli.models.opencode_zen_free_runtime). No OpenCode account needed.
|
||||
Select via ``hermes model`` or ``/model free``.
|
||||
KEYLESS: the relay serves free-tier models anonymously and 401s any bearer it
|
||||
doesn't recognize, so this provider never sends a credential (the runtime
|
||||
resolver pins the keyless placeholder and an empty Authorization header; see
|
||||
hermes_cli.models.opencode_zen_free_runtime). Select via ``/model free``.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
@@ -15,25 +12,13 @@ from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
# Attribution headers, same values as the opencode-zen/go profiles, plus the
|
||||
# empty Authorization override that keeps the SDK's "Bearer <placeholder>"
|
||||
# off the wire (the free tier 401s any unrecognized bearer).
|
||||
_KEYLESS_HEADERS = {
|
||||
"Authorization": "",
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
}
|
||||
|
||||
|
||||
class OpenCodeFreeProfile(ProviderProfile):
|
||||
"""OpenCode Free — keyless, with Ox Alpha reasoning controls.
|
||||
|
||||
Ox Alpha (x-preview-f-free) is reachable through this provider as well
|
||||
as opencode-zen; both share the same wire contract (reasoning_effort
|
||||
accepts exactly low/high/max — anything else 400s). The translation
|
||||
lives in the zen plugin; resolve it through the registered zen profile's
|
||||
module so the two providers can never drift.
|
||||
Ox Alpha (x-preview-f-free) is also reachable via opencode-zen with the same
|
||||
wire contract; the translation lives in the zen plugin and is resolved through
|
||||
the registered zen profile's module so the two providers can never drift.
|
||||
"""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
@@ -44,8 +29,7 @@ class OpenCodeFreeProfile(ProviderProfile):
|
||||
|
||||
from providers import get_provider_profile
|
||||
|
||||
zen_profile = get_provider_profile("opencode-zen")
|
||||
zen_module = sys.modules[type(zen_profile).__module__]
|
||||
zen_module = sys.modules[type(get_provider_profile("opencode-zen")).__module__]
|
||||
return zen_module._build_ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
except Exception:
|
||||
return {}, {}
|
||||
@@ -58,9 +42,17 @@ opencode_free = OpenCodeFreeProfile(
|
||||
base_url="https://opencode.ai/zen/v1",
|
||||
display_name="OpenCode Free",
|
||||
description="OpenCode free models — keyless, no account needed",
|
||||
default_headers=dict(_KEYLESS_HEADERS),
|
||||
# Attribution headers (same values as opencode-zen/go) plus the empty
|
||||
# Authorization override that keeps the SDK's "Bearer <placeholder>" off the
|
||||
# wire (the free tier 401s any unrecognized bearer).
|
||||
default_headers={
|
||||
"Authorization": "",
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
},
|
||||
# laguna is the fastest non-UA-gated free model; big-pickle 429s every
|
||||
# client except the opencode CLI's own User-Agent (verified 2026-08-21).
|
||||
# client except the opencode CLI's own User-Agent.
|
||||
default_aux_model="laguna-s-2.1-free",
|
||||
)
|
||||
|
||||
|
||||
@@ -1,25 +1,20 @@
|
||||
"""OpenCode provider profiles (Zen + Go).
|
||||
|
||||
Both use per-model api_mode routing:
|
||||
- OpenCode Zen: Claude → anthropic_messages, GPT-5/Codex/Grok → codex_responses,
|
||||
Muse Spark → codex_responses, everything else → chat_completions (this profile)
|
||||
- OpenCode Go: GPT / Grok / Muse Spark → codex_responses, MiniMax/Qwen → anthropic_messages,
|
||||
GLM/Kimi/DeepSeek/MiMo → chat_completions (this profile)
|
||||
Both route api_mode per model in core; these profiles carry the
|
||||
chat_completions reasoning translations (GLM-5.2, Kimi K2, DeepSeek, Ox Alpha).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent import reasoning_effort as re_
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
# Attribution headers sent on every OpenCode request. Same values we send
|
||||
# to OpenRouter, Vercel AI Gateway, and Fireworks. Going through
|
||||
# profile.default_headers means they survive model switches and credential
|
||||
# rotation. Without them OpenCode only sees the OpenAI SDK's generic
|
||||
# "OpenAI/Python x.y.z" User-Agent and can't tell the traffic is Hermes Agent.
|
||||
# Attribution headers (same values as OpenRouter / Vercel / Fireworks); via
|
||||
# default_headers so they survive model switches and credential rotation.
|
||||
_ATTRIBUTION_HEADERS = {
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
@@ -32,15 +27,9 @@ def _flat_model_name(model: str | None) -> str:
|
||||
return (model or "").strip().rsplit("/", 1)[-1].lower()
|
||||
|
||||
|
||||
def _is_kimi_k2_model(model: str | None) -> bool:
|
||||
return _flat_model_name(model).startswith("kimi-k2")
|
||||
|
||||
|
||||
def _is_deepseek_thinking_model(model: str | None) -> bool:
|
||||
m = _flat_model_name(model)
|
||||
if m.startswith("deepseek-v") and not m.startswith("deepseek-v3"):
|
||||
return True
|
||||
return m == "deepseek-reasoner"
|
||||
return (m.startswith("deepseek-v") and not m.startswith("deepseek-v3")) or m == "deepseek-reasoner"
|
||||
|
||||
|
||||
def _is_glm_5_2_model(model: str | None) -> bool:
|
||||
@@ -49,143 +38,70 @@ def _is_glm_5_2_model(model: str | None) -> bool:
|
||||
return any(token in m for token in ("glm-5.2", "glm-5-2", "glm-5p2"))
|
||||
|
||||
|
||||
def _requested_effort(reasoning_config: dict | None) -> str | None:
|
||||
"""Normalized effort when reasoning is enabled and an effort is set, else None."""
|
||||
effort = re_.requested_effort(reasoning_config)
|
||||
return None if effort == "none" else effort
|
||||
|
||||
|
||||
def _thinking_toggle_extras(
|
||||
reasoning_config: dict | None, efforts, overrides=None
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Moonshot/DeepSeek wire shape: extra_body.thinking XOR top-level reasoning_effort
|
||||
(sending both is an HTTP 400)."""
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
clamped = re_.clamp_effort(_requested_effort(reasoning_config), efforts, overrides)
|
||||
if clamped in efforts:
|
||||
return {}, {"reasoning_effort": clamped}
|
||||
return {"thinking": {"type": "enabled"}}, {}
|
||||
|
||||
|
||||
class OpenCodeGoProfile(ProviderProfile):
|
||||
"""OpenCode Go - model-specific reasoning controls."""
|
||||
|
||||
# Per-model completion-token cap. The opencode-go relay's default is
|
||||
# too large for mimo-v2.5-pro — it sends max_tokens=262144 but Xiaomi
|
||||
# only supports 131072 completion tokens and 400s the request.
|
||||
# Setting an explicit cap here prevents the relay default from being
|
||||
# applied. Keys are normalized via _flat_model_name().
|
||||
# The relay's default max_tokens (262144) exceeds what Xiaomi accepts for
|
||||
# mimo-v2.5-pro and 400s; keys are normalized via _flat_model_name().
|
||||
_MODEL_MAX_TOKENS: dict[str, int] = {
|
||||
"mimo-v2.5-pro": 131072,
|
||||
}
|
||||
|
||||
def get_max_tokens(self, model: str | None) -> int | None:
|
||||
cap = self._MODEL_MAX_TOKENS.get(_flat_model_name(model))
|
||||
if cap is not None:
|
||||
return cap
|
||||
return self.default_max_tokens
|
||||
return self.default_max_tokens if cap is None else cap
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
top_level: dict[str, Any] = {}
|
||||
|
||||
if _is_glm_5_2_model(model):
|
||||
# GLM-5.2 on OpenCode Go uses its native OpenAI-compatible
|
||||
# reasoning_effort knob (high/max — declared in
|
||||
# agent.reasoning_effort, shared with the zai profile); leave the
|
||||
# server default alone when reasoning is disabled or unset.
|
||||
# Native reasoning_effort knob (high/max); server default when unset/disabled.
|
||||
effort = _requested_effort(reasoning_config)
|
||||
if effort is None:
|
||||
return {}, {}
|
||||
clamped = re_.clamp_effort(effort, re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES)
|
||||
return {}, {"reasoning_effort": clamped if clamped in re_.GLM52_EFFORTS else "high"}
|
||||
if _flat_model_name(model).startswith("kimi-k2"):
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return extra_body, top_level
|
||||
if reasoning_config.get("enabled") is False:
|
||||
return extra_body, top_level
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if not effort or effort == "none":
|
||||
return extra_body, top_level
|
||||
from agent.reasoning_effort import (
|
||||
GLM52_EFFORTS,
|
||||
GLM52_OVERRIDES,
|
||||
clamp_effort,
|
||||
return {}, {}
|
||||
return _thinking_toggle_extras(reasoning_config, re_.KIMI_K2_EFFORTS)
|
||||
if _is_deepseek_thinking_model(model):
|
||||
return _thinking_toggle_extras(
|
||||
reasoning_config, re_.DEEPSEEK_V4_EFFORTS, re_.DEEPSEEK_V4_OVERRIDES
|
||||
)
|
||||
|
||||
clamped = clamp_effort(effort, GLM52_EFFORTS, GLM52_OVERRIDES)
|
||||
top_level["reasoning_effort"] = (
|
||||
clamped if clamped in GLM52_EFFORTS else "high"
|
||||
)
|
||||
return extra_body, top_level
|
||||
|
||||
if _is_kimi_k2_model(model):
|
||||
# Kimi K2 on OpenCode Go uses Moonshot's native wire shape:
|
||||
# extra_body.thinking (binary toggle) + top-level reasoning_effort
|
||||
# (low|medium|high). Mirrors the KimiProfile (api.moonshot.ai/v1).
|
||||
if not isinstance(reasoning_config, dict):
|
||||
# No config → leave server defaults alone.
|
||||
return extra_body, top_level
|
||||
|
||||
enabled = reasoning_config.get("enabled") is not False
|
||||
if not enabled:
|
||||
extra_body["thinking"] = {"type": "disabled"}
|
||||
return extra_body, top_level
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if effort and effort != "none":
|
||||
from agent.reasoning_effort import KIMI_K2_EFFORTS, clamp_effort
|
||||
|
||||
clamped = clamp_effort(effort, KIMI_K2_EFFORTS)
|
||||
if clamped in KIMI_K2_EFFORTS:
|
||||
top_level["reasoning_effort"] = clamped
|
||||
|
||||
# Avoid "cannot specify both 'thinking' and 'reasoning_effort'" HTTP 400:
|
||||
# only send extra_body["thinking"] when no reasoning_effort is set.
|
||||
if "reasoning_effort" not in top_level:
|
||||
extra_body["thinking"] = {"type": "enabled"}
|
||||
return extra_body, top_level
|
||||
|
||||
if not _is_deepseek_thinking_model(model):
|
||||
return extra_body, top_level
|
||||
|
||||
enabled = True
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
enabled = False
|
||||
|
||||
if not enabled:
|
||||
extra_body["thinking"] = {"type": "disabled"}
|
||||
return extra_body, top_level
|
||||
|
||||
if isinstance(reasoning_config, dict):
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if effort and effort != "none":
|
||||
from agent.reasoning_effort import (
|
||||
DEEPSEEK_V4_EFFORTS,
|
||||
DEEPSEEK_V4_OVERRIDES,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
clamped = clamp_effort(
|
||||
effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES
|
||||
)
|
||||
if clamped in DEEPSEEK_V4_EFFORTS:
|
||||
top_level["reasoning_effort"] = clamped
|
||||
|
||||
# Avoid "cannot specify both 'thinking' and 'reasoning_effort'" HTTP 400:
|
||||
# only send extra_body["thinking"] when no reasoning_effort is set.
|
||||
if "reasoning_effort" not in top_level:
|
||||
extra_body["thinking"] = {"type": "enabled"}
|
||||
|
||||
return extra_body, top_level
|
||||
return {}, {}
|
||||
|
||||
|
||||
def _build_ox_alpha_reasoning_extras(
|
||||
reasoning_config: dict | None, model: str | None
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Shared Ox Alpha (x-preview-f-free) reasoning_effort translation.
|
||||
|
||||
Used by both the opencode-zen profile and the opencode-free keyless
|
||||
profile — the model is reachable through either provider and the wire
|
||||
contract is identical (low/high/max only; anything else 400s).
|
||||
"""
|
||||
"""Ox Alpha (x-preview-f-free) reasoning_effort translation, shared with the
|
||||
opencode-free profile (low/high/max only; anything else 400s)."""
|
||||
if _flat_model_name(model) != "x-preview-f-free":
|
||||
return {}, {}
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return {}, {}
|
||||
if reasoning_config.get("enabled") is False:
|
||||
return {}, {}
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if not effort or effort == "none":
|
||||
return {}, {}
|
||||
|
||||
from agent.reasoning_effort import (
|
||||
OX_ALPHA_EFFORTS,
|
||||
OX_ALPHA_OVERRIDES,
|
||||
clamp_effort,
|
||||
clamped = re_.clamp_effort(
|
||||
_requested_effort(reasoning_config), re_.OX_ALPHA_EFFORTS, re_.OX_ALPHA_OVERRIDES
|
||||
)
|
||||
|
||||
clamped = clamp_effort(effort, OX_ALPHA_EFFORTS, OX_ALPHA_OVERRIDES)
|
||||
if clamped not in OX_ALPHA_EFFORTS:
|
||||
if clamped not in re_.OX_ALPHA_EFFORTS:
|
||||
return {}, {}
|
||||
return {}, {"reasoning_effort": clamped}
|
||||
|
||||
|
||||
@@ -12,14 +12,10 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
_CACHE: list[str] | None = None
|
||||
|
||||
# Anthropic model families that still accept an explicit "disable thinking"
|
||||
# request (the manual ``thinking: {type: "disabled"}`` form OpenRouter emits
|
||||
# for ``reasoning: {enabled: false}``). Everything Claude 4.6 and newer —
|
||||
# including future date-stamped / named models (fable, mythos-class, …) —
|
||||
# mandates reasoning and returns HTTP 400 on any disable form. We therefore
|
||||
# default *unknown* Anthropic models to "cannot disable" (the modern contract)
|
||||
# and keep only this explicit legacy allowlist of models that can. Mirrors the
|
||||
# default-to-newest philosophy in agent/anthropic_adapter._get_anthropic_max_output.
|
||||
# Legacy allowlist of Anthropic models that still accept an explicit "disable
|
||||
# thinking" request. Claude 4.6+ and newer named models mandate reasoning and
|
||||
# 400 on any disable form, so *unknown* Anthropic models default to "cannot
|
||||
# disable" (mirrors agent/anthropic_adapter._get_anthropic_max_output).
|
||||
_ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS = (
|
||||
"claude-3", # 3, 3.5, 3.7
|
||||
"claude-opus-4-0", "claude-opus-4.0", "claude-opus-4-1", "claude-opus-4.1",
|
||||
@@ -32,48 +28,44 @@ _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS = (
|
||||
|
||||
|
||||
def _anthropic_reasoning_is_mandatory(model: str | None) -> bool:
|
||||
"""Return True for Anthropic models that reject any disable-thinking form.
|
||||
|
||||
Claude 4.6+ (adaptive thinking) and newer named models have no "off"
|
||||
switch — sending ``reasoning: {enabled: false}`` makes OpenRouter emit
|
||||
``thinking: {type: "disabled"}``, which these models 400 on. Unknown /
|
||||
new Anthropic model names default to mandatory so the next un-numbered
|
||||
release doesn't reintroduce the 400.
|
||||
"""
|
||||
"""True for Anthropic models that reject any disable-thinking form (unknown -> True)."""
|
||||
m = (model or "").lower()
|
||||
if not m.startswith(("anthropic/", "claude")) and "claude" not in m:
|
||||
return False
|
||||
return not any(sub in m for sub in _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS)
|
||||
|
||||
|
||||
def _sticky_key(session_id: str | None) -> str | None:
|
||||
"""Declared routing scope, then ambient conversation, then explicit session_id.
|
||||
|
||||
Aux call sites (compression, titles, vision, MoA…) pass no ``session_id``,
|
||||
so the ambient lineage ROOT keeps them pinned to their conversation.
|
||||
"""
|
||||
return _cache_scope_from_session_id(
|
||||
get_affinity_scope() or get_conversation_context() or session_id
|
||||
)
|
||||
|
||||
|
||||
class OpenRouterProfile(ProviderProfile):
|
||||
"""OpenRouter aggregator — provider preferences, reasoning config passthrough."""
|
||||
|
||||
@staticmethod
|
||||
def _clamp_reasoning_to_catalog(cfg: dict[str, Any], model: str | None) -> dict[str, Any]:
|
||||
"""Clamp ``cfg["effort"]`` to the model's catalog-advertised levels.
|
||||
"""Clamp ``cfg["effort"]`` to the nearest LOWER catalog-advertised level.
|
||||
|
||||
OpenRouter's /v1/models entries publish ``reasoning.supported_efforts``
|
||||
per model (ported from PrimeIntellect-ai/prime-agent#1258). Sending an
|
||||
unsupported effort (e.g. ``ultra`` to a route that stops at ``high``)
|
||||
yields provider 4xx errors; clamp to the nearest LOWER supported level
|
||||
instead. No-op when the catalog is unreachable, the model is unlisted,
|
||||
or no supported_efforts list is published (None = all levels accepted).
|
||||
No-op when the catalog is unreachable, the model is unlisted, or no
|
||||
supported_efforts list is published (None = all levels accepted).
|
||||
"""
|
||||
effort = cfg.get("effort")
|
||||
if not effort or cfg.get("enabled") is False:
|
||||
return cfg
|
||||
try:
|
||||
from hermes_cli.models import (
|
||||
clamp_reasoning_effort_to_supported,
|
||||
openrouter_model_reasoning_capabilities,
|
||||
)
|
||||
from hermes_cli.models import clamp_reasoning_effort_to_supported, openrouter_model_reasoning_capabilities
|
||||
|
||||
caps = openrouter_model_reasoning_capabilities(model)
|
||||
if not caps or not caps.get("supports_reasoning"):
|
||||
return cfg
|
||||
clamped = clamp_reasoning_effort_to_supported(
|
||||
effort, caps.get("supported_efforts")
|
||||
)
|
||||
clamped = clamp_reasoning_effort_to_supported(effort, caps.get("supported_efforts"))
|
||||
except Exception:
|
||||
return cfg
|
||||
if clamped and clamped != effort:
|
||||
@@ -82,83 +74,46 @@ class OpenRouterProfile(ProviderProfile):
|
||||
"(catalog supported_efforts=%s)",
|
||||
effort, clamped, model, caps.get("supported_efforts"),
|
||||
)
|
||||
cfg = dict(cfg)
|
||||
cfg["effort"] = clamped
|
||||
cfg = {**cfg, "effort": clamped}
|
||||
return cfg
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Fetch from public OpenRouter catalog — no auth required.
|
||||
|
||||
Note: Tool-call capability filtering is applied by hermes_cli/models.py
|
||||
via fetch_openrouter_models() → _openrouter_model_supports_tools(), not
|
||||
here. The picker early-returns via the dedicated openrouter path before
|
||||
reaching this method, so filtering here would be unreachable.
|
||||
"""
|
||||
"""Fetch from the public OpenRouter catalog (no auth). Tool-call filtering
|
||||
happens in hermes_cli/models.py, which the picker reaches first."""
|
||||
global _CACHE # noqa: PLW0603
|
||||
if _CACHE is not None:
|
||||
return _CACHE
|
||||
try:
|
||||
result = super().fetch_models(api_key=None, base_url=base_url, timeout=timeout)
|
||||
if result is not None:
|
||||
_CACHE = result
|
||||
return result
|
||||
except Exception as exc:
|
||||
logger.debug("fetch_models(openrouter): %s", exc)
|
||||
return None
|
||||
if result is not None:
|
||||
_CACHE = result
|
||||
return result
|
||||
|
||||
def build_extra_body(
|
||||
self, *, session_id: str | None = None, **context: Any
|
||||
) -> dict[str, Any]:
|
||||
def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]:
|
||||
body: dict[str, Any] = {}
|
||||
# Top-level session_id → OpenRouter's sticky routing key. Per their
|
||||
# prompt-caching docs it is used directly as the routing key instead of
|
||||
# hashing the opening messages, and it activates stickiness on the
|
||||
# first successful request rather than only after a cache hit.
|
||||
#
|
||||
# Resolve it from the declared routing scope first (set only by a host
|
||||
# that names its own conversation, #96811), then the ambient conversation
|
||||
# contextvar, with the explicit argument as fallback. The gap this closes is the auxiliary call sites
|
||||
# — compression, title generation, vision, web_extract, session_search,
|
||||
# MoA slots — which funnel through ``agent.auxiliary_client``. That
|
||||
# module has no session handle and passes no ``session_id``, so those
|
||||
# calls sent NO sticky key at all and each routed independently of the
|
||||
# conversation it belonged to (#70820).
|
||||
#
|
||||
# Mirrors the Nous Portal profile, which resolves the same way
|
||||
# (f2f4df064d). The ambient value is the session-lineage ROOT, so it
|
||||
# also stays stable for installs that opt out of the default
|
||||
# ``compression.in_place: true`` and across delegate-subagent trees.
|
||||
sticky_key = _cache_scope_from_session_id(
|
||||
get_affinity_scope() or get_conversation_context() or session_id
|
||||
)
|
||||
# Top-level session_id is OpenRouter's sticky routing key (used directly,
|
||||
# not hashed from the opening messages; active from the first request).
|
||||
sticky_key = _sticky_key(session_id)
|
||||
if sticky_key:
|
||||
body["session_id"] = sticky_key
|
||||
prefs = context.get("provider_preferences")
|
||||
if prefs:
|
||||
body["provider"] = prefs
|
||||
|
||||
# Pareto Code router — model-gated. The plugins block is only
|
||||
# meaningful for openrouter/pareto-code; sending it on any other
|
||||
# model has no documented effect and would be confusing in logs.
|
||||
# See: https://openrouter.ai/docs/guides/routing/routers/pareto-router
|
||||
model = (context.get("model") or "")
|
||||
if model == "openrouter/pareto-code":
|
||||
score = context.get("openrouter_min_coding_score")
|
||||
if score is not None and score != "":
|
||||
try:
|
||||
score_f = float(score)
|
||||
except (TypeError, ValueError):
|
||||
score_f = None
|
||||
if score_f is not None and 0.0 <= score_f <= 1.0:
|
||||
body["plugins"] = [
|
||||
{"id": "pareto-router", "min_coding_score": score_f}
|
||||
]
|
||||
# Pareto Code router plugin is only meaningful for openrouter/pareto-code.
|
||||
score = context.get("openrouter_min_coding_score")
|
||||
if (context.get("model") or "") == "openrouter/pareto-code" and score is not None and score != "":
|
||||
try:
|
||||
score_f = float(score)
|
||||
except (TypeError, ValueError):
|
||||
score_f = None
|
||||
if score_f is not None and 0.0 <= score_f <= 1.0:
|
||||
body["plugins"] = [{"id": "pareto-router", "min_coding_score": score_f}]
|
||||
return body
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
@@ -170,69 +125,29 @@ class OpenRouterProfile(ProviderProfile):
|
||||
session_id: str | None = None,
|
||||
**context: Any,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""OpenRouter passes the full reasoning_config dict as extra_body.reasoning.
|
||||
|
||||
For xAI Grok models routed through OpenRouter, attach the
|
||||
``x-grok-conv-id`` header so that xAI's prompt cache stays pinned to
|
||||
the same backend server across turns.
|
||||
"""
|
||||
"""Pass reasoning_config as extra_body.reasoning; pin Grok's cache via x-grok-conv-id."""
|
||||
extra_body: dict[str, Any] = {}
|
||||
top_level: dict[str, Any] = {}
|
||||
extra_headers: dict[str, Any] = {}
|
||||
if supports_reasoning:
|
||||
# Reasoning-mandatory Anthropic models (Claude 4.6+ / fable /
|
||||
# future named models) use *adaptive* thinking: the model decides
|
||||
# how much to think, and OpenRouter ignores ``reasoning.effort`` for
|
||||
# them entirely. Sending any ``reasoning`` field is therefore both
|
||||
# pointless and actively harmful:
|
||||
# - ``{enabled: false}`` → OpenRouter emits Anthropic's manual
|
||||
# ``thinking: {type: "disabled"}``, which these models 400 on.
|
||||
# - any enabled form, on a tool-continuation turn whose prior
|
||||
# assistant tool_call carries no thinking block (chat_completions
|
||||
# never replays signed thinking blocks), ALSO makes OpenRouter
|
||||
# emit ``thinking: {type: "disabled"}`` → the same 400 on every
|
||||
# turn after the first tool call.
|
||||
# The only reliable behavior is to omit ``reasoning`` and let the
|
||||
# model default to adaptive. See hermes-agent#42991 (disable case)
|
||||
# and the tool-replay follow-up.
|
||||
#
|
||||
# ``reasoning.effort`` being ignored does NOT mean these models have
|
||||
# no effort lever — OpenRouter honors the requested effort on the
|
||||
# top-level ``verbosity`` field instead (it maps to Anthropic's
|
||||
# ``output_config.effort``; ``reasoning.effort`` is accepted but
|
||||
# ignored — confirmed by OpenRouter's Claude migration docs and a
|
||||
# live token-spend probe in hermes-agent#43432). Route the existing
|
||||
# ``reasoning_config["effort"]`` (sourced from
|
||||
# ``agent.reasoning_effort``) onto ``verbosity`` so the knob the user
|
||||
# already sets keeps working for these models. We still send NO
|
||||
# ``reasoning`` field, preserving the #42991 400 fix.
|
||||
# Reasoning-mandatory Anthropic models use adaptive thinking: any
|
||||
# ``reasoning`` field (disable, or an enabled form on a tool-continuation
|
||||
# turn without a replayed thinking block) makes OpenRouter emit
|
||||
# ``thinking: {type: "disabled"}`` -> 400. Omit it; the user's effort
|
||||
# still reaches Anthropic's output_config.effort via top-level ``verbosity``.
|
||||
if _anthropic_reasoning_is_mandatory(model):
|
||||
cfg = reasoning_config or {}
|
||||
effort = cfg.get("effort")
|
||||
# Only emit when effort is actually requested and reasoning
|
||||
# isn't explicitly disabled. Otherwise omit ``verbosity`` so the
|
||||
# model keeps its own adaptive default (``high``).
|
||||
if cfg.get("enabled", True) is not False and effort and effort != "none":
|
||||
top_level["verbosity"] = effort
|
||||
elif reasoning_config is not None:
|
||||
extra_body["reasoning"] = self._clamp_reasoning_to_catalog(
|
||||
dict(reasoning_config), model
|
||||
)
|
||||
extra_body["reasoning"] = self._clamp_reasoning_to_catalog(dict(reasoning_config), model)
|
||||
else:
|
||||
extra_body["reasoning"] = {"enabled": True, "effort": "medium"}
|
||||
|
||||
# Same resolution as build_extra_body: xAI's prompt cache is pinned per
|
||||
# backend server via this header, and aux calls pass no session_id, so
|
||||
# reading the ambient conversation keeps compression/vision/MoA traffic
|
||||
# on the same Grok backend as the conversation it belongs to.
|
||||
grok_conv_id = _cache_scope_from_session_id(
|
||||
get_affinity_scope() or get_conversation_context() or session_id
|
||||
)
|
||||
# xAI's prompt cache is pinned per backend server via this header.
|
||||
grok_conv_id = _sticky_key(session_id)
|
||||
if grok_conv_id and model and model.startswith(("x-ai/grok-", "xai/grok-")):
|
||||
extra_headers["x-grok-conv-id"] = grok_conv_id
|
||||
if extra_headers:
|
||||
top_level["extra_headers"] = extra_headers
|
||||
|
||||
top_level["extra_headers"] = {"x-grok-conv-id": grok_conv_id}
|
||||
return extra_body, top_level
|
||||
|
||||
|
||||
|
||||
@@ -5,31 +5,34 @@ from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
def _normalize_parts(content: list) -> list | None:
|
||||
"""List content -> list-of-dict parts (str -> text part, image_url dicts copied,
|
||||
other junk dropped). None when nothing changed (copy-on-write)."""
|
||||
parts, changed = [], False
|
||||
for part in content:
|
||||
if isinstance(part, str):
|
||||
parts.append({"type": "text", "text": part})
|
||||
changed = True
|
||||
elif isinstance(part, dict):
|
||||
if isinstance(part.get("image_url"), dict):
|
||||
part = {**part, "image_url": dict(part["image_url"])}
|
||||
changed = True
|
||||
parts.append(part)
|
||||
else:
|
||||
changed = True
|
||||
return parts if parts and changed else None
|
||||
|
||||
|
||||
class QwenProfile(ProviderProfile):
|
||||
"""Qwen Portal — message normalization, vl_high_resolution, metadata top-level."""
|
||||
|
||||
@staticmethod
|
||||
def _copy_part_if_request_mutable(part: dict[str, Any]) -> tuple[dict[str, Any], bool]:
|
||||
image_url = part.get("image_url")
|
||||
if isinstance(image_url, dict):
|
||||
copied = dict(part)
|
||||
copied["image_url"] = dict(image_url)
|
||||
return copied, True
|
||||
return part, False
|
||||
|
||||
def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""Normalize content to list-of-dicts format.
|
||||
|
||||
Inject cache_control on system message.
|
||||
|
||||
Matches the behavior of run_agent.py:_qwen_prepare_chat_messages().
|
||||
"""
|
||||
"""Normalize content to list-of-dicts and inject cache_control on the system
|
||||
message. Copy-on-write: only touched messages/parts are copied."""
|
||||
if not messages:
|
||||
return []
|
||||
|
||||
prepared = list(messages)
|
||||
system_idx: int | None = None
|
||||
|
||||
for idx, msg in enumerate(messages):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
@@ -37,49 +40,22 @@ class QwenProfile(ProviderProfile):
|
||||
system_idx = idx
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str):
|
||||
msg_copy = dict(msg)
|
||||
msg_copy["content"] = [{"type": "text", "text": content}]
|
||||
prepared[idx] = msg_copy
|
||||
prepared[idx] = {**msg, "content": [{"type": "text", "text": content}]}
|
||||
elif isinstance(content, list):
|
||||
normalized_parts = []
|
||||
changed = False
|
||||
for part in content:
|
||||
if isinstance(part, str):
|
||||
normalized_parts.append({"type": "text", "text": part})
|
||||
changed = True
|
||||
elif isinstance(part, dict):
|
||||
normalized_part, copied = self._copy_part_if_request_mutable(part)
|
||||
normalized_parts.append(normalized_part)
|
||||
changed = changed or copied
|
||||
else:
|
||||
changed = True
|
||||
if normalized_parts and changed:
|
||||
msg_copy = dict(msg)
|
||||
msg_copy["content"] = normalized_parts
|
||||
prepared[idx] = msg_copy
|
||||
parts = _normalize_parts(content)
|
||||
if parts is not None:
|
||||
prepared[idx] = {**msg, "content": parts}
|
||||
|
||||
# Inject cache_control on the last part of the system message.
|
||||
if system_idx is not None:
|
||||
msg = prepared[system_idx]
|
||||
if isinstance(msg, dict):
|
||||
content = msg.get("content")
|
||||
if (
|
||||
isinstance(content, list)
|
||||
and content
|
||||
and isinstance(content[-1], dict)
|
||||
):
|
||||
msg_copy = dict(msg)
|
||||
content_copy = list(content)
|
||||
content_copy[-1] = dict(content_copy[-1])
|
||||
content_copy[-1]["cache_control"] = {"type": "ephemeral"}
|
||||
msg_copy["content"] = content_copy
|
||||
prepared[system_idx] = msg_copy
|
||||
|
||||
content = msg.get("content")
|
||||
if isinstance(content, list) and content and isinstance(content[-1], dict):
|
||||
content_copy = list(content)
|
||||
content_copy[-1] = {**content_copy[-1], "cache_control": {"type": "ephemeral"}}
|
||||
prepared[system_idx] = {**msg, "content": content_copy}
|
||||
return prepared
|
||||
|
||||
def build_extra_body(
|
||||
self, *, session_id: str | None = None, **context
|
||||
) -> dict[str, Any]:
|
||||
def build_extra_body(self, *, session_id: str | None = None, **context) -> dict[str, Any]:
|
||||
return {"vl_high_resolution_images": True}
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
@@ -90,10 +66,7 @@ class QwenProfile(ProviderProfile):
|
||||
**context,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Qwen metadata goes to top-level api_kwargs, not extra_body."""
|
||||
top_level = {}
|
||||
if qwen_session_metadata:
|
||||
top_level["metadata"] = qwen_session_metadata
|
||||
return {}, top_level
|
||||
return {}, {"metadata": qwen_session_metadata} if qwen_session_metadata else {}
|
||||
|
||||
|
||||
qwen = QwenProfile(
|
||||
|
||||
@@ -1,45 +1,18 @@
|
||||
"""Ramp Router (router.com) provider plugin for Hermes Agent.
|
||||
"""Ramp Router (router.com) provider profile: Responses-only LLM gateway.
|
||||
|
||||
Provider profile for `Ramp Router <https://docs.router.com>`_, Ramp's LLM
|
||||
gateway: one OpenAI Responses-compatible endpoint at
|
||||
``https://api.router.com/v1`` that routes each request across upstream
|
||||
providers (OpenAI, Anthropic, xAI, Fireworks, ...) and handles fallbacks and
|
||||
spend controls server-side.
|
||||
|
||||
Wire notes (verified live against api.router.com, Aug 2026):
|
||||
|
||||
* **Responses API is the native wire.** Router serves ``GET /v1/models``
|
||||
and ``POST /v1/responses``; ``POST /v1/chat/completions`` is only a
|
||||
minimal compatibility shim (added Aug 2026) that translates onto
|
||||
Responses. Per-model reasoning-effort validation, reasoning summaries,
|
||||
and prompt caching are Responses-surface features, so
|
||||
``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate
|
||||
in ``hermes_cli/providers.py`` keep every path on the native wire —
|
||||
the same shape as the ``api.openai.com`` mandate.
|
||||
* **Account-scoped catalog.** Valid model IDs are whatever the key's
|
||||
``GET /v1/models`` returns (BYOK accounts see extra entries), so this
|
||||
profile ships **no** ``fallback_models`` — the picker relies on the live
|
||||
fetch, per Router's own guidance to never hardcode model names.
|
||||
* **Strict reasoning-effort validation.** Router validates
|
||||
``reasoning.effort`` against each model's catalog-declared vocabulary and
|
||||
returns HTTP 400 ``invalid-argument`` on a level the model does not accept
|
||||
(e.g. ``max`` on grok-4.6), and 400 ``unsupported_parameter`` when a
|
||||
non-reasoning model (gpt-4.1 family, gpt-4o, ...) receives any reasoning
|
||||
field. The catalog publishes the vocabulary per model
|
||||
(``router.capabilities.reasoning``), so ``supported_reasoning_efforts``
|
||||
below feeds the codex transport's clamp from a cached copy of it.
|
||||
* **Everything else passes through.** ``store: false``, ``prompt_cache_key``,
|
||||
``include: ["reasoning.encrypted_content"]``, and ``reasoning.summary`` are
|
||||
accepted on all models (ignored where a backend cannot honor them), tools /
|
||||
``parallel_tool_calls`` / streaming SSE work across backends, and encrypted
|
||||
reasoning replay round-trips on OpenAI-served models — so the generic
|
||||
Responses transport path needs no Router-specific request surgery.
|
||||
|
||||
The capability cache mirrors the OpenRouter reasoning-caps design in
|
||||
``hermes_cli/models.py``: cache-only lookups on the per-request hot path
|
||||
(never HTTP), seeded for free whenever ``fetch_models()`` runs (picker,
|
||||
setup, doctor), hydrated from a disk mirror across processes, and refreshed
|
||||
by a background warmer when cold or stale.
|
||||
Wire notes (verified live against api.router.com):
|
||||
* Responses API is the native wire; ``/chat/completions`` is only a thin shim.
|
||||
``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate in
|
||||
``hermes_cli/providers.py`` keep every path on it.
|
||||
* The catalog is account-scoped (BYOK accounts see extra IDs), so this profile
|
||||
ships no ``fallback_models`` — the picker relies on ``fetch_models()``.
|
||||
* Router 400s on ``reasoning.effort`` levels outside a model's published
|
||||
vocabulary and on any reasoning field for non-reasoning models. The efforts
|
||||
map from ``GET /v1/models`` is cached (memory + disk mirror, background
|
||||
warmer; never HTTP on the request hot path) and fed to the codex transport's
|
||||
clamp via ``supported_reasoning_efforts``.
|
||||
* ``store: false``, ``prompt_cache_key``, encrypted reasoning replay, tools and
|
||||
streaming pass through unchanged — no Router-specific request surgery.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -52,6 +25,7 @@ import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
from agent.reasoning_effort import EFFORT_LADDER
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile, _profile_user_agent
|
||||
@@ -60,43 +34,32 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
ROUTER_DEFAULT_BASE_URL = "https://api.router.com/v1"
|
||||
|
||||
#: Efforts-by-model cache: ``model id -> list of accepted effort levels``.
|
||||
#: ``[]`` means the catalog says the model accepts NO reasoning parameters
|
||||
#: (``reasoning.supported: false``) — the transport must omit reasoning
|
||||
#: entirely. A model absent from the dict is unknown (custom/BYOK route or
|
||||
#: vocabulary not published) and callers fall back to their defaults.
|
||||
#: model id -> accepted effort levels. ``[]`` = model accepts NO reasoning
|
||||
#: fields; absent = unknown (callers keep their defaults).
|
||||
_efforts_cache: Optional[dict[str, list[str]]] = None
|
||||
_efforts_lock = threading.Lock()
|
||||
_warm_started = False
|
||||
_disk_checked = False
|
||||
|
||||
#: Disk-mirror staleness bound. Vocabularies change rarely; a stale verdict
|
||||
#: beats no verdict, so a past-TTL mirror is still served while a background
|
||||
#: refresh runs (same policy as the OpenRouter caps mirror).
|
||||
# A stale verdict beats no verdict: a past-TTL mirror is still served while a
|
||||
# background refresh runs.
|
||||
_DISK_TTL_SECONDS = 24 * 60 * 60
|
||||
|
||||
|
||||
def _base_url() -> str:
|
||||
"""Allow a base-URL override via ``RAMP_ROUTER_BASE_URL``."""
|
||||
return os.getenv("RAMP_ROUTER_BASE_URL", "").strip().rstrip("/") or ROUTER_DEFAULT_BASE_URL
|
||||
|
||||
|
||||
def _resolve_api_key() -> str:
|
||||
"""Resolve the Router key from .env / environment, preferring dotenv.
|
||||
|
||||
``RAMP_ROUTER_API_KEY`` is Router's documented variable;
|
||||
``ROUTER_API_KEY`` is accepted as a convenience alias. Falls back to the
|
||||
raw environment when the hermes_cli helper is unavailable (e.g. stripped
|
||||
test environments).
|
||||
"""
|
||||
resolvers = []
|
||||
"""Resolve the Router key (documented var, then alias), preferring dotenv;
|
||||
plain os.environ is the fallback when the dotenv resolver is unavailable or raises."""
|
||||
resolvers: list = [lambda var: os.environ.get(var, "")]
|
||||
try:
|
||||
from hermes_cli.config import get_env_value_prefer_dotenv
|
||||
|
||||
resolvers.append(get_env_value_prefer_dotenv)
|
||||
resolvers.insert(0, get_env_value_prefer_dotenv)
|
||||
except Exception:
|
||||
pass
|
||||
resolvers.append(lambda var: os.environ.get(var, ""))
|
||||
for resolve in resolvers:
|
||||
for var in ("RAMP_ROUTER_API_KEY", "ROUTER_API_KEY"):
|
||||
try:
|
||||
@@ -108,63 +71,45 @@ def _resolve_api_key() -> str:
|
||||
return ""
|
||||
|
||||
|
||||
def _parse_efforts(items: Any) -> Optional[dict[str, list[str]]]:
|
||||
"""Parse a Router ``/v1/models`` ``data`` array into the efforts map.
|
||||
def _dig(obj: Any, *keys: str) -> Any:
|
||||
"""Nested dict lookup; None as soon as a level is missing or not a dict."""
|
||||
for key in keys:
|
||||
obj = obj.get(key) if isinstance(obj, dict) else None
|
||||
return obj
|
||||
|
||||
Returns None when the array has no usable entries, which callers treat
|
||||
as a failed fetch rather than caching an empty verdict.
|
||||
|
||||
def _parse_efforts(items: Any) -> Optional[dict[str, list[str]]]:
|
||||
"""Parse a ``/v1/models`` ``data`` array into the efforts map (None if unusable).
|
||||
|
||||
Ladder-unknown levels are dropped: clamp_effort ignores them, so an
|
||||
all-unknown vocabulary would pass the effort through unclamped to a Router
|
||||
400. ``supported=True`` with no recognized level leaves the model out
|
||||
(unknown) so the transport keeps its default clamp behavior.
|
||||
"""
|
||||
if not isinstance(items, list):
|
||||
return None
|
||||
try:
|
||||
from agent.reasoning_effort import EFFORT_LADDER
|
||||
|
||||
known_levels = set(EFFORT_LADDER)
|
||||
except Exception:
|
||||
known_levels = None
|
||||
efforts_by_id: dict[str, list[str]] = {}
|
||||
for item in items:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
mid = str(item.get("id") or "").strip()
|
||||
if not mid:
|
||||
continue
|
||||
router_meta = item.get("router")
|
||||
reasoning = None
|
||||
if isinstance(router_meta, dict):
|
||||
capabilities = router_meta.get("capabilities")
|
||||
if isinstance(capabilities, dict):
|
||||
reasoning = capabilities.get("reasoning")
|
||||
if not isinstance(reasoning, dict):
|
||||
mid = str(item.get("id") or "").strip() if isinstance(item, dict) else ""
|
||||
reasoning = _dig(item, "router", "capabilities", "reasoning")
|
||||
if not mid or not isinstance(reasoning, dict):
|
||||
continue
|
||||
if reasoning.get("supported") is False:
|
||||
# Definitive negative: any reasoning field 400s on this model.
|
||||
efforts_by_id[mid] = []
|
||||
continue
|
||||
levels = [
|
||||
str(entry.get("value") or "").strip()
|
||||
for entry in reasoning.get("efforts") or []
|
||||
if isinstance(entry, dict) and str(entry.get("value") or "").strip()
|
||||
]
|
||||
if known_levels is not None:
|
||||
# clamp_effort silently ignores ladder-unknown levels, and an
|
||||
# all-unknown vocabulary would pass the requested effort through
|
||||
# unclamped straight to a Router 400 — so a new vendor tier is
|
||||
# dropped at ingest and fails loudly here instead.
|
||||
unknown = [level for level in levels if level not in known_levels]
|
||||
if unknown:
|
||||
logger.info(
|
||||
"router: model %s publishes unrecognized reasoning effort "
|
||||
"level(s) %s; ignoring them (update agent/reasoning_effort "
|
||||
"EFFORT_LADDER to adopt new vendor tiers)",
|
||||
mid,
|
||||
unknown,
|
||||
)
|
||||
levels = [level for level in levels if level in known_levels]
|
||||
values = [str(e.get("value") or "").strip() for e in reasoning.get("efforts") or [] if isinstance(e, dict)]
|
||||
levels = [v for v in values if v]
|
||||
unknown = [level for level in levels if level not in EFFORT_LADDER]
|
||||
if unknown:
|
||||
logger.info(
|
||||
"router: model %s publishes unrecognized reasoning effort "
|
||||
"level(s) %s; ignoring them (update agent/reasoning_effort "
|
||||
"EFFORT_LADDER to adopt new vendor tiers)",
|
||||
mid, unknown,
|
||||
)
|
||||
levels = [level for level in levels if level in EFFORT_LADDER]
|
||||
if levels:
|
||||
efforts_by_id[mid] = levels
|
||||
# supported=True with no (recognized) vocabulary -> leave the model
|
||||
# out (unknown), so the transport keeps its default clamp behavior.
|
||||
return efforts_by_id or None
|
||||
|
||||
|
||||
@@ -184,16 +129,14 @@ def _save_disk(efforts_by_id: dict[str, list[str]]) -> None:
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".tmp")
|
||||
tmp.write_text(
|
||||
json.dumps({"ts": time.time(), "efforts": efforts_by_id}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
tmp.write_text(json.dumps({"ts": time.time(), "efforts": efforts_by_id}), encoding="utf-8")
|
||||
tmp.replace(path)
|
||||
except Exception as exc:
|
||||
logger.debug("router: caps disk mirror write failed: %s", exc)
|
||||
|
||||
|
||||
def _load_disk() -> tuple[Optional[dict[str, list[str]]], float]:
|
||||
"""Disk mirror -> (efforts map or None, age in seconds; TTL when ``ts`` is unparseable)."""
|
||||
path = _disk_path()
|
||||
if path is None:
|
||||
return None, 0.0
|
||||
@@ -202,11 +145,7 @@ def _load_disk() -> tuple[Optional[dict[str, list[str]]], float]:
|
||||
efforts = data.get("efforts")
|
||||
if not isinstance(efforts, dict) or not efforts:
|
||||
return None, 0.0
|
||||
parsed = {
|
||||
str(mid): [str(level) for level in levels]
|
||||
for mid, levels in efforts.items()
|
||||
if isinstance(levels, list)
|
||||
}
|
||||
parsed = {str(mid): [str(lv) for lv in levels] for mid, levels in efforts.items() if isinstance(levels, list)}
|
||||
try:
|
||||
age = max(0.0, time.time() - float(data.get("ts") or 0))
|
||||
except (TypeError, ValueError):
|
||||
@@ -220,11 +159,10 @@ def _seed_efforts(items: Any) -> Optional[dict[str, list[str]]]:
|
||||
"""Seed memory + disk caches from a ``/v1/models`` payload."""
|
||||
global _efforts_cache
|
||||
parsed = _parse_efforts(items)
|
||||
if parsed is None:
|
||||
return None
|
||||
with _efforts_lock:
|
||||
_efforts_cache = parsed
|
||||
_save_disk(parsed)
|
||||
if parsed is not None:
|
||||
with _efforts_lock:
|
||||
_efforts_cache = parsed
|
||||
_save_disk(parsed)
|
||||
return parsed
|
||||
|
||||
|
||||
@@ -232,17 +170,16 @@ def _fetch_catalog_items(
|
||||
*, api_key: str = "", base_url: str = "", timeout: float = 8.0
|
||||
) -> Optional[list]:
|
||||
"""Fetch the raw ``/v1/models`` ``data`` array. None on any failure."""
|
||||
url = (base_url or _base_url()).rstrip("/") + "/models"
|
||||
import urllib.request
|
||||
|
||||
from hermes_cli.urllib_security import open_credentialed_url
|
||||
|
||||
req = urllib.request.Request(url)
|
||||
req = urllib.request.Request((base_url or _base_url()).rstrip("/") + "/models")
|
||||
key = api_key or _resolve_api_key()
|
||||
if key:
|
||||
req.add_header("Authorization", f"Bearer {key}")
|
||||
req.add_header("Accept", "application/json")
|
||||
# Router sits behind a WAF that rejects the default Python-urllib UA.
|
||||
# Router's WAF rejects the default Python-urllib UA.
|
||||
req.add_header("User-Agent", _profile_user_agent())
|
||||
try:
|
||||
with open_credentialed_url(req, timeout=timeout) as resp:
|
||||
@@ -255,14 +192,12 @@ def _fetch_catalog_items(
|
||||
|
||||
|
||||
def _efforts_cache_only() -> Optional[dict[str, list[str]]]:
|
||||
"""Memory, else the disk mirror. Never HTTP (hot-path safe)."""
|
||||
"""Memory, else the disk mirror (checked once per process). Never HTTP (hot-path safe)."""
|
||||
global _efforts_cache, _disk_checked
|
||||
with _efforts_lock:
|
||||
cached = _efforts_cache
|
||||
if cached is not None:
|
||||
if cached is not None or _disk_checked:
|
||||
return cached
|
||||
if _disk_checked:
|
||||
return None
|
||||
_disk_checked = True
|
||||
parsed, age = _load_disk()
|
||||
if parsed is None:
|
||||
@@ -277,19 +212,19 @@ def _efforts_cache_only() -> Optional[dict[str, list[str]]]:
|
||||
|
||||
|
||||
def _warm_efforts_async() -> None:
|
||||
"""Refresh the efforts cache in the background, at most once per process."""
|
||||
"""Refresh the efforts cache in the background, at most once per process.
|
||||
|
||||
Skipped under pytest (a mid-suite fetch makes cache state timing-dependent)
|
||||
and without a key (it would 401; the first authenticated fetch_models() seeds).
|
||||
"""
|
||||
global _warm_started
|
||||
if os.environ.get("PYTEST_CURRENT_TEST"):
|
||||
# Match the canonical caps warmer (hermes_cli/models.py): a mid-suite
|
||||
# background fetch would make cache state timing-dependent in tests.
|
||||
return
|
||||
with _efforts_lock:
|
||||
if _warm_started:
|
||||
return
|
||||
_warm_started = True
|
||||
if not _resolve_api_key():
|
||||
# Without a key the fetch would 401; the first authenticated
|
||||
# fetch_models() (picker/setup/doctor) seeds the cache instead.
|
||||
return
|
||||
|
||||
def _refresh() -> None:
|
||||
@@ -298,9 +233,7 @@ def _warm_efforts_async() -> None:
|
||||
_seed_efforts(items)
|
||||
|
||||
try:
|
||||
threading.Thread(
|
||||
target=_refresh, name="router-caps-warm", daemon=True
|
||||
).start()
|
||||
threading.Thread(target=_refresh, name="router-caps-warm", daemon=True).start()
|
||||
except Exception as exc:
|
||||
logger.debug("router: caps warmer failed to start: %s", exc)
|
||||
|
||||
@@ -309,46 +242,19 @@ class RouterProfile(ProviderProfile):
|
||||
"""Ramp Router — Responses-only gateway with catalog-declared efforts."""
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: Optional[str] = None, base_url: Optional[str] = None, timeout: float = 8.0
|
||||
) -> Optional[list[str]]:
|
||||
"""Fetch the live, key-scoped catalog and seed the caps cache.
|
||||
|
||||
One request serves both consumers: the picker gets the model IDs and
|
||||
the reasoning-vocabulary mirror is left warm at no extra network
|
||||
cost (the same document carries both).
|
||||
"""
|
||||
items = _fetch_catalog_items(
|
||||
api_key=api_key or "", base_url=base_url or "", timeout=timeout
|
||||
)
|
||||
"""Fetch the live, key-scoped catalog; the same payload seeds the caps cache.
|
||||
Deduped but not sorted: Router's listing order is deliberate presentation."""
|
||||
items = _fetch_catalog_items(api_key=api_key or "", base_url=base_url or "", timeout=timeout)
|
||||
if items is None:
|
||||
return None
|
||||
_seed_efforts(items)
|
||||
# Deduped but not sorted: Router's listing order is deliberate
|
||||
# presentation (featured/current models first), so the picker keeps it.
|
||||
ids = list(
|
||||
dict.fromkeys(
|
||||
str(item["id"])
|
||||
for item in items
|
||||
if isinstance(item, dict) and item.get("id")
|
||||
)
|
||||
)
|
||||
ids = list(dict.fromkeys(str(i["id"]) for i in items if isinstance(i, dict) and i.get("id")))
|
||||
return ids or None
|
||||
|
||||
def supported_reasoning_efforts(
|
||||
self, model: Optional[str]
|
||||
) -> Optional[tuple[str, ...]]:
|
||||
"""Catalog-declared effort vocabulary for *model* (cache-only).
|
||||
|
||||
Router 400s on efforts outside a model's published set and on any
|
||||
reasoning field for non-reasoning models, so the codex transport
|
||||
clamps (or suppresses) from this verdict. Cold cache returns None —
|
||||
the transport keeps its defaults — and kicks a background warmer so
|
||||
the next turn is covered.
|
||||
"""
|
||||
def supported_reasoning_efforts(self, model: Optional[str]) -> Optional[tuple[str, ...]]:
|
||||
"""Catalog-declared effort vocabulary (cache-only; cold cache -> None + warm)."""
|
||||
mid = str(model or "").strip()
|
||||
if not mid:
|
||||
return None
|
||||
@@ -356,10 +262,7 @@ class RouterProfile(ProviderProfile):
|
||||
if efforts_by_id is None:
|
||||
_warm_efforts_async()
|
||||
return None
|
||||
levels = efforts_by_id.get(mid)
|
||||
if levels is None:
|
||||
return None
|
||||
return tuple(levels)
|
||||
return None if mid not in efforts_by_id else tuple(efforts_by_id[mid])
|
||||
|
||||
|
||||
router = RouterProfile(
|
||||
@@ -369,26 +272,14 @@ router = RouterProfile(
|
||||
display_name="Ramp Router",
|
||||
description="Ramp Router (router.com) — routes each request to the cheapest model that clears your quality bar",
|
||||
signup_url="https://app.router.com/keys",
|
||||
# RAMP_ROUTER_API_KEY is Router's documented variable; ROUTER_API_KEY is
|
||||
# a convenience alias. RAMP_ROUTER_BASE_URL overrides the endpoint
|
||||
# (auth.py picks it up as the registry's base_url_env_var).
|
||||
env_vars=("RAMP_ROUTER_API_KEY", "ROUTER_API_KEY", "RAMP_ROUTER_BASE_URL"),
|
||||
base_url=_base_url(),
|
||||
auth_type="api_key",
|
||||
# Identify Hermes traffic to the gateway (Router attributes coding-agent
|
||||
# clients by User-Agent prefix, the way it already recognizes OpenCode's
|
||||
# versioned UA) — and Router's WAF rejects blank/default client UAs.
|
||||
# Router attributes coding-agent clients by UA prefix; its WAF rejects default UAs.
|
||||
default_headers={"User-Agent": f"Hermes-Agent/{_HERMES_VERSION}"},
|
||||
# Most of the catalog's frontier routes accept image input; capability is
|
||||
# still model-dependent and governed by the live catalog.
|
||||
supports_vision=True,
|
||||
# Cheap, reasoning-capable, and vision-capable — safe for auxiliary tasks
|
||||
# (compaction, titles, vision) when Router is the main provider. Also the
|
||||
# model Router's own docs use as their example.
|
||||
default_aux_model="gpt-5.4-mini",
|
||||
# Deliberately empty: model IDs are account-scoped (BYOK accounts see
|
||||
# extra entries) and Router's docs say to read the catalog at runtime
|
||||
# rather than hardcode names. The picker uses fetch_models() above.
|
||||
# Empty on purpose: model IDs are account-scoped; the picker uses fetch_models().
|
||||
fallback_models=(),
|
||||
)
|
||||
|
||||
|
||||
@@ -1,102 +1,49 @@
|
||||
"""Upstage Solar provider profile."""
|
||||
"""Upstage Solar provider profile: top-level ``reasoning_effort`` (low|medium|high).
|
||||
|
||||
Solar's server default is ``minimal`` (reasoning off) — wrong for agentic work —
|
||||
so an unset reasoning_config defaults reasoning ON at ``medium``, matching the
|
||||
"medium (default)" the /reasoning panel shows. Explicit settings always win.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import EFFORT_LADDER, SOLAR_EFFORTS, clamp_effort
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
|
||||
# Model-name markers for Solar families that do NOT accept ``reasoning_effort``.
|
||||
# Deny-list on purpose: newly released Solar models are assumed
|
||||
# reasoning-capable by default, so only the known non-reasoning families are
|
||||
# listed here. Substring match (not startswith) so dated variants like
|
||||
# ``solar-mini-250127`` are covered too.
|
||||
# Deny-list on purpose: new Solar models are assumed reasoning-capable; only
|
||||
# these known non-reasoning families ignore reasoning_effort. Substring match
|
||||
# so dated variants (``solar-mini-250127``) are covered.
|
||||
_NON_REASONING_MODEL_MARKERS = ("solar-mini", "syn-pro")
|
||||
|
||||
# When the user hasn't picked a reasoning effort, Hermes passes
|
||||
# reasoning_config=None. Solar's own server default is "minimal" (reasoning
|
||||
# off), which is the wrong default for an agentic workload. We default reasoning
|
||||
# ON at this effort — matching the "medium (default)" that Hermes' /reasoning
|
||||
# panel shows for an unset config, so the displayed default and the real wire
|
||||
# value agree. An explicit saved setting or a `/reasoning <level>` change is
|
||||
# always honored over this default; `/reasoning none` disables it.
|
||||
_DEFAULT_REASONING_EFFORT = "medium"
|
||||
|
||||
|
||||
def _model_supports_reasoning(model: str | None) -> bool:
|
||||
"""Solar reasoning-capable models — True unless the model is deny-listed.
|
||||
|
||||
The Solar Pro family (``solar-pro``, ``solar-pro2``, ``solar-pro3`` and
|
||||
dated variants like ``solar-pro3-250127``) and the Solar Open family
|
||||
(``solar-open*``) accept ``reasoning_effort``; only ``solar-mini`` /
|
||||
``syn-pro`` ignore the parameter, so we deny-list those and treat every
|
||||
other (incl. future) Solar model as reasoning-capable.
|
||||
|
||||
``None``/empty model → True: the provider default (``fallback_models[0]``,
|
||||
``solar-pro3``) is reasoning-capable, so an unset model gets the same
|
||||
default-on behaviour.
|
||||
"""
|
||||
m = (model or "").strip().lower()
|
||||
return not any(marker in m for marker in _NON_REASONING_MODEL_MARKERS)
|
||||
|
||||
|
||||
class UpstageProfile(ProviderProfile):
|
||||
"""Upstage Solar — top-level ``reasoning_effort`` control.
|
||||
|
||||
Solar Pro/Open expose reasoning through a top-level ``reasoning_effort``
|
||||
field (``minimal`` | ``low`` | ``medium`` | ``high``), mirroring OpenAI's
|
||||
shape. Unlike DeepSeek/Kimi it does NOT require echoing ``reasoning_content``
|
||||
back on later turns, so only the request field needs wiring. We emit at most
|
||||
``low`` | ``medium`` | ``high`` — the explicit values both Solar Pro 2 and
|
||||
Pro 3 accept.
|
||||
|
||||
Default-on: Solar's own server default is ``minimal`` (off), but for an
|
||||
agentic workload we default reasoning ON (``_DEFAULT_REASONING_EFFORT``)
|
||||
when the user hasn't picked an effort. The user can still set any level or
|
||||
turn it off with ``/reasoning none``.
|
||||
"""
|
||||
"""Upstage Solar — top-level ``reasoning_effort`` control (no reasoning_content echo needed)."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
top_level: dict[str, Any] = {}
|
||||
|
||||
# solar-mini / syn-pro (the deny-list) ignore reasoning_effort — send
|
||||
# nothing. Everything else, including future Solar models, gets it.
|
||||
if not _model_supports_reasoning(model):
|
||||
return {}, top_level
|
||||
|
||||
# Unset (reasoning_config is None) → default reasoning ON for agents.
|
||||
m = (model or "").strip().lower()
|
||||
if any(marker in m for marker in _NON_REASONING_MODEL_MARKERS):
|
||||
return {}, {}
|
||||
# Unset -> default reasoning ON for agents.
|
||||
if not reasoning_config or not isinstance(reasoning_config, dict):
|
||||
return {}, {"reasoning_effort": _DEFAULT_REASONING_EFFORT}
|
||||
|
||||
# Explicitly disabled (`/reasoning none`) → omit the field so Solar
|
||||
# applies its own default (minimal = off).
|
||||
return {}, {"reasoning_effort": "medium"}
|
||||
# Explicitly disabled -> omit so Solar applies its own default (minimal = off).
|
||||
if reasoning_config.get("enabled") is False:
|
||||
return {}, top_level
|
||||
|
||||
# Map Hermes' effort vocabulary onto Solar's accepted set via the
|
||||
# shared clamp (agent.reasoning_effort). minimal → omit (Solar's
|
||||
# minimal means off); unknown-but-enabled bespoke levels collapse to
|
||||
# high rather than silently downgrading (#62650 precedent).
|
||||
return {}, {}
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if not effort:
|
||||
top_level["reasoning_effort"] = _DEFAULT_REASONING_EFFORT
|
||||
return {}, top_level
|
||||
return {}, {"reasoning_effort": "medium"}
|
||||
if effort == "minimal":
|
||||
return {}, top_level
|
||||
|
||||
from agent.reasoning_effort import EFFORT_LADDER, SOLAR_EFFORTS, clamp_effort
|
||||
|
||||
return {}, {}
|
||||
mapped = clamp_effort(effort, SOLAR_EFFORTS)
|
||||
if mapped not in SOLAR_EFFORTS:
|
||||
# Bespoke level outside the ladder — Solar precedent is to run
|
||||
# at full strength rather than quietly fall to the default.
|
||||
# Bespoke level outside the ladder runs at full strength rather
|
||||
# than quietly falling to the default; ladder levels that still
|
||||
# don't map are omitted.
|
||||
mapped = "high" if effort not in EFFORT_LADDER else None
|
||||
|
||||
if mapped:
|
||||
top_level["reasoning_effort"] = mapped
|
||||
return {}, top_level
|
||||
return {}, {"reasoning_effort": mapped} if mapped else {}
|
||||
|
||||
|
||||
upstage = UpstageProfile(
|
||||
@@ -108,11 +55,8 @@ upstage = UpstageProfile(
|
||||
env_vars=("UPSTAGE_API_KEY", "UPSTAGE_BASE_URL"),
|
||||
base_url="https://api.upstage.ai/v1",
|
||||
auth_type="api_key",
|
||||
# default_aux_model left empty → auxiliary side tasks use the main model.
|
||||
# entry [0] is the setup default — solar-pro3, the current Solar Pro flagship.
|
||||
fallback_models=(
|
||||
"solar-pro3",
|
||||
),
|
||||
# No default_aux_model: auxiliary tasks use the main model. [0] is the setup default.
|
||||
fallback_models=("solar-pro3",),
|
||||
)
|
||||
|
||||
register_provider(upstage)
|
||||
|
||||
@@ -1,20 +1,10 @@
|
||||
"""Google Vertex AI provider profile.
|
||||
"""Google Vertex AI provider profile: Gemini via Google Cloud's OpenAI-compatible
|
||||
endpoint.
|
||||
|
||||
vertex: Gemini models via Google Cloud's OpenAI-compatible endpoint.
|
||||
|
||||
Auth is OAuth2 — short-lived access tokens minted from a service-account JSON
|
||||
or Application Default Credentials (ADC), NOT a static API key. Token
|
||||
resolution and refresh live in ``agent/vertex_adapter.py``; runtime_provider.py
|
||||
calls it to obtain a fresh ``(token, base_url)`` pair, then hands the token to
|
||||
the standard OpenAI client as ``api_key``. Because the wire format is the
|
||||
OpenAI-compatible chat/completions surface, no message translation is needed —
|
||||
the only Gemini-specific concern is the ``thinking_config`` reasoning hook,
|
||||
which is emitted here exactly as the ``gemini`` provider does for its
|
||||
OpenAI-compat subpath (``extra_body.google.thinking_config``).
|
||||
|
||||
``auth_type="vertex"`` marks this as an OAuth-token provider (resolved
|
||||
specially, like bedrock's ``aws_sdk``) so it is never treated as an
|
||||
api_key provider that would mistake a credentials-file path for a key.
|
||||
Auth is OAuth2 (service-account JSON or ADC), not a static key: ``agent/
|
||||
vertex_adapter.py`` mints ``(token, base_url)`` and the token is passed as
|
||||
``api_key``. ``auth_type="vertex"`` keeps it out of the api_key provider path so
|
||||
a credentials-file path is never mistaken for a key.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
@@ -26,39 +16,24 @@ from providers.base import ProviderProfile
|
||||
class VertexProfile(ProviderProfile):
|
||||
"""Vertex AI — reuse Gemini's thinking_config translation for extra_body."""
|
||||
|
||||
def build_extra_body(
|
||||
self, *, session_id: str | None = None, **context: Any
|
||||
) -> dict[str, Any]:
|
||||
"""Emit ``extra_body.google.thinking_config`` for the OpenAI-compat
|
||||
Vertex surface, mirroring the ``gemini`` provider's behavior.
|
||||
"""
|
||||
def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]:
|
||||
"""Emit ``extra_body.google.thinking_config`` like the ``gemini`` provider's
|
||||
OpenAI-compat subpath."""
|
||||
from agent.transports.chat_completions import (
|
||||
_build_gemini_thinking_config,
|
||||
_snake_case_gemini_thinking_config,
|
||||
)
|
||||
|
||||
model = context.get("model") or ""
|
||||
reasoning_config = context.get("reasoning_config")
|
||||
|
||||
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
|
||||
if not raw_thinking_config:
|
||||
return {}
|
||||
|
||||
thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config)
|
||||
raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config"))
|
||||
thinking_config = _snake_case_gemini_thinking_config(raw) if raw else None
|
||||
if not thinking_config:
|
||||
return {}
|
||||
return {"extra_body": {"google": {"thinking_config": thinking_config}}}
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
timeout: float = 8.0,
|
||||
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
|
||||
) -> list[str] | None:
|
||||
"""Vertex's OpenAI-compat endpoint has no ``/models`` listing route;
|
||||
model discovery is not available. The setup wizard ships a curated list.
|
||||
"""
|
||||
"""No ``/models`` route on the OpenAI-compat endpoint; setup ships a curated list."""
|
||||
return None
|
||||
|
||||
|
||||
|
||||
@@ -1,27 +1,7 @@
|
||||
"""ZAI / GLM provider profile.
|
||||
|
||||
Z.AI's GLM-4.5-and-later chat models default to thinking-mode ON when the
|
||||
request omits ``thinking``. Hermes' ``reasoning_config = {"enabled": False}``
|
||||
was previously a silent no-op on this route — the base profile emits nothing,
|
||||
so users who turned thinking off (desktop toggle, ``/reasoning none``,
|
||||
``reasoning_effort: none``/``false`` in config.yaml) kept burning thinking
|
||||
tokens on every turn.
|
||||
|
||||
:meth:`ZaiProfile.build_api_kwargs_extras` translates the Hermes reasoning
|
||||
config into the wire shape Z.AI's OpenAI-compat endpoint expects:
|
||||
|
||||
{"extra_body": {"thinking": {"type": "enabled" | "disabled"}}}
|
||||
|
||||
When no reasoning preference is set (``reasoning_config is None``) the field
|
||||
is omitted so the server default applies, matching prior behavior. GLM
|
||||
models before 4.5 (e.g. ``glm-4-9b``) don't accept ``thinking`` and are left
|
||||
untouched.
|
||||
|
||||
GLM-5.2 additionally exposes a native ``reasoning_effort`` knob with exactly
|
||||
two enabled levels — ``high`` and ``max`` — on the OpenAI-compatible endpoint
|
||||
(per Z.AI / BigModel docs). Hermes' richer effort scale is collapsed onto
|
||||
those two so the user's effort preference actually reaches the model instead
|
||||
of being silently dropped.
|
||||
GLM-4.5+ defaults to thinking ON, so ``reasoning_config`` is translated to
|
||||
``extra_body.thinking``; GLM-5.2/5.3 also take a native ``reasoning_effort``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,93 +9,40 @@ from __future__ import annotations
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from agent import reasoning_effort as re_
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
_GLM_VERSION_RE = re.compile(r"^glm-(\d+)(?:\.(\d+))?")
|
||||
# Alias spellings seen on relays (Fireworks ``glm-5p2``, ``zai-org-glm-5-2``…).
|
||||
_GLM_5_3_TOKENS = ("glm-5.3", "glm-5-3", "glm-5p3")
|
||||
_GLM_5_2_TOKENS = ("glm-5.2", "glm-5-2", "glm-5p2") + _GLM_5_3_TOKENS
|
||||
|
||||
|
||||
def _model_supports_thinking(model: str | None) -> bool:
|
||||
"""GLM thinking-capable model families: glm-4.5 and later (4.5, 4.6, 5…)."""
|
||||
match = _GLM_VERSION_RE.match((model or "").strip().lower())
|
||||
return bool(match) and (int(match.group(1)), int(match.group(2) or 0)) >= (4, 5)
|
||||
|
||||
|
||||
def _has_token(model: str | None, tokens: tuple[str, ...]) -> bool:
|
||||
m = (model or "").strip().lower()
|
||||
match = _GLM_VERSION_RE.match(m)
|
||||
if not match:
|
||||
return False
|
||||
major = int(match.group(1))
|
||||
minor = int(match.group(2) or 0)
|
||||
return (major, minor) >= (4, 5)
|
||||
return bool(m) and any(token in m for token in tokens)
|
||||
|
||||
|
||||
def _is_glm_5_2(model: str | None) -> bool:
|
||||
"""Detect GLM-5.2/5.3 (reasoning_effort-capable) across alias spellings.
|
||||
def _glm_5_2_reasoning_effort(reasoning_config: dict | None, *, model: str | None = None) -> str | None:
|
||||
"""Map Hermes effort onto GLM's vocabulary (5.2: high/max; 5.3: low..max).
|
||||
|
||||
Covers the canonical ``glm-5.2``/``glm-5.3`` plus the ``glm-5-2`` /
|
||||
``glm-5p2`` variants seen on relays (Fireworks ``glm-5p2``, etc.) and any
|
||||
vendor-prefixed form (``z-ai/glm-5.2``, ``zai-org-glm-5-2``). GLM-5.3
|
||||
uses the same base model as 5.2 (post-training gains only) and exposes
|
||||
the same ``reasoning_effort`` knob (verified live 2026-08-14: the
|
||||
coding-plan endpoint accepts ``reasoning_effort: high`` for glm-5.3).
|
||||
Below-floor efforts clamp to the floor; disabled/unset leaves the server default.
|
||||
"""
|
||||
m = (model or "").strip().lower()
|
||||
if not m:
|
||||
return False
|
||||
return any(
|
||||
token in m
|
||||
for token in ("glm-5.2", "glm-5-2", "glm-5p2", "glm-5.3", "glm-5-3", "glm-5p3")
|
||||
)
|
||||
|
||||
|
||||
def _is_glm_5_3(model: str | None) -> bool:
|
||||
"""Detect GLM-5.3 specifically — it has a wider effort vocabulary.
|
||||
|
||||
5.2 accepts only ``high``/``max``; 5.3 accepts a graded
|
||||
``low``/``medium``/``high``/``max`` scale (verified live, issue #91789),
|
||||
so effort mapping must pick the vocabulary per model.
|
||||
"""
|
||||
m = (model or "").strip().lower()
|
||||
if not m:
|
||||
return False
|
||||
return any(token in m for token in ("glm-5.3", "glm-5-3", "glm-5p3"))
|
||||
|
||||
|
||||
def _glm_5_2_reasoning_effort(
|
||||
reasoning_config: dict | None, *, model: str | None = None
|
||||
) -> str | None:
|
||||
"""Map Hermes reasoning effort onto GLM's native vocabulary.
|
||||
|
||||
GLM-5.2 supports two enabled effort levels (``high``/``max``);
|
||||
GLM-5.3 supports the graded ``low``/``medium``/``high``/``max`` scale.
|
||||
``xhigh``/``max``/``ultra`` request the top tier; anything below the
|
||||
model's floor clamps to that floor. When reasoning is explicitly
|
||||
disabled, or no effort preference is supplied, the server default is
|
||||
left untouched.
|
||||
"""
|
||||
if not isinstance(reasoning_config, dict):
|
||||
effort = re_.requested_effort(reasoning_config)
|
||||
if effort is None or effort == "none":
|
||||
return None
|
||||
if reasoning_config.get("enabled") is False:
|
||||
return None
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if not effort or effort == "none":
|
||||
return None
|
||||
|
||||
# Per-model vocabulary declared in agent.reasoning_effort; xhigh rounds
|
||||
# up to max on both. 5.2 cannot think less than high; 5.3 accepts a
|
||||
# graded scale down to low (issue #91789).
|
||||
from agent.reasoning_effort import (
|
||||
GLM52_EFFORTS,
|
||||
GLM52_OVERRIDES,
|
||||
GLM53_EFFORTS,
|
||||
GLM53_OVERRIDES,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
if _is_glm_5_3(model):
|
||||
efforts, overrides, floor = GLM53_EFFORTS, GLM53_OVERRIDES, "low"
|
||||
if _has_token(model, _GLM_5_3_TOKENS):
|
||||
efforts, overrides, floor = re_.GLM53_EFFORTS, re_.GLM53_OVERRIDES, "low"
|
||||
else:
|
||||
efforts, overrides, floor = GLM52_EFFORTS, GLM52_OVERRIDES, "high"
|
||||
|
||||
clamped = clamp_effort(effort, efforts, overrides)
|
||||
efforts, overrides, floor = re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES, "high"
|
||||
clamped = re_.clamp_effort(effort, efforts, overrides)
|
||||
return clamped if clamped in efforts else floor
|
||||
|
||||
|
||||
@@ -127,21 +54,19 @@ class ZaiProfile(ProviderProfile):
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
top_level: dict[str, Any] = {}
|
||||
|
||||
if not _model_supports_thinking(model) and not _is_glm_5_2(model):
|
||||
is_5_2 = _has_token(model, _GLM_5_2_TOKENS)
|
||||
if not _model_supports_thinking(model) and not is_5_2:
|
||||
return extra_body, top_level
|
||||
|
||||
# Only emit when the user expressed a preference; omitting the field
|
||||
# keeps the server default (enabled) exactly as before.
|
||||
# Only emit when the user expressed a preference (server default = enabled).
|
||||
if isinstance(reasoning_config, dict):
|
||||
enabled = reasoning_config.get("enabled") is not False
|
||||
extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"}
|
||||
|
||||
if _is_glm_5_2(model):
|
||||
if is_5_2:
|
||||
effort = _glm_5_2_reasoning_effort(reasoning_config, model=model)
|
||||
if effort is not None:
|
||||
top_level["reasoning_effort"] = effort
|
||||
|
||||
return extra_body, top_level
|
||||
|
||||
|
||||
@@ -152,11 +77,7 @@ zai = ZaiProfile(
|
||||
display_name="Z.AI (GLM)",
|
||||
description="Z.AI / GLM — Zhipu AI models",
|
||||
signup_url="https://z.ai/",
|
||||
fallback_models=(
|
||||
"glm-5.2",
|
||||
"glm-5",
|
||||
"glm-4-9b",
|
||||
),
|
||||
fallback_models=("glm-5.2", "glm-5", "glm-4-9b"),
|
||||
base_url="https://api.z.ai/api/paas/v4",
|
||||
default_aux_model="glm-4.5-flash",
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user