refactor(plugins/model-providers): reuse agent.reasoning_effort clamps, per-model dict tables, compact profiles

This commit is contained in:
Teknium
2026-09-02 11:23:23 -07:00
parent 5d78bd817d
commit 313b244e8f
32 changed files with 595 additions and 1685 deletions
+8 -21
View File
@@ -14,10 +14,11 @@ from providers.base import ProviderProfile, _profile_user_agent
logger = logging.getLogger(__name__)
DEFAULT_ACTUAL_BASE_URL = "https://api.actual.inc/v1"
DEFAULT_ACTUAL_LOCAL_BASE_URL = "http://127.0.0.1:8080/v1"
_LOCAL_HOSTS = {"localhost", "127.0.0.1", "::1", "0.0.0.0"}
def _normalize_actual_base_url(base_url: str) -> str:
"""Append /v1 to a bare hosted or local host; pass anything else through."""
url = str(base_url or "").strip().rstrip("/")
if not url:
return DEFAULT_ACTUAL_BASE_URL
@@ -27,42 +28,28 @@ def _normalize_actual_base_url(base_url: str) -> str:
path = parsed.path.rstrip("/")
except Exception:
return url
if host == "api.actual.inc" and path in {"", "/"}:
return url + "/v1"
if host in {"localhost", "127.0.0.1", "::1", "0.0.0.0"} and path in {"", "/"}:
if (host == "api.actual.inc" or host in _LOCAL_HOSTS) and path in {"", "/"}:
return url + "/v1"
return url
class ActualProfile(ProviderProfile):
"""Actual Computer provider.
Hosted inference defaults to api.actual.inc. Local inference is exposed by
the Actual client only when it runs in offline mode, so users opt into it by
setting ACTUAL_BASE_URL to the local API URL.
"""
"""Actual Computer: hosted at api.actual.inc; local (offline-mode client)
inference opted into via ACTUAL_BASE_URL."""
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
from hermes_cli.urllib_security import open_credentialed_url
base_url = _normalize_actual_base_url(
os.getenv("ACTUAL_BASE_URL", "").strip() or base_url or self.base_url
)
if not base_url:
return None
req = urllib.request.Request(base_url + "/models")
if api_key:
req.add_header("Authorization", f"Bearer {api_key}")
req.add_header("Accept", "application/json")
req.add_header("User-Agent", _profile_user_agent())
from hermes_cli.urllib_security import open_credentialed_url
try:
with open_credentialed_url(req, timeout=timeout) as resp:
data = json.loads(resp.read().decode())
+6 -16
View File
@@ -1,8 +1,4 @@
"""Vercel AI Gateway provider profile.
AI Gateway routes to multiple backends. Hermes sends attribution
headers and full reasoning config passthrough.
"""
"""Vercel AI Gateway provider profile: attribution headers + reasoning passthrough."""
from typing import Any
@@ -14,18 +10,12 @@ class VercelAIGatewayProfile(ProviderProfile):
"""Vercel AI Gateway — attribution headers + reasoning passthrough."""
def build_api_kwargs_extras(
self,
*,
reasoning_config: dict | None = None,
supports_reasoning: bool = True,
**ctx: Any,
self, *, reasoning_config: dict | None = None, supports_reasoning: bool = True, **ctx: Any
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
if supports_reasoning and reasoning_config is not None:
extra_body["reasoning"] = dict(reasoning_config)
elif supports_reasoning:
extra_body["reasoning"] = {"enabled": True, "effort": "medium"}
return extra_body, {}
if not supports_reasoning:
return {}, {}
reasoning = dict(reasoning_config) if reasoning_config is not None else {"enabled": True, "effort": "medium"}
return {"reasoning": reasoning}, {}
vercel = VercelAIGatewayProfile(
@@ -1,18 +1,8 @@
"""Alibaba Cloud Coding Plan provider profiles.
"""Alibaba Cloud Coding Plan provider profiles (intl + CN): a dedicated endpoint
and key tier separate from ``alibaba``. Names match models.dev catalog keys.
Separate from the standard `alibaba` profile because it hits a different
endpoint (coding-intl.dashscope.aliyuncs.com) with a dedicated API key tier.
Region split, mirroring the base DashScope pair (#73265):
- ``alibaba-coding-plan`` → coding-intl.dashscope.aliyuncs.com (international)
- ``alibaba-coding-plan-cn`` → coding.dashscope.aliyuncs.com (mainland China)
Profile names match the models.dev catalog keys exactly so model metadata
lines up and ``model.provider: alibaba-coding-plan-cn`` resolves at runtime.
The CN profile checks its own ``ALIBABA_CODING_PLAN_CN_API_KEY`` first (#101122,
mirroring kimi-coding-cn) and keeps the shared vars as ordered fallbacks so
existing CN users configured with the shared key keep working.
The CN profile checks its own key first and keeps the shared vars as ordered
fallbacks so existing CN users configured with the shared key keep working.
"""
from providers import register_provider
+4 -15
View File
@@ -1,19 +1,8 @@
"""Alibaba Cloud DashScope provider profiles.
"""Alibaba Cloud DashScope provider profiles (intl + CN, plus the Model Studio
Token Plan flat-token tier with its own key/endpoints — one module per vendor).
DashScope has region-split endpoints with the same key type:
- ``alibaba`` → dashscope-intl.aliyuncs.com (international)
- ``alibaba-cn`` → dashscope.aliyuncs.com (mainland China)
The Model Studio Token Plan (flat-token tier of the SAME vendor/service,
same OpenAI-compatible protocol, its own key + endpoints) registers here
too rather than as a new plugin directory — one module per vendor, matching
how the kimi module carries both of its endpoint variants:
- ``alibaba-token-plan`` → token-plan.ap-southeast-1.maas.aliyuncs.com
- ``alibaba-token-plan-cn`` → token-plan.cn-beijing.maas.aliyuncs.com
Profile names match the models.dev catalog keys exactly
(``alibaba`` / ``alibaba-cn``) so model metadata lines up and
``model.provider: alibaba-cn`` resolves at runtime (#73265).
Profile names match models.dev catalog keys exactly so model metadata lines up
and ``model.provider: alibaba-cn`` resolves at runtime.
"""
from providers import register_provider
+2 -10
View File
@@ -15,11 +15,7 @@ class AnthropicProfile(ProviderProfile):
"""Native Anthropic — uses x-api-key header, not Bearer."""
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Anthropic uses x-api-key header and anthropic-version."""
if not api_key:
@@ -31,11 +27,7 @@ class AnthropicProfile(ProviderProfile):
req.add_header("Accept", "application/json")
with open_credentialed_url(req, timeout=timeout) as resp:
data = json.loads(resp.read().decode())
return [
m["id"]
for m in data.get("data", [])
if isinstance(m, dict) and "id" in m
]
return [m["id"] for m in data.get("data", []) if isinstance(m, dict) and "id" in m]
except Exception as exc:
logger.debug("fetch_models(anthropic): %s", exc)
return None
@@ -1,8 +1,5 @@
"""Microsoft Foundry provider profile.
Azure Foundry exposes an OpenAI-compatible endpoint; users supply their own
base URL at setup since endpoints are per-resource.
"""
"""Microsoft Foundry provider profile: OpenAI-compatible, per-resource base URL
supplied by the user at setup."""
from providers import register_provider
from providers.base import ProviderProfile
+1 -5
View File
@@ -8,11 +8,7 @@ class BedrockProfile(ProviderProfile):
"""AWS Bedrock — no REST /v1/models endpoint; uses AWS SDK."""
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Bedrock model listing requires AWS SDK, not a REST call."""
return None
+29 -96
View File
@@ -1,28 +1,6 @@
"""CommandCode provider profile.
CommandCode provides a unified API that fronts 20+ models from DeepSeek, Qwen,
Kimi, GLM, MiniMax, StepFun, Xiaomi Mimo, Google Gemini, and OpenAI GPT — all
accessible through either OpenAI-compatible chat completions or Anthropic
Messages endpoints from a single base URL and API key.
Two provider profiles are registered:
``commandcode``
``api_mode=chat_completions`` — standard OpenAI-compatible endpoint.
Model prefix: ``deepseek/deepseek-v4-pro``, ``Qwen/Qwen3.7-Max``, etc.
``commandcode-anthropic``
``api_mode=anthropic_messages`` — Anthropic Messages API-compatible.
Model names: ``claude-sonnet-4-6``, ``claude-opus-4-7``,
``claude-haiku-4-5-20251001``.
Both use the same ``COMMANDCODE_API_KEY`` env var and
``https://api.commandcode.ai/provider/v1`` base URL. The
``commandcode-anthropic`` profile relies on ``agent/anthropic_adapter.py``
recognizing the ``api.commandcode.ai`` hostname for Bearer auth (the
CommandCode /anthropic endpoint uses ``Authorization: Bearer``, not
Anthropic's native ``x-api-key`` header).
"""
"""CommandCode provider profiles: ``commandcode`` (chat_completions) and
``commandcode-anthropic`` (anthropic_messages, Bearer auth — see
``agent/anthropic_adapter.py``). Same key and base URL for both."""
from __future__ import annotations
@@ -35,74 +13,63 @@ from providers.base import ProviderProfile, _profile_user_agent
logger = logging.getLogger(__name__)
# ── Shared constants ──────────────────────────────────────────────────────────
_COMMANDCODE_BASE = "https://api.commandcode.ai/provider/v1"
_COMMANDCODE_MODELS_URL = f"{_COMMANDCODE_BASE}/models"
# Both profiles authenticate with the same key; each carries its own base-URL
# override var so each renders its own card on the desktop Keys tab (rows are
# keyed by env var, and the shared API key attributes to the first profile).
_COMMANDCODE_ENV = ("COMMANDCODE_API_KEY", "COMMANDCODE_BASE_URL")
_COMMANDCODE_ANTHROPIC_ENV = ("COMMANDCODE_API_KEY", "COMMANDCODE_ANTHROPIC_BASE_URL")
def _fetch_commandcode_models(
timeout: float = 10.0,
base_url: str | None = None,
) -> list[str] | None:
"""Fetch the live model list from the CommandCode /models endpoint.
"""Fetch model IDs from the public (unauthenticated) /models endpoint.
Returns a flat list of model IDs or None on failure.
No auth required — the public models endpoint is open.
``base_url`` overrides the endpoint only when the caller passed a URL
that differs from the default ``_COMMANDCODE_BASE`` (a user-configured
``model.base_url`` / ``COMMANDCODE_BASE_URL`` pointing at a proxy or
custom deployment). The picker passes base_url unconditionally, falling
back to the profile default — equality means "not customised".
The picker passes base_url unconditionally, so only a value differing from
the default counts as a customised endpoint.
"""
caller_base = (base_url or "").strip()
if caller_base and caller_base.rstrip("/") != _COMMANDCODE_BASE.rstrip("/"):
models_url = caller_base.rstrip("/") + "/models"
else:
models_url = _COMMANDCODE_MODELS_URL
caller_base = (base_url or "").strip().rstrip("/")
custom = caller_base and caller_base != _COMMANDCODE_BASE
models_url = caller_base + "/models" if custom else _COMMANDCODE_MODELS_URL
try:
req = urllib.request.Request(models_url)
req.add_header("Accept", "application/json")
req.add_header("User-Agent", _profile_user_agent())
with urllib.request.urlopen(req, timeout=timeout) as resp:
data = json.loads(resp.read().decode())
# Response shape: {"object": "list", "data": [{"id": "..."}, ...]}
return [
m["id"]
for m in data.get("data", [])
if isinstance(m, dict) and "id" in m
]
return [m["id"] for m in data.get("data", []) if isinstance(m, dict) and "id" in m]
except Exception as exc:
logger.debug("fetch_models(commandcode): %s", exc)
return None
# ── Chat Completions profile ──────────────────────────────────────────────────
class CommandCodeProfile(ProviderProfile):
"""CommandCode — OpenAI-compatible chat completions endpoint."""
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Fetch from the public CommandCode /models endpoint."""
return _fetch_commandcode_models(timeout=timeout, base_url=base_url)
class CommandCodeAnthropicProfile(ProviderProfile):
"""CommandCode — Anthropic Messages API-compatible endpoint."""
def fetch_models(
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Public /models endpoint, filtered to Anthropic-family models."""
all_models = _fetch_commandcode_models(timeout=timeout, base_url=base_url)
if all_models is None:
return None
return [m for m in all_models if m.startswith("claude-")]
commandcode = CommandCodeProfile(
name="commandcode",
aliases=("commandcode-chat",),
api_mode="chat_completions",
env_vars=_COMMANDCODE_ENV,
# Same key as the anthropic profile; distinct base-URL override vars so each
# profile renders its own card on the desktop Keys tab (rows keyed by env var).
env_vars=("COMMANDCODE_API_KEY", "COMMANDCODE_BASE_URL"),
display_name="CommandCode",
description="CommandCode — 20+ models via OpenAI-compatible API",
signup_url="https://commandcode.ai/",
@@ -124,53 +91,19 @@ commandcode = CommandCodeProfile(
default_aux_model="deepseek/deepseek-v4-flash",
)
# ── Anthropic Messages profile ────────────────────────────────────────────────
class CommandCodeAnthropicProfile(ProviderProfile):
"""CommandCode — Anthropic Messages API-compatible endpoint.
Uses Bearer auth (same API key), not Anthropic's native x-api-key header.
``agent/anthropic_adapter.py`` must recognize ``api.commandcode.ai``
as a Bearer-auth domain for this to work.
"""
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
) -> list[str] | None:
"""Fetch from the public CommandCode /models endpoint.
Filter to Anthropic-family models only (claude-*).
"""
all_models = _fetch_commandcode_models(timeout=timeout, base_url=base_url)
if all_models is None:
return None
return [m for m in all_models if m.startswith("claude-")]
commandcode_anthropic = CommandCodeAnthropicProfile(
name="commandcode-anthropic",
aliases=("commandcode-claude",),
api_mode="anthropic_messages",
env_vars=_COMMANDCODE_ANTHROPIC_ENV,
env_vars=("COMMANDCODE_API_KEY", "COMMANDCODE_ANTHROPIC_BASE_URL"),
display_name="CommandCode (Anthropic)",
description="CommandCode — Claude models via Anthropic Messages API",
signup_url="https://commandcode.ai/",
base_url=_COMMANDCODE_BASE,
models_url=_COMMANDCODE_MODELS_URL,
fallback_models=(
"claude-sonnet-4-6",
"claude-opus-4-7",
"claude-haiku-4-5-20251001",
),
fallback_models=("claude-sonnet-4-6", "claude-opus-4-7", "claude-haiku-4-5-20251001"),
default_aux_model="claude-haiku-4-5-20251001",
)
# ── Registration ──────────────────────────────────────────────────────────────
register_provider(commandcode)
register_provider(commandcode_anthropic)
@@ -1,11 +1,9 @@
"""GitHub Copilot ACP provider profile.
copilot-acp does not speak OpenAI-over-HTTP: it drives an external ACP
subprocess over stdio. The profile therefore supplies its own client through
:meth:`ProviderProfile.create_client` instead of letting the core build an
``openai.OpenAI``. That hook is the registration seam — this profile is its
in-tree consumer, and an out-of-tree ACP provider registered from
``~/.hermes/plugins/model-providers/`` or a pip entry point uses the exact same
subprocess over stdio, so the profile supplies its own client via
:meth:`ProviderProfile.create_client`. An out-of-tree ACP provider registered
from ``~/.hermes/plugins/model-providers/`` or a pip entry point uses the same
three lines without touching core.
"""
@@ -25,11 +23,7 @@ class CopilotACPProfile(ProviderProfile):
return CopilotACPClient(**client_kwargs)
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Model listing is handled by the ACP subprocess."""
return None
+24 -47
View File
@@ -1,13 +1,8 @@
"""Copilot / GitHub Models provider profile.
Copilot uses per-model api_mode routing:
- GPT-5+ / Codex models → codex_responses
- Claude models → anthropic_messages
- Everything else → chat_completions (this profile covers that subset)
Key quirks for the chat_completions subset:
- Editor attribution headers (via copilot_default_headers())
- GitHub Models reasoning extra_body (model-catalog gated)
Core routes GPT-5+/Codex -> codex_responses and Claude -> anthropic_messages;
this profile covers the chat_completions remainder: editor attribution headers
(copilot_default_headers()) and catalog-gated GitHub Models reasoning.
"""
from typing import Any
@@ -27,46 +22,28 @@ class CopilotProfile(ProviderProfile):
supports_reasoning: bool = False,
**ctx,
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
if supports_reasoning and model:
try:
from hermes_cli.models import github_model_reasoning_efforts
if not (supports_reasoning and model):
return {}, {}
try:
from hermes_cli.models import clamp_reasoning_effort_to_supported, github_model_reasoning_efforts
supported_efforts = github_model_reasoning_efforts(model)
if supported_efforts and reasoning_config:
effort = reasoning_config.get("effort", "medium")
# Honor the requested level when the live Copilot catalog
# lists it as supported: gpt-5.5/gpt-5.4 DO support
# ``xhigh``. Otherwise clamp to the nearest WEAKER
# supported level via the shared ladder helper — the old
# ad-hoc rules dropped everything unrecognized to
# ``medium``, which inverted the ladder: ``ultra`` (the
# strongest ask) resolved weaker than an explicit
# ``high`` (#74295).
if effort not in supported_efforts:
from hermes_cli.models import (
clamp_reasoning_effort_to_supported,
)
effort = clamp_reasoning_effort_to_supported(
effort, list(supported_efforts)
)
if effort not in supported_efforts:
# Unrecognized/bespoke level the ladder can't
# place — fall back to medium, then to the
# catalog's first entry.
effort = (
"medium"
if "medium" in supported_efforts
else supported_efforts[0]
)
if effort in supported_efforts:
extra_body["reasoning"] = {"effort": effort}
elif supported_efforts:
extra_body["reasoning"] = {"effort": "medium"}
except Exception:
pass
return extra_body, {}
supported = github_model_reasoning_efforts(model)
if not supported:
return {}, {}
if not reasoning_config:
return {"reasoning": {"effort": "medium"}}, {}
effort = reasoning_config.get("effort", "medium")
# Honor a level the live catalog lists; otherwise clamp to the nearest
# WEAKER supported level (never drop straight to medium, which inverted
# the ladder: ultra < high). Bespoke levels the ladder can't place fall
# to medium (or the first supported level).
if effort not in supported:
effort = clamp_reasoning_effort_to_supported(effort, list(supported))
if effort not in supported:
effort = "medium" if "medium" in supported else supported[0]
return {"reasoning": {"effort": effort}}, {}
except Exception:
return {}, {}
copilot = CopilotProfile(
+25 -96
View File
@@ -1,127 +1,65 @@
"""Custom / Ollama (local) provider profile.
Covers any endpoint registered as provider="custom", including local
Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on
Volcengine ARK, vLLM, llama.cpp). Key quirks:
- ollama_num_ctx → extra_body.options.num_ctx (local context window)
- reasoning_config disabled → top-level reasoning_effort="none"
(Ollama /v1/chat/completions ignores think=False — ollama#14820)
+ extra_body.think = False only on Ollama URLs (/api/chat and proxies)
- reasoning_config enabled + effort → top-level reasoning_effort
(the native OpenAI-compatible format GLM/ARK expect; unset omits it
so the endpoint's server default applies)
"""
"""Custom / Ollama (local) provider profile: any endpoint registered as
provider="custom" (Ollama, vLLM, llama.cpp, GLM-5.2 on ARK, …)."""
from typing import Any
from urllib.parse import urlparse
from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort
from providers import register_provider
from providers.base import ProviderProfile
def _looks_like_ollama_endpoint(base_url: str | None) -> bool:
"""True when ``base_url`` is an Ollama host, not a generic OpenAI-compat relay.
"""True only for explicit Ollama signatures (port 11434 or an ``ollama`` host label).
``think`` is an Ollama-native extra_body field. Strict hosts (Mistral
``extra=forbid``, Groq, …) reject it with HTTP 422. Match only explicit
Ollama signatures — default port 11434, or ``ollama`` as a hostname
label — not arbitrary localhost (llama.cpp / vLLM / LM Studio).
``think`` is Ollama-native; strict hosts (Mistral, Groq) 422 on it, and
arbitrary localhost may be llama.cpp / vLLM / LM Studio.
"""
raw = (base_url or "").strip()
if not raw:
return False
parsed = urlparse(raw if "://" in raw else f"//{raw}")
# urlparse raises ValueError for non-integer / out-of-range ports
# ("http://host:99999/v1" parses fine in the OpenAI client, so the URL
# is reachable here). Treat a malformed port as "not Ollama" instead of
# killing the whole kwargs build — same try/except shape the 11434
# check in hermes_cli/models.py uses, not the same detection logic.
# urlparse raises ValueError on malformed ports ("host:99999"); treat as not-Ollama.
try:
if parsed.port == 11434:
return True
except ValueError:
return False
host = (parsed.hostname or "").lower().rstrip(".")
if not host:
return False
if host == "ollama.com" or host.endswith(".ollama.com"):
return True
return "ollama" in host.split(".")
return bool(host) and (host == "ollama.com" or host.endswith(".ollama.com") or "ollama" in host.split("."))
class CustomProfile(ProviderProfile):
"""Custom/Ollama local provider — think=false and num_ctx support."""
def build_api_kwargs_extras(
self,
*,
reasoning_config: dict | None = None,
ollama_num_ctx: int | None = None,
**ctx: Any,
self, *, reasoning_config: dict | None = None, ollama_num_ctx: int | None = None, **ctx: Any
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
top_level: dict[str, Any] = {}
# Ollama context window
if ollama_num_ctx:
options = extra_body.get("options", {})
options["num_ctx"] = ollama_num_ctx
extra_body["options"] = options
extra_body["options"] = {"num_ctx": ollama_num_ctx}
# Reasoning / thinking control for custom OpenAI-compatible endpoints
# (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …).
#
# - disabled → top-level reasoning_effort="none"; extra_body.think
# = False only on Ollama URLs (Ollama's thinking-off flag)
# - enabled + effort set → TOP-LEVEL reasoning_effort string, the
# format GLM-5.2/ARK and other OpenAI-compatible reasoning APIs
# expect (GLM documents "high" and "max"; "max" is its default).
# - enabled + no effort → omit both, so the endpoint applies its own
# server-side default (do NOT force a level the user didn't pick).
#
# We deliberately do NOT emit ``think=True`` on enable: it is an
# Ollama-only flag and thinking is already server-default-on for these
# backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't
# recognize it. Mirrors the DeepSeek/Zai profile precedent. The same
# constraint applies to ``think=False`` on disable — Mistral/Groq
# reject unknown fields (HTTP 422 extra_forbidden) rather than ignoring
# them, so that flag stays Ollama-URL-gated.
# disabled -> top-level reasoning_effort="none" (Ollama's /v1 ignores
# extra_body.think) plus think=False only on Ollama URLs; enabled+effort ->
# top-level reasoning_effort clamped to the OpenAI-compat wire (GLM/ARK,
# vLLM and SGLang all top out at "max"; "ultra" verbatim 400s); enabled
# without effort -> omit so the server default applies. Never emit
# think=True (Ollama-only flag).
if reasoning_config and isinstance(reasoning_config, dict):
_effort = (reasoning_config.get("effort") or "").strip().lower()
_enabled = reasoning_config.get("enabled", True)
if _effort == "none" or _enabled is False:
# Ollama's /v1/chat/completions silently ignores
# extra_body.think (only /api/chat honours it — ollama#14820)
# but respects the top-level reasoning_effort field (#25758).
# Always emit reasoning_effort="none"; only add think=False
# when the URL is actually Ollama.
effort = (reasoning_config.get("effort") or "").strip().lower()
if effort == "none" or reasoning_config.get("enabled", True) is False:
top_level["reasoning_effort"] = "none"
if _looks_like_ollama_endpoint(ctx.get("base_url")):
extra_body["think"] = False
elif _effort:
# Clamp the internal ladder onto the widest OpenAI-compatible
# wire vocabulary (shared policy in agent.reasoning_effort) —
# GLM/ARK, vLLM and SGLang all top out at "max"; forwarding
# "ultra" verbatim is a guaranteed 400 (#89503).
from agent.reasoning_effort import (
OPENAI_COMPAT_WIRE_EFFORTS,
clamp_effort,
)
top_level["reasoning_effort"] = clamp_effort(
_effort, OPENAI_COMPAT_WIRE_EFFORTS
)
elif effort:
top_level["reasoning_effort"] = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS)
return extra_body, top_level
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Custom/Ollama: base_url is user-configured; fetch if set."""
"""base_url is user-configured; fetch only if set."""
if not (base_url or self.base_url):
return None
return super().fetch_models(api_key=api_key, base_url=base_url, timeout=timeout)
@@ -129,20 +67,11 @@ class CustomProfile(ProviderProfile):
custom = CustomProfile(
name="custom",
aliases=(
"ollama",
"local",
"vllm",
"llamacpp",
"llama.cpp",
"llama-cpp",
),
aliases=("ollama", "local", "vllm", "llamacpp", "llama.cpp", "llama-cpp"),
env_vars=(), # No fixed key — custom endpoint
base_url="", # User-configured
# Without this, no max_tokens is sent and Ollama falls back to its internal
# num_predict=128, truncating responses after a few tokens (#39281). This is
# only a floor used when the user hasn't set model.max_tokens — they can
# override per-model — so we set it generously rather than lowballing it.
# Floor only (user model.max_tokens overrides); without it Ollama falls
# back to num_predict=128 and truncates.
default_max_tokens=65536,
)
+14 -40
View File
@@ -1,33 +1,20 @@
"""DeepInfra provider profile.
DeepInfra is an OpenAI-compatible inference gateway that hosts 100+ open
models (Step, GLM, Kimi, DeepSeek, MiniMax, Nemotron, Mistral, Qwen, …) as
well as image-gen / TTS / STT / embedding endpoints. The chat surface is
wired in through this profile; non-chat surfaces are wired in through
their respective plugin subsystems (``plugins/image_gen/deepinfra`` and
the TTS/STT dispatchers in ``tools/``).
"""
"""DeepInfra provider profile (chat surface; image-gen/TTS/STT are wired via
their own plugin subsystems)."""
from providers import register_provider
from providers.base import ProviderProfile
class _DeepInfraProfile(ProviderProfile):
"""DeepInfra profile with live vision-default discovery.
Owns its own vision default so shared vision resolution in
``agent/auxiliary_client.py`` stays provider-agnostic (a
``default_vision_model()`` hook call instead of an ``if provider ==
"deepinfra"`` branch reaching into the catalog helpers).
"""
"""DeepInfra profile with live vision-default discovery, so shared vision
resolution in ``agent/auxiliary_client.py`` stays provider-agnostic."""
def default_vision_model(self): # type: ignore[override]
"""First vision-capable *chat* model from the live catalog, or None.
Key-gated so a box without ``DEEPINFRA_API_KEY`` never pays the
catalog round-trip. Requires the ``chat`` surface tag (not just the
``vision`` capability) so an image-gen/edit model that merely carries
a ``vision`` tag can't be picked as a chat-completions vision backend.
Key-gated so a box without DEEPINFRA_API_KEY never pays the round-trip.
Requires the ``chat`` surface tag so an image-gen model carrying a
``vision`` tag can't be picked as a chat-completions vision backend.
"""
from agent.secret_scope import get_secret
@@ -41,10 +28,8 @@ class _DeepInfraProfile(ProviderProfile):
for item in items or []:
metadata = item.get("metadata") or {}
tags = metadata.get("tags") if isinstance(metadata, dict) else None
if isinstance(tags, list) and "vision" in tags:
model_id = item.get("id")
if model_id:
return model_id
if isinstance(tags, list) and "vision" in tags and item.get("id"):
return item["id"]
return None
@@ -57,24 +42,13 @@ deepinfra = _DeepInfraProfile(
env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"),
base_url="https://api.deepinfra.com/v1/openai",
auth_type="api_key",
# The catalog spans models with different output limits. Omitting a
# provider-wide default lets DeepInfra apply its documented per-model cap;
# an explicit user ``agent.max_tokens`` still passes through normally.
# No provider-wide cap: DeepInfra applies its documented per-model limit.
default_max_tokens=None,
# Auxiliary model — cheap/fast chat model the same provider uses for
# side tasks (context compression, session search, web extract,
# vision). This is the *only* hardcoded DeepInfra model in the
# integration: aux resolution is synchronous (no time for a catalog
# round-trip on every agent turn), so we need one explicit choice.
# Every other surface (chat picker, image-gen, tts, stt, pricing)
# discovers models live from
# ``api.deepinfra.com/v1/openai/models?filter=true&sort_by=hermes``.
# The only hardcoded DeepInfra model: aux resolution is synchronous, so it
# can't wait on a catalog round-trip. Everything else is discovered live.
default_aux_model="deepseek-ai/DeepSeek-V4-Flash",
# ``fallback_models`` deliberately empty — the live catalog at
# ``hermes_cli/models.py::_fetch_deepinfra_models`` is the source of
# truth. When the live fetch fails (network/DNS), the picker shows
# no options, which is preferable to silently routing the user to a
# model that may have been retired upstream.
# Empty on purpose: the live catalog is the source of truth; an empty picker
# beats silently routing to a retired model.
fallback_models=(),
)
+24 -78
View File
@@ -1,96 +1,45 @@
"""DeepSeek provider profile.
DeepSeek's V4 family defaults to thinking-mode ON when ``extra_body.thinking``
is unset. The API then returns ``reasoning_content`` and starts enforcing
the contract that subsequent turns echo it back; combined with how Hermes
replays history this lands on the notorious HTTP 400
``reasoning_content must be passed back`` error after the first tool call
(#15700, #17212, #17825).
This profile overrides :meth:`build_api_kwargs_extras` to mirror the Kimi /
Moonshot wire shape that DeepSeek's OpenAI-compat endpoint expects:
{"reasoning_effort": "<low|medium|high|max>",
"extra_body": {"thinking": {"type": "enabled" | "disabled"}}}
Non-thinking models (``deepseek-v3-*`` variants) are left as no-ops so we
don't perturb the V3 wire format.
The legacy aliases ``deepseek-chat`` / ``deepseek-reasoner`` were retired on
2026-07-24. Use ``deepseek-v4-flash`` or ``deepseek-v4-pro``; Hermes remaps
the retired IDs in ``hermes_cli.model_normalize``.
V4 defaults to thinking ON when ``extra_body.thinking`` is unset, and then
requires ``reasoning_content`` to be echoed back on later turns (HTTP 400 after
the first tool call otherwise). This profile sets ``thinking`` explicitly and
maps effort onto DeepSeek's ``reasoning_effort``; V3 models are left untouched.
Retired ``deepseek-chat``/``deepseek-reasoner`` IDs are remapped in
``hermes_cli.model_normalize`` before reaching here.
"""
from __future__ import annotations
from typing import Any
from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, clamp_effort
from providers import register_provider
from providers.base import ProviderProfile
def _model_supports_thinking(model: str | None) -> bool:
"""DeepSeek thinking-capable model families.
Currently covers the V4 family (``deepseek-v4-pro``, ``deepseek-v4-flash``,
and any future ``deepseek-v4-*`` variants). Retired aliases are remapped
before requests leave Hermes, so they are not listed here.
"""
m = (model or "").strip().lower()
if not m:
return False
if m.startswith("deepseek-v") and not m.startswith("deepseek-v3"):
# deepseek-v4-*, deepseek-v5-*, etc. — every V4+ generation has
# thinking. v3 explicitly excluded.
return True
return False
class DeepSeekProfile(ProviderProfile):
"""DeepSeek — extra_body.thinking + top-level reasoning_effort."""
def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
m = (model or "").strip().lower()
# deepseek-v4-* and every later generation; v3 explicitly excluded.
if not m.startswith("deepseek-v") or m.startswith("deepseek-v3"):
return {}, {}
rc = reasoning_config if isinstance(reasoning_config, dict) else None
# Always set explicitly (default enabled, matching the API default) to
# avoid the reasoning_content echo trap on subsequent turns.
if rc is not None and rc.get("enabled") is False:
return {"thinking": {"type": "disabled"}}, {}
top_level: dict[str, Any] = {}
if not _model_supports_thinking(model):
# V3 / unknown — leave wire format untouched, current behavior.
return extra_body, top_level
# Determine enabled/disabled. Default is enabled to match DeepSeek's
# API default; the API requires this to be set explicitly to avoid the
# reasoning_content echo trap on subsequent turns.
enabled = True
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
enabled = False
extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"}
if not enabled:
return extra_body, top_level
# Effort mapping via the shared vocabulary in agent.reasoning_effort
# (DeepSeek V4: low/medium/high/max, xhigh rounds up to max). When no
# effort is set we omit reasoning_effort so DeepSeek applies its
# server default (currently high).
if isinstance(reasoning_config, dict):
from agent.reasoning_effort import (
DEEPSEEK_V4_EFFORTS,
DEEPSEEK_V4_OVERRIDES,
clamp_effort,
)
effort = (reasoning_config.get("effort") or "").strip().lower()
if effort and effort != "none":
clamped = clamp_effort(
effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES
)
if clamped in DEEPSEEK_V4_EFFORTS:
top_level["reasoning_effort"] = clamped
return extra_body, top_level
# No effort -> omit reasoning_effort so DeepSeek applies its server default.
effort = (rc.get("effort") or "").strip().lower() if rc is not None else ""
if effort and effort != "none":
clamped = clamp_effort(effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES)
if clamped in DEEPSEEK_V4_EFFORTS:
top_level["reasoning_effort"] = clamped
return {"thinking": {"type": "enabled"}}, top_level
deepseek = DeepSeekProfile(
@@ -100,10 +49,7 @@ deepseek = DeepSeekProfile(
display_name="DeepSeek",
description="DeepSeek — native DeepSeek API",
signup_url="https://platform.deepseek.com/",
fallback_models=(
"deepseek-v4-pro",
"deepseek-v4-flash",
),
fallback_models=("deepseek-v4-pro", "deepseek-v4-flash"),
base_url="https://api.deepseek.com/v1",
default_aux_model="deepseek-v4-flash",
)
+5 -17
View File
@@ -1,13 +1,5 @@
"""Fireworks AI provider profile.
Fireworks AI serves fast, production-grade inference for open and proprietary
models through an OpenAI-compatible chat-completions endpoint.
Address models directly by their catalog ID, e.g.
``accounts/fireworks/models/kimi-k2p6`` or ``accounts/fireworks/models/glm-5p2``.
Model IDs here track the canonical Fireworks catalog (fw-ai/fireconnect
``setup-cli``).
"""
"""Fireworks AI provider profile. Models are addressed by full catalog ID
(``accounts/fireworks/models/<slug>``), tracking fw-ai/fireconnect ``setup-cli``."""
from hermes_cli import __version__ as _HERMES_VERSION
from providers import register_provider
@@ -23,19 +15,15 @@ fireworks = ProviderProfile(
env_vars=("FIREWORKS_API_KEY",),
base_url="https://api.fireworks.ai/inference/v1",
auth_type="api_key",
# Attribution headers sent on every Fireworks request. Values match the
# canonical Hermes set in agent/auxiliary_client.py. Applied through the
# generic profile.default_headers path, so they survive switch_model and
# credential rotation.
# Attribution headers (canonical Hermes set); via default_headers so they
# survive switch_model and credential rotation.
default_headers={
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
},
# Auxiliary model for cheap tasks (compaction, title generation, vision).
# A standard pay-as-you-go catalog ``/models/`` ID.
default_aux_model="accounts/fireworks/models/glm-5p2",
# Curated safety net shown in the picker when the live catalog fetch fails.
# Picker safety net when the live catalog fetch fails.
fallback_models=(
"accounts/fireworks/models/kimi-k2p6",
"accounts/fireworks/models/glm-5p2",
+12 -29
View File
@@ -1,12 +1,7 @@
"""Google Gemini provider profiles.
"""Google Gemini (AI Studio) provider profile.
gemini: Google AI Studio (API key) — uses GeminiNativeClient
Reports api_mode="chat_completions" but uses a custom native client
that bypasses the standard OpenAI transport. The profile captures auth
and endpoint metadata for auth.py / runtime_provider.py migration, and
carries the thinking_config translation hook so the transport's profile
path produces the same extra_body shape the legacy flag path did.
Reports api_mode="chat_completions" but runs on GeminiNativeClient; this
profile carries auth/endpoint metadata and the thinking_config translation hook.
"""
from typing import Any
@@ -18,34 +13,22 @@ from providers.base import ProviderProfile
class GeminiProfile(ProviderProfile):
"""Gemini — translate reasoning_config to thinking_config in extra_body."""
def build_extra_body(
self, *, session_id: str | None = None, **context: Any
) -> dict[str, Any]:
"""Emit extra_body.thinking_config (native) or extra_body.extra_body.google.thinking_config
(OpenAI-compat /openai subpath), mirroring the legacy path's behavior.
"""
def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]:
"""Native: ``thinking_config``; OpenAI-compat /openai subpath:
``extra_body.google.thinking_config`` (snake_case)."""
from agent.transports.chat_completions import (
_build_gemini_thinking_config,
_is_gemini_openai_compat_base_url,
_snake_case_gemini_thinking_config,
)
model = context.get("model") or ""
reasoning_config = context.get("reasoning_config")
base_url = context.get("base_url") or self.base_url
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
if not raw_thinking_config:
raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config"))
if not raw:
return {}
body: dict[str, Any] = {}
if self.name == "gemini" and _is_gemini_openai_compat_base_url(base_url):
thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config)
if thinking_config:
body["extra_body"] = {"google": {"thinking_config": thinking_config}}
else:
body["thinking_config"] = raw_thinking_config
return body
if self.name == "gemini" and _is_gemini_openai_compat_base_url(context.get("base_url") or self.base_url):
thinking_config = _snake_case_gemini_thinking_config(raw)
return {"extra_body": {"google": {"thinking_config": thinking_config}}} if thinking_config else {}
return {"thinking_config": raw}
gemini = GeminiProfile(
+1 -3
View File
@@ -13,9 +13,7 @@ gmi = ProviderProfile(
env_vars=("GMI_API_KEY", "GMI_BASE_URL"),
base_url="https://api.gmi-serving.com/v1",
auth_type="api_key",
# Attribution so GMI can identify traffic from Hermes Agent.
# The generic profile.default_headers fallback in run_agent.py and
# agent/auxiliary_client.py picks this up at client construction time.
# Attribution so GMI can identify Hermes Agent traffic.
default_headers={"User-Agent": f"HermesAgent/{_HERMES_VERSION}"},
default_aux_model="google/gemini-3.1-flash-lite-preview",
fallback_models=(
@@ -10,10 +10,7 @@ huggingface = ProviderProfile(
display_name="HuggingFace",
description="HuggingFace Inference API",
signup_url="https://huggingface.co/settings/tokens",
fallback_models=(
"Qwen/Qwen3.5-72B-Instruct",
"deepseek-ai/DeepSeek-V3.2",
),
fallback_models=("Qwen/Qwen3.5-72B-Instruct", "deepseek-ai/DeepSeek-V3.2"),
base_url="https://router.huggingface.co/v1",
)
+28 -85
View File
@@ -1,36 +1,32 @@
"""Kimi / Moonshot provider profiles.
Kimi has dual endpoints:
- sk-kimi-* keys → api.kimi.com/coding (Anthropic Messages API)
- legacy keys → api.moonshot.ai/v1 (OpenAI chat completions)
This module covers the chat_completions path (/v1 endpoint).
"""
"""Kimi / Moonshot provider profiles (chat_completions path; sk-kimi-* keys are
redirected to api.kimi.com/coding by core)."""
from typing import Any
from urllib.parse import urlparse
from agent.reasoning_effort import KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, clamp_effort, requested_effort
from hermes_cli import __version__ as _HERMES_VERSION
from providers import register_provider
from providers.base import OMIT_TEMPERATURE, ProviderProfile
_HEADERS = {
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
}
def _is_confirmed_kimi_coding_url(base_url: str) -> bool:
"""Return True only for Kimi Code's canonical HTTPS API surfaces."""
try:
parsed = urlparse(base_url)
port = parsed.port
p = urlparse(base_url)
port = p.port
except ValueError:
return False
return (
parsed.scheme.lower() == "https"
and (parsed.hostname or "").lower() == "api.kimi.com"
and port in (None, 443)
and parsed.username is None
and parsed.password is None
and parsed.path.rstrip("/") in {"/coding", "/coding/v1"}
and not parsed.query
and not parsed.fragment
p.scheme.lower() == "https" and (p.hostname or "").lower() == "api.kimi.com" and port in (None, 443)
and p.username is None and p.password is None
and p.path.rstrip("/") in {"/coding", "/coding/v1"} and not p.query and not p.fragment
)
@@ -38,22 +34,15 @@ class KimiProfile(ProviderProfile):
"""Kimi/Moonshot — temperature omitted, thinking xor reasoning_effort."""
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Use Kimi Code's OpenAI-compatible surface for model discovery."""
"""Use Kimi Code's OpenAI-compatible surface for model discovery; the bare
``k3`` slug is only served there, so it is filtered off other endpoints."""
effective_base = (base_url or self.base_url or "").rstrip("/")
confirmed_coding_endpoint = _is_confirmed_kimi_coding_url(effective_base)
if confirmed_coding_endpoint and urlparse(effective_base).path.rstrip("/") == "/coding":
effective_base += "/v1"
models = super().fetch_models(
api_key=api_key,
base_url=effective_base or None,
timeout=timeout,
)
models = super().fetch_models(api_key=api_key, base_url=effective_base or None, timeout=timeout)
if models is None or confirmed_coding_endpoint:
return models
return [model for model in models if model.strip().lower() != "k3"]
@@ -61,53 +50,15 @@ class KimiProfile(ProviderProfile):
def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, **context
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Kimi reasoning controls.
Moonshot's wire shape treats ``extra_body.thinking`` (a binary toggle)
and a top-level ``reasoning_effort`` as mutually exclusive — sending
both is at best redundant and risks "cannot specify both 'thinking' and
'reasoning_effort'" (HTTP 400). This mirrors the kimi-k2 handling on the
opencode-go relay: send effort when one is requested, otherwise fall
back to ``extra_body.thinking`` — never both.
"""
extra_body = {}
top_level = {}
if not reasoning_config or not isinstance(reasoning_config, dict):
# No config → thinking enabled, let the server pick the depth.
# (Previously also sent reasoning_effort="medium", which paired
# thinking + effort on every default call.)
extra_body["thinking"] = {"type": "enabled"}
return extra_body, top_level
enabled = reasoning_config.get("enabled", True)
if enabled is False:
extra_body["thinking"] = {"type": "disabled"}
return extra_body, top_level
# Enabled: prefer an explicit effort; only fall back to extra_body
# thinking when no recognized effort is requested.
# K3's vocabulary (low/high/max, default high) and its documented
# rounding (medium→high, xhigh→max) are declared in
# agent.reasoning_effort — shared with the chat-completions
# transport's Kimi path so both stay in sync.
from agent.reasoning_effort import (
KIMI_K3_EFFORTS,
KIMI_K3_OVERRIDES,
clamp_effort,
)
effort = (reasoning_config.get("effort") or "").strip().lower()
if effort and effort != "none":
k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES)
else:
k3_effort = None
"""Moonshot treats extra_body.thinking and reasoning_effort as mutually
exclusive (400 on both): send effort when requested, else the toggle."""
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled", True) is False:
return {"thinking": {"type": "disabled"}}, {}
effort = requested_effort(reasoning_config)
k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES) if effort != "none" else None
if k3_effort in KIMI_K3_EFFORTS:
top_level["reasoning_effort"] = k3_effort
else:
extra_body["thinking"] = {"type": "enabled"}
return extra_body, top_level
return {}, {"reasoning_effort": k3_effort}
return {"thinking": {"type": "enabled"}}, {}
kimi = KimiProfile(
@@ -117,11 +68,7 @@ kimi = KimiProfile(
base_url="https://api.moonshot.ai/v1",
fixed_temperature=OMIT_TEMPERATURE,
default_max_tokens=32000,
default_headers={
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
},
default_headers=dict(_HEADERS),
default_aux_model="kimi-k2-turbo-preview",
)
@@ -132,11 +79,7 @@ kimi_cn = KimiProfile(
base_url="https://api.moonshot.cn/v1",
fixed_temperature=OMIT_TEMPERATURE,
default_max_tokens=32000,
default_headers={
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
},
default_headers=dict(_HEADERS),
default_aux_model="kimi-k2-turbo-preview",
)
+25 -75
View File
@@ -1,29 +1,9 @@
"""Meta Model API (Muse Spark) provider plugin for Hermes Agent.
"""Meta Model API (Muse Spark) provider profile — https://api.meta.ai/v1.
Provider profile for Meta Superintelligence Labs' Muse Spark family, served
via the OpenAI-compatible Meta Model API at ``https://api.meta.ai/v1``.
Bundled from https://github.com/albertodepaola/hermes-meta-provider. Hermes'
provider discovery (``providers/__init__.py``) imports it on first
``get_provider_profile()`` / ``list_providers()`` call, and the module-level
``register_provider()`` below wires it into the registry.
Design notes
------------
* **Zero core edits.** Everything rides on ``ProviderProfile`` hooks. No changes
to hermes' ``model_metadata.py`` / ``models.py`` / ``run_agent.py`` are needed:
- Context window (1M), reasoning and vision capabilities already resolve from
models.dev for the muse-spark family, so no static ctx table entry is required.
- The reasoning dial is emitted as a **top-level ``reasoning_effort``** kwarg
(returned in the ``top_level`` slot of ``build_api_kwargs_extras``), which the
chat-completions transport merges unconditionally. This deliberately avoids
the ``extra_body.reasoning`` path, whose emission is gated by a hardcoded
host allowlist in core (``AIAgent._supports_reasoning_extra_body``) that a
third-party plugin must not edit.
* **Meta 400 on ``reasoning_effort: "none"``.** Muse rejects ``none``; disabling
reasoning maps to ``"minimal"`` instead.
* **``default_max_tokens=16384``.** Muse spends completion budget on hidden
reasoning tokens first; small caps can finish with empty content.
Bundled from albertodepaola/hermes-meta-provider; rides entirely on
ProviderProfile hooks (zero core edits). The reasoning dial is emitted as a
top-level ``reasoning_effort`` kwarg — not ``extra_body.reasoning``, whose
emission is gated by a core host allowlist a third-party plugin must not edit.
"""
from __future__ import annotations
@@ -31,30 +11,11 @@ from __future__ import annotations
import os
from typing import Any
from agent.reasoning_effort import META_AI_EFFORTS, clamp_effort
from providers import register_provider
from providers.base import ProviderProfile
def _resolve_effort(reasoning_config: dict | None) -> str:
"""Map Hermes' reasoning_config to a Meta-safe ``reasoning_effort`` value.
Meta's vocabulary (minimal..xhigh; rejects ``none``) is declared in
agent.reasoning_effort. Disabled/"none" maps to ``minimal`` (the closest
Meta has to off); unset/bespoke levels fall to ``medium``.
"""
rc = reasoning_config or {}
if rc.get("enabled") is False:
return "minimal"
effort = str(rc.get("effort") or "").strip().lower()
if effort in {"", "none"}:
return "minimal" if effort == "none" else "medium"
from agent.reasoning_effort import META_AI_EFFORTS, clamp_effort
clamped = clamp_effort(effort, META_AI_EFFORTS)
return clamped if clamped in META_AI_EFFORTS else "medium"
class MetaAIProfile(ProviderProfile):
"""Meta Model API — top-level reasoning_effort, self-contained."""
@@ -65,19 +26,20 @@ class MetaAIProfile(ProviderProfile):
supports_reasoning: bool = False, # noqa: ARG002 — we self-gate below
**context: Any,
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Emit ``reasoning_effort`` as a top-level api kwarg.
"""Ignores the core ``supports_reasoning`` gate (host-allowlist driven);
Muse Spark always accepts ``reasoning_effort``.
We ignore the core ``supports_reasoning`` gate on purpose: that flag is
driven by a host allowlist in core we cannot (and should not) edit from
an out-of-tree plugin. Muse Spark always accepts ``reasoning_effort``,
so we resolve it from ``reasoning_config`` directly.
Muse 400s on ``none``: disabled/"none" -> ``minimal`` (closest to off);
unset/bespoke levels -> ``medium``.
"""
return {}, {"reasoning_effort": _resolve_effort(reasoning_config)}
def _base_url() -> str:
"""Allow a base-URL override via ``META_BASE_URL`` without editing config."""
return os.getenv("META_BASE_URL", "").strip() or "https://api.meta.ai/v1"
rc = reasoning_config or {}
effort = str(rc.get("effort") or "").strip().lower()
if rc.get("enabled") is False or effort == "none":
mapped = "minimal"
else:
clamped = clamp_effort(effort, META_AI_EFFORTS)
mapped = clamped if clamped in META_AI_EFFORTS else "medium"
return {}, {"reasoning_effort": mapped}
meta_ai = MetaAIProfile(
@@ -88,30 +50,18 @@ meta_ai = MetaAIProfile(
signup_url="https://developer.meta.com/ai/",
# MODEL_API_KEY is Meta's documented env var; the aliases are conveniences.
env_vars=("MODEL_API_KEY", "META_API_KEY", "META_MODEL_API_KEY", "META_BASE_URL"),
base_url=_base_url(),
base_url=os.getenv("META_BASE_URL", "").strip() or "https://api.meta.ai/v1",
auth_type="api_key",
# Responses API is the wire that engages Muse prompt caching: measured
# 0 cached tokens on /v1/chat/completions vs 93-99% cache hits on
# /v1/responses with prompt_cache_retention (see host_mandated_api_mode
# in hermes_cli/providers.py and the retention hint in
# agent/transports/codex.py). The MetaAIProfile chat-completions hook
# above still covers custom OpenAI-compatible endpoints configured with
# a non-api.meta.ai base URL, which fall through to chat_completions.
# Responses API engages Muse prompt caching (0 cached tokens on
# chat/completions vs 93-99% hits on /v1/responses); the chat-completions
# hook above still covers custom non-api.meta.ai base URLs.
api_mode="codex_responses",
# Muse Spark is natively multimodal (image/video/pdf/audio in, text out).
supports_vision=True,
# Cheap contributor tier is a good default for auxiliary tasks
# (compaction, title generation, vision) when this is the main provider.
default_aux_model="muse-spark-1.2-contributor",
# Muse spends completion budget on hidden reasoning tokens first; a low cap
# can finish with empty content. 16k is a safe floor.
# Muse spends completion budget on hidden reasoning first; low caps can
# finish with empty content.
default_max_tokens=16384,
# Curated safety net shown in the picker when the live /v1/models fetch
# fails or no credentials are configured yet.
fallback_models=(
"muse-spark-1.2",
"muse-spark-1.2-contributor",
),
fallback_models=("muse-spark-1.2", "muse-spark-1.2-contributor"),
)
register_provider(meta_ai)
+10 -28
View File
@@ -1,9 +1,8 @@
"""MiniMax provider profiles (international + China).
"""MiniMax provider profiles (international, China, OAuth).
The default API-key routes use anthropic_messages because their base URLs end
with /anthropic. Users can opt MiniMax-M3 into the OpenAI-compatible endpoint
with base_url=https://api.minimax.io/v1; that route needs MiniMax-specific
reasoning controls in extra_body.
Default routes use anthropic_messages (base URLs end in /anthropic). Users can
opt MiniMax-M3 into the OpenAI-compatible https://api.minimax.io/v1 route,
which needs MiniMax-specific reasoning controls in extra_body.
"""
from typing import Any
@@ -15,15 +14,7 @@ from providers.base import ProviderProfile
def _is_minimax_global_openai_base_url(base_url: str | None) -> bool:
parsed = urlparse(str(base_url or "").strip())
if (parsed.hostname or "").lower() != "api.minimax.io":
return False
path = parsed.path.rstrip("/").lower()
return path == "/v1"
def _is_minimax_m3(model: str | None) -> bool:
normalized = str(model or "").strip().lower()
return normalized in {"minimax-m3", "minimax/minimax-m3"}
return (parsed.hostname or "").lower() == "api.minimax.io" and parsed.path.rstrip("/").lower() == "/v1"
class MiniMaxProfile(ProviderProfile):
@@ -37,25 +28,16 @@ class MiniMaxProfile(ProviderProfile):
base_url: str | None = None,
**context: Any,
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Emit M3 reasoning controls for api.minimax.io/v1.
MiniMax-M3's OpenAI-compatible endpoint keeps thinking inline unless
``reasoning_split`` is sent, so always request the split format on that
route. ``thinking`` controls the M3 mode; Hermes' effort levels are not
a MiniMax depth knob here, so they only select adaptive vs disabled.
"""
if not _is_minimax_global_openai_base_url(base_url) or not _is_minimax_m3(model):
"""M3 on api.minimax.io/v1 keeps thinking inline unless ``reasoning_split``
is sent; effort levels only select adaptive vs disabled ``thinking``."""
is_m3 = str(model or "").strip().lower() in {"minimax-m3", "minimax/minimax-m3"}
if not _is_minimax_global_openai_base_url(base_url) or not is_m3:
return {}, {}
extra_body: dict[str, Any] = {"reasoning_split": True}
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
extra_body["thinking"] = {"type": "disabled"}
return extra_body, {}
if reasoning_config is not None:
elif reasoning_config is not None:
extra_body["thinking"] = {"type": "adaptive"}
return extra_body, {}
@@ -4,33 +4,21 @@ from __future__ import annotations
from typing import Any
from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort
from providers import register_provider
from providers.base import ProviderProfile
def _flat_model_name(model: str | None) -> str:
"""Return a lowercase model id, tolerating vendor-prefixed IDs."""
return (model or "").strip().rsplit("/", 1)[-1].lower()
def _model_supports_reasoning_effort(model: str | None) -> bool:
"""Conservative allowlist for Nebius models that expose reasoning effort."""
model_name = _flat_model_name(model)
if not model_name:
return False
return any(
marker in model_name
for marker in (
"deepseek-r1",
"deepseek-v4",
"deepseek-reasoner",
"gpt-oss",
"glm-5",
"kimi-k2",
"minimax-m2",
"qwen3",
)
)
# Conservative allowlist of model families that expose reasoning effort.
_REASONING_MARKERS = (
"deepseek-r1",
"deepseek-v4",
"deepseek-reasoner",
"gpt-oss",
"glm-5",
"kimi-k2",
"minimax-m2",
"qwen3",
)
class NebiusTokenFactoryProfile(ProviderProfile):
@@ -44,46 +32,25 @@ class NebiusTokenFactoryProfile(ProviderProfile):
supports_reasoning: bool = False,
**context: Any,
) -> tuple[dict[str, Any], dict[str, Any]]:
if not supports_reasoning and not _model_supports_reasoning_effort(model):
model_name = (model or "").strip().rsplit("/", 1)[-1].lower()
if not supports_reasoning and not any(marker in model_name for marker in _REASONING_MARKERS):
return {}, {}
if isinstance(reasoning_config, dict):
enabled = reasoning_config.get("enabled", True)
raw_effort = reasoning_config.get("effort", "medium")
else:
enabled = True
raw_effort = "medium"
effort = str(raw_effort or "medium").strip().lower()
if enabled is False or effort in {"none", "off", "disabled"}:
rc = reasoning_config if isinstance(reasoning_config, dict) else {}
# Unset/blank effort defaults to medium (reasoning ON).
effort = str(rc.get("effort", "medium") or "medium").strip().lower()
if rc.get("enabled", True) is False or effort in {"none", "off", "disabled"}:
return {}, {}
# Canonical clamp (nearest weaker supported level, never escalate,
# monotonic) — the hand-rolled map this replaces inverted the ladder:
# ultra fell through to medium while xhigh mapped to high.
from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort
effort = clamp_effort(effort, NEBIUS_EFFORTS) or "medium"
return {}, {"reasoning_effort": effort}
# Canonical clamp: nearest weaker supported level, never escalate.
return {}, {"reasoning_effort": clamp_effort(effort, NEBIUS_EFFORTS) or "medium"}
nebius_token_factory = NebiusTokenFactoryProfile(
name="nebius-token-factory",
aliases=(
"nebius",
"nebius-tokenfactory",
"nebius-tf",
"token-factory",
"tokenfactory",
),
aliases=("nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory"),
display_name="Nebius Token Factory",
description="Nebius Token Factory — OpenAI-compatible inference",
signup_url="https://tokenfactory.nebius.com/",
env_vars=(
"NEBIUS_API_KEY",
"NEBIUS_TOKEN_FACTORY_API_KEY",
"NEBIUS_BASE_URL",
),
env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY", "NEBIUS_BASE_URL"),
base_url="https://api.tokenfactory.nebius.com/v1",
models_url="https://api.tokenfactory.nebius.com/v1/models?verbose=true",
auth_type="api_key",
+22 -78
View File
@@ -16,14 +16,7 @@ class NousProfile(ProviderProfile):
"""Nous Portal — product tags, reasoning with Nous-specific omission."""
def resolve_aux_model(self, *, vision: bool = False) -> str:
"""Ask the Portal which cheap model it currently recommends.
``/api/nous/recommended-models`` is the authoritative, tier-aware
source (free vs paid), so the auxiliary fast tier tracks the live
catalog instead of a hardcoded id that 404s the day Nous retires it.
The underlying fetch is memory- and disk-cached with a last-known-good
fallback, so this is cheap to call and safe offline.
"""
"""Portal's tier-aware ``/api/nous/recommended-models`` pick (cached, offline-safe)."""
try:
from hermes_cli.models import get_nous_recommended_aux_model
@@ -31,37 +24,12 @@ class NousProfile(ProviderProfile):
except Exception:
return ""
def build_extra_body(
self, *, session_id: str | None = None, **context
) -> dict[str, Any]:
def build_extra_body(self, *, session_id: str | None = None, **context) -> dict[str, Any]:
body: dict[str, Any] = {"tags": nous_portal_tags(session_id=session_id)}
# Top-level session_id → provider sticky routing key. Pins every
# turn of a session to the same upstream endpoint so explicit
# Anthropic cache_control breakpoints stay warm instead of
# cold-writing a fresh cache on each reroute (Anthropic/Vertex/
# Bedrock caches are instance-local). Mirrors the OpenRouter
# profile; without it the portal falls back to hashing the opening
# messages, which breaks pinning whenever those shift.
#
# Resolve it exactly like ``nous_portal_tags`` resolves the
# ``conversation=`` tag: ambient context first (the lineage ROOT id
# published by the agent loop), explicit argument as fallback.
#
# The gap this closes is the auxiliary call sites — compression,
# title generation, vision, web_extract, session_search, MoA slots.
# They funnel through ``agent.auxiliary_client`` which has no session
# handle, so they never pass ``session_id``: they carried the
# ``conversation=`` tag but NO sticky key at all, and each one routed
# independently of the conversation it belongs to. Reading the same
# ambient contextvar the tag already uses fixes that with zero
# per-call-site plumbing; a host-declared routing scope (#96811) wins
# over it when one was published for this turn.
#
# For the main loop the two agree anyway under the default
# ``compression.in_place: true`` (#38763), where compaction keeps the
# session id; the ambient root additionally keeps the key stable for
# installs that opt back into rotating compaction, and across
# delegate-subagent trees.
# Top-level session_id = sticky routing key, so Anthropic-style cache
# breakpoints stay warm on one upstream instance. Resolved like the
# ``conversation=`` tag: declared scope, then the ambient lineage ROOT
# (covers aux call sites that pass no session_id), then the explicit argument.
sticky_key = _cache_scope_from_session_id(
get_affinity_scope() or get_conversation_context() or session_id
)
@@ -74,23 +42,13 @@ class NousProfile(ProviderProfile):
@staticmethod
def _cannot_disable_reasoning(model: str | None) -> bool:
"""True when a disable can't safely be sent for *model*.
"""True when ``reasoning: {enabled: false}`` would 400 on *model*.
Reasoning-mandatory routes answer ``reasoning: {enabled: false}``
with HTTP 400 ("Reasoning is mandatory for this model"), so the
catalog decides. Cache-only, and an unknown model (catalog cold,
unlisted, or unreachable) also answers True: a cold first turn errs
toward the old omit-everything behavior rather than risking a 400.
A route the catalog says takes no reasoning parameter at all is
treated the same way — sending it a disable is sending a parameter
the Portal has told us it doesn't accept.
Cache-only catalog lookup; unknown/cold (warmer kicked) and
no-reasoning-parameter routes both answer True (omit rather than risk a 400).
"""
try:
from hermes_cli.models import (
nous_model_reasoning_capabilities,
warm_nous_reasoning_caps_async,
)
from hermes_cli.models import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async
caps = nous_model_reasoning_capabilities(model)
if caps is None:
@@ -98,9 +56,7 @@ class NousProfile(ProviderProfile):
return True
except Exception:
return True
if not caps.get("supports_reasoning"):
return True
return bool(caps.get("mandatory"))
return not caps.get("supports_reasoning") or bool(caps.get("mandatory"))
def build_api_kwargs_extras(
self,
@@ -110,25 +66,16 @@ class NousProfile(ProviderProfile):
model: str | None = None,
**context,
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Nous: passes the full reasoning_config, disable included.
The Portal honors ``reasoning: {enabled: false}`` — it is the only
wire shape that does. Sending nothing means the *upstream* default,
which for a thinking-first model like ``deepseek/deepseek-v4-pro``
(catalog: ``default_effort: high``) is thinking ON, so omitting a
disable silently ignored the user's "thinking off".
"""
extra_body = {}
if supports_reasoning:
if reasoning_config is not None:
rc = dict(reasoning_config)
if rc.get("enabled") is False and self._cannot_disable_reasoning(model):
pass # route rejects a disable — let the model think
else:
extra_body["reasoning"] = rc
else:
extra_body["reasoning"] = {"enabled": True, "effort": "medium"}
return extra_body, {}
"""Pass the full reasoning_config, disable included (the Portal honors it;
omitting it means the upstream default, thinking ON for V4-class models)."""
if not supports_reasoning:
return {}, {}
if reasoning_config is None:
return {"reasoning": {"enabled": True, "effort": "medium"}}, {}
rc = dict(reasoning_config)
if rc.get("enabled") is False and self._cannot_disable_reasoning(model):
return {}, {}
return {"reasoning": rc}, {}
nous = NousProfile(
@@ -138,10 +85,7 @@ nous = NousProfile(
display_name="Nous Research",
description="Nous Research — Hermes model family",
signup_url="https://nousresearch.com/",
fallback_models=(
"hermes-3-405b",
"hermes-3-70b",
),
fallback_models=("hermes-3-405b", "hermes-3-70b"),
base_url="https://inference-api.nousresearch.com/v1",
auth_type="oauth_device_code",
)
+13 -29
View File
@@ -9,36 +9,23 @@ from providers.base import ProviderProfile
class NvidiaProviderProfile(ProviderProfile):
"""NVIDIA NIM accepts a stricter ToolMessage schema than most OpenAI-compatible APIs."""
def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
needs_sanitize = any(
@staticmethod
def _needs_strip(msg: Any) -> bool:
return (
isinstance(msg, dict)
and msg.get("role") == "tool"
and ("name" in msg or "tool_name" in msg)
for msg in messages
)
if not needs_sanitize:
return messages
# Copy-on-write: shallow outer-list copy, then a shallow dict copy
# only for the role:"tool" messages that actually need a field
# dropped. Avoids recursively deep-copying every message's content
# (including large tool outputs and attachments) for a turn that
# only ever needs to touch two top-level keys on a handful of
# messages. Matches the pattern already used by the shared
# sanitizer in agent/transports/chat_completions.py and by
# QwenProfile.prepare_messages().
sanitized = list(messages)
for idx, msg in enumerate(messages):
if (
isinstance(msg, dict)
and msg.get("role") == "tool"
and ("name" in msg or "tool_name" in msg)
):
msg_copy = dict(msg)
msg_copy.pop("name", None)
msg_copy.pop("tool_name", None)
sanitized[idx] = msg_copy
return sanitized
def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Copy-on-write: only tool messages that lose a field are copied
(no deep copy of large tool outputs); untouched input returned as-is."""
if not any(self._needs_strip(msg) for msg in messages):
return messages
return [
{k: v for k, v in msg.items() if k not in ("name", "tool_name")} if self._needs_strip(msg) else msg
for msg in messages
]
nvidia = NvidiaProviderProfile(
@@ -48,10 +35,7 @@ nvidia = NvidiaProviderProfile(
display_name="NVIDIA NIM",
description="NVIDIA NIM — accelerated inference",
signup_url="https://build.nvidia.com/",
fallback_models=(
"nvidia/llama-3.1-nemotron-70b-instruct",
"nvidia/llama-3.3-70b-instruct",
),
fallback_models=("nvidia/llama-3.1-nemotron-70b-instruct", "nvidia/llama-3.3-70b-instruct"),
base_url="https://integrate.api.nvidia.com/v1",
default_max_tokens=16384,
)
@@ -1,24 +1,15 @@
"""Ollama Cloud provider profile.
Ollama Cloud's OpenAI-compatible ``/v1/chat/completions`` endpoint
supports top-level ``reasoning_effort`` with values ``none``, ``low``,
``medium``, ``high``, and ``max`` (the last being undocumented but
empirically confirmed for DeepSeek V4 — ``max`` produces ~2.5× more
thinking tokens than ``high``).
This profile maps Hermes's ``xhigh`` → ``max`` to unlock DeepSeek V4's
"Max thinking" tier through Ollama Cloud. ``low`` / ``medium`` / ``high``
pass through unchanged.
When reasoning is explicitly disabled (``enabled: false`` or
``effort: "none"``), ``reasoning_effort`` is omitted entirely so the
model runs in non-thinking mode.
Top-level ``reasoning_effort`` on /v1/chat/completions accepts none|low|medium|
high|max (``max`` is undocumented but real — ~2.5x more thinking tokens on
DeepSeek V4); Hermes' ``xhigh`` maps to ``max``.
"""
from __future__ import annotations
from typing import Any
from agent.reasoning_effort import OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES, clamp_effort
from providers import register_provider
from providers.base import ProviderProfile
@@ -27,61 +18,23 @@ class OllamaCloudProfile(ProviderProfile):
"""Ollama Cloud — maps xhigh→max via top-level reasoning_effort."""
def build_api_kwargs_extras(
self,
*,
reasoning_config: dict | None = None,
supports_reasoning: bool = False,
**ctx: Any,
self, *, reasoning_config: dict | None = None, supports_reasoning: bool = False, **ctx: Any
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Emit top-level ``reasoning_effort`` for Ollama Cloud thinking models.
Gated on ``supports_reasoning``, which the transport resolves from the
model's native ``/api/show`` ``capabilities`` (``thinking``). Models
without the thinking capability (e.g. ``gemma3``, ``qwen3-coder``) get
no ``reasoning_effort`` at all — emitting it there is a no-op the API
ignores, and gating avoids sending a meaningless field.
"""
top_level: dict[str, Any] = {}
if not supports_reasoning:
"""Gated on ``supports_reasoning`` (resolved from the model's /api/show
``thinking`` capability) so non-thinking models get no meaningless field."""
if not supports_reasoning or not reasoning_config or not isinstance(reasoning_config, dict):
return {}, {}
if reasoning_config and isinstance(reasoning_config, dict):
enabled = reasoning_config.get("enabled", True)
if enabled is False:
# Ollama Cloud defaults to thinking ON, and ignores the
# extra_body.thinking:{type:disabled} shape (verified live).
# The ONLY way to actually suppress thinking on its
# /v1/chat/completions endpoint is top-level
# reasoning_effort:"none" — omitting the field leaves
# thinking on.
return {}, {"reasoning_effort": "none"}
effort = (reasoning_config.get("effort") or "").strip().lower()
if not effort:
# No explicit effort requested — let the model decide
# (Ollama Cloud's server default is thinking ON).
return {}, {}
if effort == "none":
return {}, {"reasoning_effort": "none"} # explicit off switch
# Accepted set {none, low, medium, high, max} is declared in
# agent.reasoning_effort ("minimal" is rejected with HTTP 400 →
# clamps to low; xhigh rounds up to max). Bespoke levels outside
# the ladder are omitted so the model applies its own default
# rather than triggering a hard 400.
from agent.reasoning_effort import (
OLLAMA_CLOUD_EFFORTS,
OLLAMA_CLOUD_OVERRIDES,
clamp_effort,
)
clamped = clamp_effort(
effort, OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES
)
if clamped in OLLAMA_CLOUD_EFFORTS:
top_level["reasoning_effort"] = clamped
return {}, top_level
# Ollama Cloud defaults to thinking ON and ignores extra_body.thinking
# (verified live); top-level reasoning_effort:"none" is the ONLY off switch.
effort = (reasoning_config.get("effort") or "").strip().lower()
if reasoning_config.get("enabled", True) is False or effort == "none":
return {}, {"reasoning_effort": "none"}
if not effort:
return {}, {} # let the server default (thinking ON) apply
# "minimal" 400s -> clamps to low; xhigh rounds up to max. Bespoke
# levels outside the ladder are omitted rather than risking a 400.
clamped = clamp_effort(effort, OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES)
return {}, {"reasoning_effort": clamped} if clamped in OLLAMA_CLOUD_EFFORTS else {}
ollama_cloud = OllamaCloudProfile(
@@ -1,12 +1,9 @@
"""OpenCode Free provider profile.
"""OpenCode Free provider profile: the free tier on the Zen relay (https://opencode.ai/zen/v1).
OpenCode's free model tier on the Zen relay (https://opencode.ai/zen/v1).
KEYLESS: the relay serves free-tier models anonymously and rejects any
Authorization bearer it doesn't recognize with 401 — so this provider
never sends a credential at all (the runtime resolver pins the keyless
placeholder and an empty Authorization header; see
hermes_cli.models.opencode_zen_free_runtime). No OpenCode account needed.
Select via ``hermes model`` or ``/model free``.
KEYLESS: the relay serves free-tier models anonymously and 401s any bearer it
doesn't recognize, so this provider never sends a credential (the runtime
resolver pins the keyless placeholder and an empty Authorization header; see
hermes_cli.models.opencode_zen_free_runtime). Select via ``/model free``.
"""
from typing import Any
@@ -15,25 +12,13 @@ from hermes_cli import __version__ as _HERMES_VERSION
from providers import register_provider
from providers.base import ProviderProfile
# Attribution headers, same values as the opencode-zen/go profiles, plus the
# empty Authorization override that keeps the SDK's "Bearer <placeholder>"
# off the wire (the free tier 401s any unrecognized bearer).
_KEYLESS_HEADERS = {
"Authorization": "",
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
}
class OpenCodeFreeProfile(ProviderProfile):
"""OpenCode Free — keyless, with Ox Alpha reasoning controls.
Ox Alpha (x-preview-f-free) is reachable through this provider as well
as opencode-zen; both share the same wire contract (reasoning_effort
accepts exactly low/high/max — anything else 400s). The translation
lives in the zen plugin; resolve it through the registered zen profile's
module so the two providers can never drift.
Ox Alpha (x-preview-f-free) is also reachable via opencode-zen with the same
wire contract; the translation lives in the zen plugin and is resolved through
the registered zen profile's module so the two providers can never drift.
"""
def build_api_kwargs_extras(
@@ -44,8 +29,7 @@ class OpenCodeFreeProfile(ProviderProfile):
from providers import get_provider_profile
zen_profile = get_provider_profile("opencode-zen")
zen_module = sys.modules[type(zen_profile).__module__]
zen_module = sys.modules[type(get_provider_profile("opencode-zen")).__module__]
return zen_module._build_ox_alpha_reasoning_extras(reasoning_config, model)
except Exception:
return {}, {}
@@ -58,9 +42,17 @@ opencode_free = OpenCodeFreeProfile(
base_url="https://opencode.ai/zen/v1",
display_name="OpenCode Free",
description="OpenCode free models — keyless, no account needed",
default_headers=dict(_KEYLESS_HEADERS),
# Attribution headers (same values as opencode-zen/go) plus the empty
# Authorization override that keeps the SDK's "Bearer <placeholder>" off the
# wire (the free tier 401s any unrecognized bearer).
default_headers={
"Authorization": "",
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
},
# laguna is the fastest non-UA-gated free model; big-pickle 429s every
# client except the opencode CLI's own User-Agent (verified 2026-08-21).
# client except the opencode CLI's own User-Agent.
default_aux_model="laguna-s-2.1-free",
)
+46 -130
View File
@@ -1,25 +1,20 @@
"""OpenCode provider profiles (Zen + Go).
Both use per-model api_mode routing:
- OpenCode Zen: Claude → anthropic_messages, GPT-5/Codex/Grok → codex_responses,
Muse Spark → codex_responses, everything else → chat_completions (this profile)
- OpenCode Go: GPT / Grok / Muse Spark → codex_responses, MiniMax/Qwen → anthropic_messages,
GLM/Kimi/DeepSeek/MiMo → chat_completions (this profile)
Both route api_mode per model in core; these profiles carry the
chat_completions reasoning translations (GLM-5.2, Kimi K2, DeepSeek, Ox Alpha).
"""
from __future__ import annotations
from typing import Any
from agent import reasoning_effort as re_
from hermes_cli import __version__ as _HERMES_VERSION
from providers import register_provider
from providers.base import ProviderProfile
# Attribution headers sent on every OpenCode request. Same values we send
# to OpenRouter, Vercel AI Gateway, and Fireworks. Going through
# profile.default_headers means they survive model switches and credential
# rotation. Without them OpenCode only sees the OpenAI SDK's generic
# "OpenAI/Python x.y.z" User-Agent and can't tell the traffic is Hermes Agent.
# Attribution headers (same values as OpenRouter / Vercel / Fireworks); via
# default_headers so they survive model switches and credential rotation.
_ATTRIBUTION_HEADERS = {
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
@@ -32,15 +27,9 @@ def _flat_model_name(model: str | None) -> str:
return (model or "").strip().rsplit("/", 1)[-1].lower()
def _is_kimi_k2_model(model: str | None) -> bool:
return _flat_model_name(model).startswith("kimi-k2")
def _is_deepseek_thinking_model(model: str | None) -> bool:
m = _flat_model_name(model)
if m.startswith("deepseek-v") and not m.startswith("deepseek-v3"):
return True
return m == "deepseek-reasoner"
return (m.startswith("deepseek-v") and not m.startswith("deepseek-v3")) or m == "deepseek-reasoner"
def _is_glm_5_2_model(model: str | None) -> bool:
@@ -49,143 +38,70 @@ def _is_glm_5_2_model(model: str | None) -> bool:
return any(token in m for token in ("glm-5.2", "glm-5-2", "glm-5p2"))
def _requested_effort(reasoning_config: dict | None) -> str | None:
"""Normalized effort when reasoning is enabled and an effort is set, else None."""
effort = re_.requested_effort(reasoning_config)
return None if effort == "none" else effort
def _thinking_toggle_extras(
reasoning_config: dict | None, efforts, overrides=None
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Moonshot/DeepSeek wire shape: extra_body.thinking XOR top-level reasoning_effort
(sending both is an HTTP 400)."""
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
return {"thinking": {"type": "disabled"}}, {}
clamped = re_.clamp_effort(_requested_effort(reasoning_config), efforts, overrides)
if clamped in efforts:
return {}, {"reasoning_effort": clamped}
return {"thinking": {"type": "enabled"}}, {}
class OpenCodeGoProfile(ProviderProfile):
"""OpenCode Go - model-specific reasoning controls."""
# Per-model completion-token cap. The opencode-go relay's default is
# too large for mimo-v2.5-pro — it sends max_tokens=262144 but Xiaomi
# only supports 131072 completion tokens and 400s the request.
# Setting an explicit cap here prevents the relay default from being
# applied. Keys are normalized via _flat_model_name().
# The relay's default max_tokens (262144) exceeds what Xiaomi accepts for
# mimo-v2.5-pro and 400s; keys are normalized via _flat_model_name().
_MODEL_MAX_TOKENS: dict[str, int] = {
"mimo-v2.5-pro": 131072,
}
def get_max_tokens(self, model: str | None) -> int | None:
cap = self._MODEL_MAX_TOKENS.get(_flat_model_name(model))
if cap is not None:
return cap
return self.default_max_tokens
return self.default_max_tokens if cap is None else cap
def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
top_level: dict[str, Any] = {}
if _is_glm_5_2_model(model):
# GLM-5.2 on OpenCode Go uses its native OpenAI-compatible
# reasoning_effort knob (high/max — declared in
# agent.reasoning_effort, shared with the zai profile); leave the
# server default alone when reasoning is disabled or unset.
# Native reasoning_effort knob (high/max); server default when unset/disabled.
effort = _requested_effort(reasoning_config)
if effort is None:
return {}, {}
clamped = re_.clamp_effort(effort, re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES)
return {}, {"reasoning_effort": clamped if clamped in re_.GLM52_EFFORTS else "high"}
if _flat_model_name(model).startswith("kimi-k2"):
if not isinstance(reasoning_config, dict):
return extra_body, top_level
if reasoning_config.get("enabled") is False:
return extra_body, top_level
effort = (reasoning_config.get("effort") or "").strip().lower()
if not effort or effort == "none":
return extra_body, top_level
from agent.reasoning_effort import (
GLM52_EFFORTS,
GLM52_OVERRIDES,
clamp_effort,
return {}, {}
return _thinking_toggle_extras(reasoning_config, re_.KIMI_K2_EFFORTS)
if _is_deepseek_thinking_model(model):
return _thinking_toggle_extras(
reasoning_config, re_.DEEPSEEK_V4_EFFORTS, re_.DEEPSEEK_V4_OVERRIDES
)
clamped = clamp_effort(effort, GLM52_EFFORTS, GLM52_OVERRIDES)
top_level["reasoning_effort"] = (
clamped if clamped in GLM52_EFFORTS else "high"
)
return extra_body, top_level
if _is_kimi_k2_model(model):
# Kimi K2 on OpenCode Go uses Moonshot's native wire shape:
# extra_body.thinking (binary toggle) + top-level reasoning_effort
# (low|medium|high). Mirrors the KimiProfile (api.moonshot.ai/v1).
if not isinstance(reasoning_config, dict):
# No config → leave server defaults alone.
return extra_body, top_level
enabled = reasoning_config.get("enabled") is not False
if not enabled:
extra_body["thinking"] = {"type": "disabled"}
return extra_body, top_level
effort = (reasoning_config.get("effort") or "").strip().lower()
if effort and effort != "none":
from agent.reasoning_effort import KIMI_K2_EFFORTS, clamp_effort
clamped = clamp_effort(effort, KIMI_K2_EFFORTS)
if clamped in KIMI_K2_EFFORTS:
top_level["reasoning_effort"] = clamped
# Avoid "cannot specify both 'thinking' and 'reasoning_effort'" HTTP 400:
# only send extra_body["thinking"] when no reasoning_effort is set.
if "reasoning_effort" not in top_level:
extra_body["thinking"] = {"type": "enabled"}
return extra_body, top_level
if not _is_deepseek_thinking_model(model):
return extra_body, top_level
enabled = True
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
enabled = False
if not enabled:
extra_body["thinking"] = {"type": "disabled"}
return extra_body, top_level
if isinstance(reasoning_config, dict):
effort = (reasoning_config.get("effort") or "").strip().lower()
if effort and effort != "none":
from agent.reasoning_effort import (
DEEPSEEK_V4_EFFORTS,
DEEPSEEK_V4_OVERRIDES,
clamp_effort,
)
clamped = clamp_effort(
effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES
)
if clamped in DEEPSEEK_V4_EFFORTS:
top_level["reasoning_effort"] = clamped
# Avoid "cannot specify both 'thinking' and 'reasoning_effort'" HTTP 400:
# only send extra_body["thinking"] when no reasoning_effort is set.
if "reasoning_effort" not in top_level:
extra_body["thinking"] = {"type": "enabled"}
return extra_body, top_level
return {}, {}
def _build_ox_alpha_reasoning_extras(
reasoning_config: dict | None, model: str | None
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Shared Ox Alpha (x-preview-f-free) reasoning_effort translation.
Used by both the opencode-zen profile and the opencode-free keyless
profile — the model is reachable through either provider and the wire
contract is identical (low/high/max only; anything else 400s).
"""
"""Ox Alpha (x-preview-f-free) reasoning_effort translation, shared with the
opencode-free profile (low/high/max only; anything else 400s)."""
if _flat_model_name(model) != "x-preview-f-free":
return {}, {}
if not isinstance(reasoning_config, dict):
return {}, {}
if reasoning_config.get("enabled") is False:
return {}, {}
effort = (reasoning_config.get("effort") or "").strip().lower()
if not effort or effort == "none":
return {}, {}
from agent.reasoning_effort import (
OX_ALPHA_EFFORTS,
OX_ALPHA_OVERRIDES,
clamp_effort,
clamped = re_.clamp_effort(
_requested_effort(reasoning_config), re_.OX_ALPHA_EFFORTS, re_.OX_ALPHA_OVERRIDES
)
clamped = clamp_effort(effort, OX_ALPHA_EFFORTS, OX_ALPHA_OVERRIDES)
if clamped not in OX_ALPHA_EFFORTS:
if clamped not in re_.OX_ALPHA_EFFORTS:
return {}, {}
return {}, {"reasoning_effort": clamped}
+52 -137
View File
@@ -12,14 +12,10 @@ logger = logging.getLogger(__name__)
_CACHE: list[str] | None = None
# Anthropic model families that still accept an explicit "disable thinking"
# request (the manual ``thinking: {type: "disabled"}`` form OpenRouter emits
# for ``reasoning: {enabled: false}``). Everything Claude 4.6 and newer —
# including future date-stamped / named models (fable, mythos-class, …) —
# mandates reasoning and returns HTTP 400 on any disable form. We therefore
# default *unknown* Anthropic models to "cannot disable" (the modern contract)
# and keep only this explicit legacy allowlist of models that can. Mirrors the
# default-to-newest philosophy in agent/anthropic_adapter._get_anthropic_max_output.
# Legacy allowlist of Anthropic models that still accept an explicit "disable
# thinking" request. Claude 4.6+ and newer named models mandate reasoning and
# 400 on any disable form, so *unknown* Anthropic models default to "cannot
# disable" (mirrors agent/anthropic_adapter._get_anthropic_max_output).
_ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS = (
"claude-3", # 3, 3.5, 3.7
"claude-opus-4-0", "claude-opus-4.0", "claude-opus-4-1", "claude-opus-4.1",
@@ -32,48 +28,44 @@ _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS = (
def _anthropic_reasoning_is_mandatory(model: str | None) -> bool:
"""Return True for Anthropic models that reject any disable-thinking form.
Claude 4.6+ (adaptive thinking) and newer named models have no "off"
switch — sending ``reasoning: {enabled: false}`` makes OpenRouter emit
``thinking: {type: "disabled"}``, which these models 400 on. Unknown /
new Anthropic model names default to mandatory so the next un-numbered
release doesn't reintroduce the 400.
"""
"""True for Anthropic models that reject any disable-thinking form (unknown -> True)."""
m = (model or "").lower()
if not m.startswith(("anthropic/", "claude")) and "claude" not in m:
return False
return not any(sub in m for sub in _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS)
def _sticky_key(session_id: str | None) -> str | None:
"""Declared routing scope, then ambient conversation, then explicit session_id.
Aux call sites (compression, titles, vision, MoA…) pass no ``session_id``,
so the ambient lineage ROOT keeps them pinned to their conversation.
"""
return _cache_scope_from_session_id(
get_affinity_scope() or get_conversation_context() or session_id
)
class OpenRouterProfile(ProviderProfile):
"""OpenRouter aggregator — provider preferences, reasoning config passthrough."""
@staticmethod
def _clamp_reasoning_to_catalog(cfg: dict[str, Any], model: str | None) -> dict[str, Any]:
"""Clamp ``cfg["effort"]`` to the model's catalog-advertised levels.
"""Clamp ``cfg["effort"]`` to the nearest LOWER catalog-advertised level.
OpenRouter's /v1/models entries publish ``reasoning.supported_efforts``
per model (ported from PrimeIntellect-ai/prime-agent#1258). Sending an
unsupported effort (e.g. ``ultra`` to a route that stops at ``high``)
yields provider 4xx errors; clamp to the nearest LOWER supported level
instead. No-op when the catalog is unreachable, the model is unlisted,
or no supported_efforts list is published (None = all levels accepted).
No-op when the catalog is unreachable, the model is unlisted, or no
supported_efforts list is published (None = all levels accepted).
"""
effort = cfg.get("effort")
if not effort or cfg.get("enabled") is False:
return cfg
try:
from hermes_cli.models import (
clamp_reasoning_effort_to_supported,
openrouter_model_reasoning_capabilities,
)
from hermes_cli.models import clamp_reasoning_effort_to_supported, openrouter_model_reasoning_capabilities
caps = openrouter_model_reasoning_capabilities(model)
if not caps or not caps.get("supports_reasoning"):
return cfg
clamped = clamp_reasoning_effort_to_supported(
effort, caps.get("supported_efforts")
)
clamped = clamp_reasoning_effort_to_supported(effort, caps.get("supported_efforts"))
except Exception:
return cfg
if clamped and clamped != effort:
@@ -82,83 +74,46 @@ class OpenRouterProfile(ProviderProfile):
"(catalog supported_efforts=%s)",
effort, clamped, model, caps.get("supported_efforts"),
)
cfg = dict(cfg)
cfg["effort"] = clamped
cfg = {**cfg, "effort": clamped}
return cfg
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Fetch from public OpenRouter catalog — no auth required.
Note: Tool-call capability filtering is applied by hermes_cli/models.py
via fetch_openrouter_models() → _openrouter_model_supports_tools(), not
here. The picker early-returns via the dedicated openrouter path before
reaching this method, so filtering here would be unreachable.
"""
"""Fetch from the public OpenRouter catalog (no auth). Tool-call filtering
happens in hermes_cli/models.py, which the picker reaches first."""
global _CACHE # noqa: PLW0603
if _CACHE is not None:
return _CACHE
try:
result = super().fetch_models(api_key=None, base_url=base_url, timeout=timeout)
if result is not None:
_CACHE = result
return result
except Exception as exc:
logger.debug("fetch_models(openrouter): %s", exc)
return None
if result is not None:
_CACHE = result
return result
def build_extra_body(
self, *, session_id: str | None = None, **context: Any
) -> dict[str, Any]:
def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]:
body: dict[str, Any] = {}
# Top-level session_id → OpenRouter's sticky routing key. Per their
# prompt-caching docs it is used directly as the routing key instead of
# hashing the opening messages, and it activates stickiness on the
# first successful request rather than only after a cache hit.
#
# Resolve it from the declared routing scope first (set only by a host
# that names its own conversation, #96811), then the ambient conversation
# contextvar, with the explicit argument as fallback. The gap this closes is the auxiliary call sites
# — compression, title generation, vision, web_extract, session_search,
# MoA slots — which funnel through ``agent.auxiliary_client``. That
# module has no session handle and passes no ``session_id``, so those
# calls sent NO sticky key at all and each routed independently of the
# conversation it belonged to (#70820).
#
# Mirrors the Nous Portal profile, which resolves the same way
# (f2f4df064d). The ambient value is the session-lineage ROOT, so it
# also stays stable for installs that opt out of the default
# ``compression.in_place: true`` and across delegate-subagent trees.
sticky_key = _cache_scope_from_session_id(
get_affinity_scope() or get_conversation_context() or session_id
)
# Top-level session_id is OpenRouter's sticky routing key (used directly,
# not hashed from the opening messages; active from the first request).
sticky_key = _sticky_key(session_id)
if sticky_key:
body["session_id"] = sticky_key
prefs = context.get("provider_preferences")
if prefs:
body["provider"] = prefs
# Pareto Code router — model-gated. The plugins block is only
# meaningful for openrouter/pareto-code; sending it on any other
# model has no documented effect and would be confusing in logs.
# See: https://openrouter.ai/docs/guides/routing/routers/pareto-router
model = (context.get("model") or "")
if model == "openrouter/pareto-code":
score = context.get("openrouter_min_coding_score")
if score is not None and score != "":
try:
score_f = float(score)
except (TypeError, ValueError):
score_f = None
if score_f is not None and 0.0 <= score_f <= 1.0:
body["plugins"] = [
{"id": "pareto-router", "min_coding_score": score_f}
]
# Pareto Code router plugin is only meaningful for openrouter/pareto-code.
score = context.get("openrouter_min_coding_score")
if (context.get("model") or "") == "openrouter/pareto-code" and score is not None and score != "":
try:
score_f = float(score)
except (TypeError, ValueError):
score_f = None
if score_f is not None and 0.0 <= score_f <= 1.0:
body["plugins"] = [{"id": "pareto-router", "min_coding_score": score_f}]
return body
def build_api_kwargs_extras(
@@ -170,69 +125,29 @@ class OpenRouterProfile(ProviderProfile):
session_id: str | None = None,
**context: Any,
) -> tuple[dict[str, Any], dict[str, Any]]:
"""OpenRouter passes the full reasoning_config dict as extra_body.reasoning.
For xAI Grok models routed through OpenRouter, attach the
``x-grok-conv-id`` header so that xAI's prompt cache stays pinned to
the same backend server across turns.
"""
"""Pass reasoning_config as extra_body.reasoning; pin Grok's cache via x-grok-conv-id."""
extra_body: dict[str, Any] = {}
top_level: dict[str, Any] = {}
extra_headers: dict[str, Any] = {}
if supports_reasoning:
# Reasoning-mandatory Anthropic models (Claude 4.6+ / fable /
# future named models) use *adaptive* thinking: the model decides
# how much to think, and OpenRouter ignores ``reasoning.effort`` for
# them entirely. Sending any ``reasoning`` field is therefore both
# pointless and actively harmful:
# - ``{enabled: false}`` → OpenRouter emits Anthropic's manual
# ``thinking: {type: "disabled"}``, which these models 400 on.
# - any enabled form, on a tool-continuation turn whose prior
# assistant tool_call carries no thinking block (chat_completions
# never replays signed thinking blocks), ALSO makes OpenRouter
# emit ``thinking: {type: "disabled"}`` → the same 400 on every
# turn after the first tool call.
# The only reliable behavior is to omit ``reasoning`` and let the
# model default to adaptive. See hermes-agent#42991 (disable case)
# and the tool-replay follow-up.
#
# ``reasoning.effort`` being ignored does NOT mean these models have
# no effort lever — OpenRouter honors the requested effort on the
# top-level ``verbosity`` field instead (it maps to Anthropic's
# ``output_config.effort``; ``reasoning.effort`` is accepted but
# ignored — confirmed by OpenRouter's Claude migration docs and a
# live token-spend probe in hermes-agent#43432). Route the existing
# ``reasoning_config["effort"]`` (sourced from
# ``agent.reasoning_effort``) onto ``verbosity`` so the knob the user
# already sets keeps working for these models. We still send NO
# ``reasoning`` field, preserving the #42991 400 fix.
# Reasoning-mandatory Anthropic models use adaptive thinking: any
# ``reasoning`` field (disable, or an enabled form on a tool-continuation
# turn without a replayed thinking block) makes OpenRouter emit
# ``thinking: {type: "disabled"}`` -> 400. Omit it; the user's effort
# still reaches Anthropic's output_config.effort via top-level ``verbosity``.
if _anthropic_reasoning_is_mandatory(model):
cfg = reasoning_config or {}
effort = cfg.get("effort")
# Only emit when effort is actually requested and reasoning
# isn't explicitly disabled. Otherwise omit ``verbosity`` so the
# model keeps its own adaptive default (``high``).
if cfg.get("enabled", True) is not False and effort and effort != "none":
top_level["verbosity"] = effort
elif reasoning_config is not None:
extra_body["reasoning"] = self._clamp_reasoning_to_catalog(
dict(reasoning_config), model
)
extra_body["reasoning"] = self._clamp_reasoning_to_catalog(dict(reasoning_config), model)
else:
extra_body["reasoning"] = {"enabled": True, "effort": "medium"}
# Same resolution as build_extra_body: xAI's prompt cache is pinned per
# backend server via this header, and aux calls pass no session_id, so
# reading the ambient conversation keeps compression/vision/MoA traffic
# on the same Grok backend as the conversation it belongs to.
grok_conv_id = _cache_scope_from_session_id(
get_affinity_scope() or get_conversation_context() or session_id
)
# xAI's prompt cache is pinned per backend server via this header.
grok_conv_id = _sticky_key(session_id)
if grok_conv_id and model and model.startswith(("x-ai/grok-", "xai/grok-")):
extra_headers["x-grok-conv-id"] = grok_conv_id
if extra_headers:
top_level["extra_headers"] = extra_headers
top_level["extra_headers"] = {"x-grok-conv-id": grok_conv_id}
return extra_body, top_level
+31 -58
View File
@@ -5,31 +5,34 @@ from providers import register_provider
from providers.base import ProviderProfile
def _normalize_parts(content: list) -> list | None:
"""List content -> list-of-dict parts (str -> text part, image_url dicts copied,
other junk dropped). None when nothing changed (copy-on-write)."""
parts, changed = [], False
for part in content:
if isinstance(part, str):
parts.append({"type": "text", "text": part})
changed = True
elif isinstance(part, dict):
if isinstance(part.get("image_url"), dict):
part = {**part, "image_url": dict(part["image_url"])}
changed = True
parts.append(part)
else:
changed = True
return parts if parts and changed else None
class QwenProfile(ProviderProfile):
"""Qwen Portal — message normalization, vl_high_resolution, metadata top-level."""
@staticmethod
def _copy_part_if_request_mutable(part: dict[str, Any]) -> tuple[dict[str, Any], bool]:
image_url = part.get("image_url")
if isinstance(image_url, dict):
copied = dict(part)
copied["image_url"] = dict(image_url)
return copied, True
return part, False
def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Normalize content to list-of-dicts format.
Inject cache_control on system message.
Matches the behavior of run_agent.py:_qwen_prepare_chat_messages().
"""
"""Normalize content to list-of-dicts and inject cache_control on the system
message. Copy-on-write: only touched messages/parts are copied."""
if not messages:
return []
prepared = list(messages)
system_idx: int | None = None
for idx, msg in enumerate(messages):
if not isinstance(msg, dict):
continue
@@ -37,49 +40,22 @@ class QwenProfile(ProviderProfile):
system_idx = idx
content = msg.get("content")
if isinstance(content, str):
msg_copy = dict(msg)
msg_copy["content"] = [{"type": "text", "text": content}]
prepared[idx] = msg_copy
prepared[idx] = {**msg, "content": [{"type": "text", "text": content}]}
elif isinstance(content, list):
normalized_parts = []
changed = False
for part in content:
if isinstance(part, str):
normalized_parts.append({"type": "text", "text": part})
changed = True
elif isinstance(part, dict):
normalized_part, copied = self._copy_part_if_request_mutable(part)
normalized_parts.append(normalized_part)
changed = changed or copied
else:
changed = True
if normalized_parts and changed:
msg_copy = dict(msg)
msg_copy["content"] = normalized_parts
prepared[idx] = msg_copy
parts = _normalize_parts(content)
if parts is not None:
prepared[idx] = {**msg, "content": parts}
# Inject cache_control on the last part of the system message.
if system_idx is not None:
msg = prepared[system_idx]
if isinstance(msg, dict):
content = msg.get("content")
if (
isinstance(content, list)
and content
and isinstance(content[-1], dict)
):
msg_copy = dict(msg)
content_copy = list(content)
content_copy[-1] = dict(content_copy[-1])
content_copy[-1]["cache_control"] = {"type": "ephemeral"}
msg_copy["content"] = content_copy
prepared[system_idx] = msg_copy
content = msg.get("content")
if isinstance(content, list) and content and isinstance(content[-1], dict):
content_copy = list(content)
content_copy[-1] = {**content_copy[-1], "cache_control": {"type": "ephemeral"}}
prepared[system_idx] = {**msg, "content": content_copy}
return prepared
def build_extra_body(
self, *, session_id: str | None = None, **context
) -> dict[str, Any]:
def build_extra_body(self, *, session_id: str | None = None, **context) -> dict[str, Any]:
return {"vl_high_resolution_images": True}
def build_api_kwargs_extras(
@@ -90,10 +66,7 @@ class QwenProfile(ProviderProfile):
**context,
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Qwen metadata goes to top-level api_kwargs, not extra_body."""
top_level = {}
if qwen_session_metadata:
top_level["metadata"] = qwen_session_metadata
return {}, top_level
return {}, {"metadata": qwen_session_metadata} if qwen_session_metadata else {}
qwen = QwenProfile(
+77 -186
View File
@@ -1,45 +1,18 @@
"""Ramp Router (router.com) provider plugin for Hermes Agent.
"""Ramp Router (router.com) provider profile: Responses-only LLM gateway.
Provider profile for `Ramp Router <https://docs.router.com>`_, Ramp's LLM
gateway: one OpenAI Responses-compatible endpoint at
``https://api.router.com/v1`` that routes each request across upstream
providers (OpenAI, Anthropic, xAI, Fireworks, ...) and handles fallbacks and
spend controls server-side.
Wire notes (verified live against api.router.com, Aug 2026):
* **Responses API is the native wire.** Router serves ``GET /v1/models``
and ``POST /v1/responses``; ``POST /v1/chat/completions`` is only a
minimal compatibility shim (added Aug 2026) that translates onto
Responses. Per-model reasoning-effort validation, reasoning summaries,
and prompt caching are Responses-surface features, so
``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate
in ``hermes_cli/providers.py`` keep every path on the native wire —
the same shape as the ``api.openai.com`` mandate.
* **Account-scoped catalog.** Valid model IDs are whatever the key's
``GET /v1/models`` returns (BYOK accounts see extra entries), so this
profile ships **no** ``fallback_models`` — the picker relies on the live
fetch, per Router's own guidance to never hardcode model names.
* **Strict reasoning-effort validation.** Router validates
``reasoning.effort`` against each model's catalog-declared vocabulary and
returns HTTP 400 ``invalid-argument`` on a level the model does not accept
(e.g. ``max`` on grok-4.6), and 400 ``unsupported_parameter`` when a
non-reasoning model (gpt-4.1 family, gpt-4o, ...) receives any reasoning
field. The catalog publishes the vocabulary per model
(``router.capabilities.reasoning``), so ``supported_reasoning_efforts``
below feeds the codex transport's clamp from a cached copy of it.
* **Everything else passes through.** ``store: false``, ``prompt_cache_key``,
``include: ["reasoning.encrypted_content"]``, and ``reasoning.summary`` are
accepted on all models (ignored where a backend cannot honor them), tools /
``parallel_tool_calls`` / streaming SSE work across backends, and encrypted
reasoning replay round-trips on OpenAI-served models — so the generic
Responses transport path needs no Router-specific request surgery.
The capability cache mirrors the OpenRouter reasoning-caps design in
``hermes_cli/models.py``: cache-only lookups on the per-request hot path
(never HTTP), seeded for free whenever ``fetch_models()`` runs (picker,
setup, doctor), hydrated from a disk mirror across processes, and refreshed
by a background warmer when cold or stale.
Wire notes (verified live against api.router.com):
* Responses API is the native wire; ``/chat/completions`` is only a thin shim.
``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate in
``hermes_cli/providers.py`` keep every path on it.
* The catalog is account-scoped (BYOK accounts see extra IDs), so this profile
ships no ``fallback_models`` — the picker relies on ``fetch_models()``.
* Router 400s on ``reasoning.effort`` levels outside a model's published
vocabulary and on any reasoning field for non-reasoning models. The efforts
map from ``GET /v1/models`` is cached (memory + disk mirror, background
warmer; never HTTP on the request hot path) and fed to the codex transport's
clamp via ``supported_reasoning_efforts``.
* ``store: false``, ``prompt_cache_key``, encrypted reasoning replay, tools and
streaming pass through unchanged — no Router-specific request surgery.
"""
from __future__ import annotations
@@ -52,6 +25,7 @@ import time
from pathlib import Path
from typing import Any, Optional
from agent.reasoning_effort import EFFORT_LADDER
from hermes_cli import __version__ as _HERMES_VERSION
from providers import register_provider
from providers.base import ProviderProfile, _profile_user_agent
@@ -60,43 +34,32 @@ logger = logging.getLogger(__name__)
ROUTER_DEFAULT_BASE_URL = "https://api.router.com/v1"
#: Efforts-by-model cache: ``model id -> list of accepted effort levels``.
#: ``[]`` means the catalog says the model accepts NO reasoning parameters
#: (``reasoning.supported: false``) — the transport must omit reasoning
#: entirely. A model absent from the dict is unknown (custom/BYOK route or
#: vocabulary not published) and callers fall back to their defaults.
#: model id -> accepted effort levels. ``[]`` = model accepts NO reasoning
#: fields; absent = unknown (callers keep their defaults).
_efforts_cache: Optional[dict[str, list[str]]] = None
_efforts_lock = threading.Lock()
_warm_started = False
_disk_checked = False
#: Disk-mirror staleness bound. Vocabularies change rarely; a stale verdict
#: beats no verdict, so a past-TTL mirror is still served while a background
#: refresh runs (same policy as the OpenRouter caps mirror).
# A stale verdict beats no verdict: a past-TTL mirror is still served while a
# background refresh runs.
_DISK_TTL_SECONDS = 24 * 60 * 60
def _base_url() -> str:
"""Allow a base-URL override via ``RAMP_ROUTER_BASE_URL``."""
return os.getenv("RAMP_ROUTER_BASE_URL", "").strip().rstrip("/") or ROUTER_DEFAULT_BASE_URL
def _resolve_api_key() -> str:
"""Resolve the Router key from .env / environment, preferring dotenv.
``RAMP_ROUTER_API_KEY`` is Router's documented variable;
``ROUTER_API_KEY`` is accepted as a convenience alias. Falls back to the
raw environment when the hermes_cli helper is unavailable (e.g. stripped
test environments).
"""
resolvers = []
"""Resolve the Router key (documented var, then alias), preferring dotenv;
plain os.environ is the fallback when the dotenv resolver is unavailable or raises."""
resolvers: list = [lambda var: os.environ.get(var, "")]
try:
from hermes_cli.config import get_env_value_prefer_dotenv
resolvers.append(get_env_value_prefer_dotenv)
resolvers.insert(0, get_env_value_prefer_dotenv)
except Exception:
pass
resolvers.append(lambda var: os.environ.get(var, ""))
for resolve in resolvers:
for var in ("RAMP_ROUTER_API_KEY", "ROUTER_API_KEY"):
try:
@@ -108,63 +71,45 @@ def _resolve_api_key() -> str:
return ""
def _parse_efforts(items: Any) -> Optional[dict[str, list[str]]]:
"""Parse a Router ``/v1/models`` ``data`` array into the efforts map.
def _dig(obj: Any, *keys: str) -> Any:
"""Nested dict lookup; None as soon as a level is missing or not a dict."""
for key in keys:
obj = obj.get(key) if isinstance(obj, dict) else None
return obj
Returns None when the array has no usable entries, which callers treat
as a failed fetch rather than caching an empty verdict.
def _parse_efforts(items: Any) -> Optional[dict[str, list[str]]]:
"""Parse a ``/v1/models`` ``data`` array into the efforts map (None if unusable).
Ladder-unknown levels are dropped: clamp_effort ignores them, so an
all-unknown vocabulary would pass the effort through unclamped to a Router
400. ``supported=True`` with no recognized level leaves the model out
(unknown) so the transport keeps its default clamp behavior.
"""
if not isinstance(items, list):
return None
try:
from agent.reasoning_effort import EFFORT_LADDER
known_levels = set(EFFORT_LADDER)
except Exception:
known_levels = None
efforts_by_id: dict[str, list[str]] = {}
for item in items:
if not isinstance(item, dict):
continue
mid = str(item.get("id") or "").strip()
if not mid:
continue
router_meta = item.get("router")
reasoning = None
if isinstance(router_meta, dict):
capabilities = router_meta.get("capabilities")
if isinstance(capabilities, dict):
reasoning = capabilities.get("reasoning")
if not isinstance(reasoning, dict):
mid = str(item.get("id") or "").strip() if isinstance(item, dict) else ""
reasoning = _dig(item, "router", "capabilities", "reasoning")
if not mid or not isinstance(reasoning, dict):
continue
if reasoning.get("supported") is False:
# Definitive negative: any reasoning field 400s on this model.
efforts_by_id[mid] = []
continue
levels = [
str(entry.get("value") or "").strip()
for entry in reasoning.get("efforts") or []
if isinstance(entry, dict) and str(entry.get("value") or "").strip()
]
if known_levels is not None:
# clamp_effort silently ignores ladder-unknown levels, and an
# all-unknown vocabulary would pass the requested effort through
# unclamped straight to a Router 400 — so a new vendor tier is
# dropped at ingest and fails loudly here instead.
unknown = [level for level in levels if level not in known_levels]
if unknown:
logger.info(
"router: model %s publishes unrecognized reasoning effort "
"level(s) %s; ignoring them (update agent/reasoning_effort "
"EFFORT_LADDER to adopt new vendor tiers)",
mid,
unknown,
)
levels = [level for level in levels if level in known_levels]
values = [str(e.get("value") or "").strip() for e in reasoning.get("efforts") or [] if isinstance(e, dict)]
levels = [v for v in values if v]
unknown = [level for level in levels if level not in EFFORT_LADDER]
if unknown:
logger.info(
"router: model %s publishes unrecognized reasoning effort "
"level(s) %s; ignoring them (update agent/reasoning_effort "
"EFFORT_LADDER to adopt new vendor tiers)",
mid, unknown,
)
levels = [level for level in levels if level in EFFORT_LADDER]
if levels:
efforts_by_id[mid] = levels
# supported=True with no (recognized) vocabulary -> leave the model
# out (unknown), so the transport keeps its default clamp behavior.
return efforts_by_id or None
@@ -184,16 +129,14 @@ def _save_disk(efforts_by_id: dict[str, list[str]]) -> None:
try:
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".tmp")
tmp.write_text(
json.dumps({"ts": time.time(), "efforts": efforts_by_id}),
encoding="utf-8",
)
tmp.write_text(json.dumps({"ts": time.time(), "efforts": efforts_by_id}), encoding="utf-8")
tmp.replace(path)
except Exception as exc:
logger.debug("router: caps disk mirror write failed: %s", exc)
def _load_disk() -> tuple[Optional[dict[str, list[str]]], float]:
"""Disk mirror -> (efforts map or None, age in seconds; TTL when ``ts`` is unparseable)."""
path = _disk_path()
if path is None:
return None, 0.0
@@ -202,11 +145,7 @@ def _load_disk() -> tuple[Optional[dict[str, list[str]]], float]:
efforts = data.get("efforts")
if not isinstance(efforts, dict) or not efforts:
return None, 0.0
parsed = {
str(mid): [str(level) for level in levels]
for mid, levels in efforts.items()
if isinstance(levels, list)
}
parsed = {str(mid): [str(lv) for lv in levels] for mid, levels in efforts.items() if isinstance(levels, list)}
try:
age = max(0.0, time.time() - float(data.get("ts") or 0))
except (TypeError, ValueError):
@@ -220,11 +159,10 @@ def _seed_efforts(items: Any) -> Optional[dict[str, list[str]]]:
"""Seed memory + disk caches from a ``/v1/models`` payload."""
global _efforts_cache
parsed = _parse_efforts(items)
if parsed is None:
return None
with _efforts_lock:
_efforts_cache = parsed
_save_disk(parsed)
if parsed is not None:
with _efforts_lock:
_efforts_cache = parsed
_save_disk(parsed)
return parsed
@@ -232,17 +170,16 @@ def _fetch_catalog_items(
*, api_key: str = "", base_url: str = "", timeout: float = 8.0
) -> Optional[list]:
"""Fetch the raw ``/v1/models`` ``data`` array. None on any failure."""
url = (base_url or _base_url()).rstrip("/") + "/models"
import urllib.request
from hermes_cli.urllib_security import open_credentialed_url
req = urllib.request.Request(url)
req = urllib.request.Request((base_url or _base_url()).rstrip("/") + "/models")
key = api_key or _resolve_api_key()
if key:
req.add_header("Authorization", f"Bearer {key}")
req.add_header("Accept", "application/json")
# Router sits behind a WAF that rejects the default Python-urllib UA.
# Router's WAF rejects the default Python-urllib UA.
req.add_header("User-Agent", _profile_user_agent())
try:
with open_credentialed_url(req, timeout=timeout) as resp:
@@ -255,14 +192,12 @@ def _fetch_catalog_items(
def _efforts_cache_only() -> Optional[dict[str, list[str]]]:
"""Memory, else the disk mirror. Never HTTP (hot-path safe)."""
"""Memory, else the disk mirror (checked once per process). Never HTTP (hot-path safe)."""
global _efforts_cache, _disk_checked
with _efforts_lock:
cached = _efforts_cache
if cached is not None:
if cached is not None or _disk_checked:
return cached
if _disk_checked:
return None
_disk_checked = True
parsed, age = _load_disk()
if parsed is None:
@@ -277,19 +212,19 @@ def _efforts_cache_only() -> Optional[dict[str, list[str]]]:
def _warm_efforts_async() -> None:
"""Refresh the efforts cache in the background, at most once per process."""
"""Refresh the efforts cache in the background, at most once per process.
Skipped under pytest (a mid-suite fetch makes cache state timing-dependent)
and without a key (it would 401; the first authenticated fetch_models() seeds).
"""
global _warm_started
if os.environ.get("PYTEST_CURRENT_TEST"):
# Match the canonical caps warmer (hermes_cli/models.py): a mid-suite
# background fetch would make cache state timing-dependent in tests.
return
with _efforts_lock:
if _warm_started:
return
_warm_started = True
if not _resolve_api_key():
# Without a key the fetch would 401; the first authenticated
# fetch_models() (picker/setup/doctor) seeds the cache instead.
return
def _refresh() -> None:
@@ -298,9 +233,7 @@ def _warm_efforts_async() -> None:
_seed_efforts(items)
try:
threading.Thread(
target=_refresh, name="router-caps-warm", daemon=True
).start()
threading.Thread(target=_refresh, name="router-caps-warm", daemon=True).start()
except Exception as exc:
logger.debug("router: caps warmer failed to start: %s", exc)
@@ -309,46 +242,19 @@ class RouterProfile(ProviderProfile):
"""Ramp Router — Responses-only gateway with catalog-declared efforts."""
def fetch_models(
self,
*,
api_key: Optional[str] = None,
base_url: Optional[str] = None,
timeout: float = 8.0,
self, *, api_key: Optional[str] = None, base_url: Optional[str] = None, timeout: float = 8.0
) -> Optional[list[str]]:
"""Fetch the live, key-scoped catalog and seed the caps cache.
One request serves both consumers: the picker gets the model IDs and
the reasoning-vocabulary mirror is left warm at no extra network
cost (the same document carries both).
"""
items = _fetch_catalog_items(
api_key=api_key or "", base_url=base_url or "", timeout=timeout
)
"""Fetch the live, key-scoped catalog; the same payload seeds the caps cache.
Deduped but not sorted: Router's listing order is deliberate presentation."""
items = _fetch_catalog_items(api_key=api_key or "", base_url=base_url or "", timeout=timeout)
if items is None:
return None
_seed_efforts(items)
# Deduped but not sorted: Router's listing order is deliberate
# presentation (featured/current models first), so the picker keeps it.
ids = list(
dict.fromkeys(
str(item["id"])
for item in items
if isinstance(item, dict) and item.get("id")
)
)
ids = list(dict.fromkeys(str(i["id"]) for i in items if isinstance(i, dict) and i.get("id")))
return ids or None
def supported_reasoning_efforts(
self, model: Optional[str]
) -> Optional[tuple[str, ...]]:
"""Catalog-declared effort vocabulary for *model* (cache-only).
Router 400s on efforts outside a model's published set and on any
reasoning field for non-reasoning models, so the codex transport
clamps (or suppresses) from this verdict. Cold cache returns None —
the transport keeps its defaults — and kicks a background warmer so
the next turn is covered.
"""
def supported_reasoning_efforts(self, model: Optional[str]) -> Optional[tuple[str, ...]]:
"""Catalog-declared effort vocabulary (cache-only; cold cache -> None + warm)."""
mid = str(model or "").strip()
if not mid:
return None
@@ -356,10 +262,7 @@ class RouterProfile(ProviderProfile):
if efforts_by_id is None:
_warm_efforts_async()
return None
levels = efforts_by_id.get(mid)
if levels is None:
return None
return tuple(levels)
return None if mid not in efforts_by_id else tuple(efforts_by_id[mid])
router = RouterProfile(
@@ -369,26 +272,14 @@ router = RouterProfile(
display_name="Ramp Router",
description="Ramp Router (router.com) — routes each request to the cheapest model that clears your quality bar",
signup_url="https://app.router.com/keys",
# RAMP_ROUTER_API_KEY is Router's documented variable; ROUTER_API_KEY is
# a convenience alias. RAMP_ROUTER_BASE_URL overrides the endpoint
# (auth.py picks it up as the registry's base_url_env_var).
env_vars=("RAMP_ROUTER_API_KEY", "ROUTER_API_KEY", "RAMP_ROUTER_BASE_URL"),
base_url=_base_url(),
auth_type="api_key",
# Identify Hermes traffic to the gateway (Router attributes coding-agent
# clients by User-Agent prefix, the way it already recognizes OpenCode's
# versioned UA) — and Router's WAF rejects blank/default client UAs.
# Router attributes coding-agent clients by UA prefix; its WAF rejects default UAs.
default_headers={"User-Agent": f"Hermes-Agent/{_HERMES_VERSION}"},
# Most of the catalog's frontier routes accept image input; capability is
# still model-dependent and governed by the live catalog.
supports_vision=True,
# Cheap, reasoning-capable, and vision-capable — safe for auxiliary tasks
# (compaction, titles, vision) when Router is the main provider. Also the
# model Router's own docs use as their example.
default_aux_model="gpt-5.4-mini",
# Deliberately empty: model IDs are account-scoped (BYOK accounts see
# extra entries) and Router's docs say to read the catalog at runtime
# rather than hardcode names. The picker uses fetch_models() above.
# Empty on purpose: model IDs are account-scoped; the picker uses fetch_models().
fallback_models=(),
)
+26 -82
View File
@@ -1,102 +1,49 @@
"""Upstage Solar provider profile."""
"""Upstage Solar provider profile: top-level ``reasoning_effort`` (low|medium|high).
Solar's server default is ``minimal`` (reasoning off) — wrong for agentic work —
so an unset reasoning_config defaults reasoning ON at ``medium``, matching the
"medium (default)" the /reasoning panel shows. Explicit settings always win.
"""
from typing import Any
from agent.reasoning_effort import EFFORT_LADDER, SOLAR_EFFORTS, clamp_effort
from providers import register_provider
from providers.base import ProviderProfile
# Model-name markers for Solar families that do NOT accept ``reasoning_effort``.
# Deny-list on purpose: newly released Solar models are assumed
# reasoning-capable by default, so only the known non-reasoning families are
# listed here. Substring match (not startswith) so dated variants like
# ``solar-mini-250127`` are covered too.
# Deny-list on purpose: new Solar models are assumed reasoning-capable; only
# these known non-reasoning families ignore reasoning_effort. Substring match
# so dated variants (``solar-mini-250127``) are covered.
_NON_REASONING_MODEL_MARKERS = ("solar-mini", "syn-pro")
# When the user hasn't picked a reasoning effort, Hermes passes
# reasoning_config=None. Solar's own server default is "minimal" (reasoning
# off), which is the wrong default for an agentic workload. We default reasoning
# ON at this effort — matching the "medium (default)" that Hermes' /reasoning
# panel shows for an unset config, so the displayed default and the real wire
# value agree. An explicit saved setting or a `/reasoning <level>` change is
# always honored over this default; `/reasoning none` disables it.
_DEFAULT_REASONING_EFFORT = "medium"
def _model_supports_reasoning(model: str | None) -> bool:
"""Solar reasoning-capable models — True unless the model is deny-listed.
The Solar Pro family (``solar-pro``, ``solar-pro2``, ``solar-pro3`` and
dated variants like ``solar-pro3-250127``) and the Solar Open family
(``solar-open*``) accept ``reasoning_effort``; only ``solar-mini`` /
``syn-pro`` ignore the parameter, so we deny-list those and treat every
other (incl. future) Solar model as reasoning-capable.
``None``/empty model → True: the provider default (``fallback_models[0]``,
``solar-pro3``) is reasoning-capable, so an unset model gets the same
default-on behaviour.
"""
m = (model or "").strip().lower()
return not any(marker in m for marker in _NON_REASONING_MODEL_MARKERS)
class UpstageProfile(ProviderProfile):
"""Upstage Solar — top-level ``reasoning_effort`` control.
Solar Pro/Open expose reasoning through a top-level ``reasoning_effort``
field (``minimal`` | ``low`` | ``medium`` | ``high``), mirroring OpenAI's
shape. Unlike DeepSeek/Kimi it does NOT require echoing ``reasoning_content``
back on later turns, so only the request field needs wiring. We emit at most
``low`` | ``medium`` | ``high`` — the explicit values both Solar Pro 2 and
Pro 3 accept.
Default-on: Solar's own server default is ``minimal`` (off), but for an
agentic workload we default reasoning ON (``_DEFAULT_REASONING_EFFORT``)
when the user hasn't picked an effort. The user can still set any level or
turn it off with ``/reasoning none``.
"""
"""Upstage Solar — top-level ``reasoning_effort`` control (no reasoning_content echo needed)."""
def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
) -> tuple[dict[str, Any], dict[str, Any]]:
top_level: dict[str, Any] = {}
# solar-mini / syn-pro (the deny-list) ignore reasoning_effort — send
# nothing. Everything else, including future Solar models, gets it.
if not _model_supports_reasoning(model):
return {}, top_level
# Unset (reasoning_config is None) → default reasoning ON for agents.
m = (model or "").strip().lower()
if any(marker in m for marker in _NON_REASONING_MODEL_MARKERS):
return {}, {}
# Unset -> default reasoning ON for agents.
if not reasoning_config or not isinstance(reasoning_config, dict):
return {}, {"reasoning_effort": _DEFAULT_REASONING_EFFORT}
# Explicitly disabled (`/reasoning none`) → omit the field so Solar
# applies its own default (minimal = off).
return {}, {"reasoning_effort": "medium"}
# Explicitly disabled -> omit so Solar applies its own default (minimal = off).
if reasoning_config.get("enabled") is False:
return {}, top_level
# Map Hermes' effort vocabulary onto Solar's accepted set via the
# shared clamp (agent.reasoning_effort). minimal → omit (Solar's
# minimal means off); unknown-but-enabled bespoke levels collapse to
# high rather than silently downgrading (#62650 precedent).
return {}, {}
effort = (reasoning_config.get("effort") or "").strip().lower()
if not effort:
top_level["reasoning_effort"] = _DEFAULT_REASONING_EFFORT
return {}, top_level
return {}, {"reasoning_effort": "medium"}
if effort == "minimal":
return {}, top_level
from agent.reasoning_effort import EFFORT_LADDER, SOLAR_EFFORTS, clamp_effort
return {}, {}
mapped = clamp_effort(effort, SOLAR_EFFORTS)
if mapped not in SOLAR_EFFORTS:
# Bespoke level outside the ladder — Solar precedent is to run
# at full strength rather than quietly fall to the default.
# Bespoke level outside the ladder runs at full strength rather
# than quietly falling to the default; ladder levels that still
# don't map are omitted.
mapped = "high" if effort not in EFFORT_LADDER else None
if mapped:
top_level["reasoning_effort"] = mapped
return {}, top_level
return {}, {"reasoning_effort": mapped} if mapped else {}
upstage = UpstageProfile(
@@ -108,11 +55,8 @@ upstage = UpstageProfile(
env_vars=("UPSTAGE_API_KEY", "UPSTAGE_BASE_URL"),
base_url="https://api.upstage.ai/v1",
auth_type="api_key",
# default_aux_model left empty → auxiliary side tasks use the main model.
# entry [0] is the setup default — solar-pro3, the current Solar Pro flagship.
fallback_models=(
"solar-pro3",
),
# No default_aux_model: auxiliary tasks use the main model. [0] is the setup default.
fallback_models=("solar-pro3",),
)
register_provider(upstage)
+13 -38
View File
@@ -1,20 +1,10 @@
"""Google Vertex AI provider profile.
"""Google Vertex AI provider profile: Gemini via Google Cloud's OpenAI-compatible
endpoint.
vertex: Gemini models via Google Cloud's OpenAI-compatible endpoint.
Auth is OAuth2 — short-lived access tokens minted from a service-account JSON
or Application Default Credentials (ADC), NOT a static API key. Token
resolution and refresh live in ``agent/vertex_adapter.py``; runtime_provider.py
calls it to obtain a fresh ``(token, base_url)`` pair, then hands the token to
the standard OpenAI client as ``api_key``. Because the wire format is the
OpenAI-compatible chat/completions surface, no message translation is needed —
the only Gemini-specific concern is the ``thinking_config`` reasoning hook,
which is emitted here exactly as the ``gemini`` provider does for its
OpenAI-compat subpath (``extra_body.google.thinking_config``).
``auth_type="vertex"`` marks this as an OAuth-token provider (resolved
specially, like bedrock's ``aws_sdk``) so it is never treated as an
api_key provider that would mistake a credentials-file path for a key.
Auth is OAuth2 (service-account JSON or ADC), not a static key: ``agent/
vertex_adapter.py`` mints ``(token, base_url)`` and the token is passed as
``api_key``. ``auth_type="vertex"`` keeps it out of the api_key provider path so
a credentials-file path is never mistaken for a key.
"""
from typing import Any
@@ -26,39 +16,24 @@ from providers.base import ProviderProfile
class VertexProfile(ProviderProfile):
"""Vertex AI — reuse Gemini's thinking_config translation for extra_body."""
def build_extra_body(
self, *, session_id: str | None = None, **context: Any
) -> dict[str, Any]:
"""Emit ``extra_body.google.thinking_config`` for the OpenAI-compat
Vertex surface, mirroring the ``gemini`` provider's behavior.
"""
def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]:
"""Emit ``extra_body.google.thinking_config`` like the ``gemini`` provider's
OpenAI-compat subpath."""
from agent.transports.chat_completions import (
_build_gemini_thinking_config,
_snake_case_gemini_thinking_config,
)
model = context.get("model") or ""
reasoning_config = context.get("reasoning_config")
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
if not raw_thinking_config:
return {}
thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config)
raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config"))
thinking_config = _snake_case_gemini_thinking_config(raw) if raw else None
if not thinking_config:
return {}
return {"extra_body": {"google": {"thinking_config": thinking_config}}}
def fetch_models(
self,
*,
api_key: str | None = None,
base_url: str | None = None,
timeout: float = 8.0,
self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0
) -> list[str] | None:
"""Vertex's OpenAI-compat endpoint has no ``/models`` listing route;
model discovery is not available. The setup wizard ships a curated list.
"""
"""No ``/models`` route on the OpenAI-compat endpoint; setup ships a curated list."""
return None
+26 -105
View File
@@ -1,27 +1,7 @@
"""ZAI / GLM provider profile.
Z.AI's GLM-4.5-and-later chat models default to thinking-mode ON when the
request omits ``thinking``. Hermes' ``reasoning_config = {"enabled": False}``
was previously a silent no-op on this route — the base profile emits nothing,
so users who turned thinking off (desktop toggle, ``/reasoning none``,
``reasoning_effort: none``/``false`` in config.yaml) kept burning thinking
tokens on every turn.
:meth:`ZaiProfile.build_api_kwargs_extras` translates the Hermes reasoning
config into the wire shape Z.AI's OpenAI-compat endpoint expects:
{"extra_body": {"thinking": {"type": "enabled" | "disabled"}}}
When no reasoning preference is set (``reasoning_config is None``) the field
is omitted so the server default applies, matching prior behavior. GLM
models before 4.5 (e.g. ``glm-4-9b``) don't accept ``thinking`` and are left
untouched.
GLM-5.2 additionally exposes a native ``reasoning_effort`` knob with exactly
two enabled levels — ``high`` and ``max`` — on the OpenAI-compatible endpoint
(per Z.AI / BigModel docs). Hermes' richer effort scale is collapsed onto
those two so the user's effort preference actually reaches the model instead
of being silently dropped.
GLM-4.5+ defaults to thinking ON, so ``reasoning_config`` is translated to
``extra_body.thinking``; GLM-5.2/5.3 also take a native ``reasoning_effort``.
"""
from __future__ import annotations
@@ -29,93 +9,40 @@ from __future__ import annotations
import re
from typing import Any
from agent import reasoning_effort as re_
from providers import register_provider
from providers.base import ProviderProfile
_GLM_VERSION_RE = re.compile(r"^glm-(\d+)(?:\.(\d+))?")
# Alias spellings seen on relays (Fireworks ``glm-5p2``, ``zai-org-glm-5-2``…).
_GLM_5_3_TOKENS = ("glm-5.3", "glm-5-3", "glm-5p3")
_GLM_5_2_TOKENS = ("glm-5.2", "glm-5-2", "glm-5p2") + _GLM_5_3_TOKENS
def _model_supports_thinking(model: str | None) -> bool:
"""GLM thinking-capable model families: glm-4.5 and later (4.5, 4.6, 5…)."""
match = _GLM_VERSION_RE.match((model or "").strip().lower())
return bool(match) and (int(match.group(1)), int(match.group(2) or 0)) >= (4, 5)
def _has_token(model: str | None, tokens: tuple[str, ...]) -> bool:
m = (model or "").strip().lower()
match = _GLM_VERSION_RE.match(m)
if not match:
return False
major = int(match.group(1))
minor = int(match.group(2) or 0)
return (major, minor) >= (4, 5)
return bool(m) and any(token in m for token in tokens)
def _is_glm_5_2(model: str | None) -> bool:
"""Detect GLM-5.2/5.3 (reasoning_effort-capable) across alias spellings.
def _glm_5_2_reasoning_effort(reasoning_config: dict | None, *, model: str | None = None) -> str | None:
"""Map Hermes effort onto GLM's vocabulary (5.2: high/max; 5.3: low..max).
Covers the canonical ``glm-5.2``/``glm-5.3`` plus the ``glm-5-2`` /
``glm-5p2`` variants seen on relays (Fireworks ``glm-5p2``, etc.) and any
vendor-prefixed form (``z-ai/glm-5.2``, ``zai-org-glm-5-2``). GLM-5.3
uses the same base model as 5.2 (post-training gains only) and exposes
the same ``reasoning_effort`` knob (verified live 2026-08-14: the
coding-plan endpoint accepts ``reasoning_effort: high`` for glm-5.3).
Below-floor efforts clamp to the floor; disabled/unset leaves the server default.
"""
m = (model or "").strip().lower()
if not m:
return False
return any(
token in m
for token in ("glm-5.2", "glm-5-2", "glm-5p2", "glm-5.3", "glm-5-3", "glm-5p3")
)
def _is_glm_5_3(model: str | None) -> bool:
"""Detect GLM-5.3 specifically — it has a wider effort vocabulary.
5.2 accepts only ``high``/``max``; 5.3 accepts a graded
``low``/``medium``/``high``/``max`` scale (verified live, issue #91789),
so effort mapping must pick the vocabulary per model.
"""
m = (model or "").strip().lower()
if not m:
return False
return any(token in m for token in ("glm-5.3", "glm-5-3", "glm-5p3"))
def _glm_5_2_reasoning_effort(
reasoning_config: dict | None, *, model: str | None = None
) -> str | None:
"""Map Hermes reasoning effort onto GLM's native vocabulary.
GLM-5.2 supports two enabled effort levels (``high``/``max``);
GLM-5.3 supports the graded ``low``/``medium``/``high``/``max`` scale.
``xhigh``/``max``/``ultra`` request the top tier; anything below the
model's floor clamps to that floor. When reasoning is explicitly
disabled, or no effort preference is supplied, the server default is
left untouched.
"""
if not isinstance(reasoning_config, dict):
effort = re_.requested_effort(reasoning_config)
if effort is None or effort == "none":
return None
if reasoning_config.get("enabled") is False:
return None
effort = (reasoning_config.get("effort") or "").strip().lower()
if not effort or effort == "none":
return None
# Per-model vocabulary declared in agent.reasoning_effort; xhigh rounds
# up to max on both. 5.2 cannot think less than high; 5.3 accepts a
# graded scale down to low (issue #91789).
from agent.reasoning_effort import (
GLM52_EFFORTS,
GLM52_OVERRIDES,
GLM53_EFFORTS,
GLM53_OVERRIDES,
clamp_effort,
)
if _is_glm_5_3(model):
efforts, overrides, floor = GLM53_EFFORTS, GLM53_OVERRIDES, "low"
if _has_token(model, _GLM_5_3_TOKENS):
efforts, overrides, floor = re_.GLM53_EFFORTS, re_.GLM53_OVERRIDES, "low"
else:
efforts, overrides, floor = GLM52_EFFORTS, GLM52_OVERRIDES, "high"
clamped = clamp_effort(effort, efforts, overrides)
efforts, overrides, floor = re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES, "high"
clamped = re_.clamp_effort(effort, efforts, overrides)
return clamped if clamped in efforts else floor
@@ -127,21 +54,19 @@ class ZaiProfile(ProviderProfile):
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
top_level: dict[str, Any] = {}
if not _model_supports_thinking(model) and not _is_glm_5_2(model):
is_5_2 = _has_token(model, _GLM_5_2_TOKENS)
if not _model_supports_thinking(model) and not is_5_2:
return extra_body, top_level
# Only emit when the user expressed a preference; omitting the field
# keeps the server default (enabled) exactly as before.
# Only emit when the user expressed a preference (server default = enabled).
if isinstance(reasoning_config, dict):
enabled = reasoning_config.get("enabled") is not False
extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"}
if _is_glm_5_2(model):
if is_5_2:
effort = _glm_5_2_reasoning_effort(reasoning_config, model=model)
if effort is not None:
top_level["reasoning_effort"] = effort
return extra_body, top_level
@@ -152,11 +77,7 @@ zai = ZaiProfile(
display_name="Z.AI (GLM)",
description="Z.AI / GLM — Zhipu AI models",
signup_url="https://z.ai/",
fallback_models=(
"glm-5.2",
"glm-5",
"glm-4-9b",
),
fallback_models=("glm-5.2", "glm-5", "glm-4-9b"),
base_url="https://api.z.ai/api/paas/v4",
default_aux_model="glm-4.5-flash",
)