From 313b244e8f76045ab5eb1540bb25dbbf9ee66eb6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:23:23 -0700 Subject: [PATCH] refactor(plugins/model-providers): reuse agent.reasoning_effort clamps, per-model dict tables, compact profiles --- plugins/model-providers/actual/__init__.py | 29 +- .../model-providers/ai-gateway/__init__.py | 22 +- .../alibaba-coding-plan/__init__.py | 18 +- plugins/model-providers/alibaba/__init__.py | 19 +- plugins/model-providers/anthropic/__init__.py | 12 +- .../model-providers/azure-foundry/__init__.py | 7 +- plugins/model-providers/bedrock/__init__.py | 6 +- .../model-providers/commandcode/__init__.py | 125 ++------- .../model-providers/copilot-acp/__init__.py | 14 +- plugins/model-providers/copilot/__init__.py | 71 ++--- plugins/model-providers/custom/__init__.py | 121 ++------ plugins/model-providers/deepinfra/__init__.py | 54 +--- plugins/model-providers/deepseek/__init__.py | 102 ++----- plugins/model-providers/fireworks/__init__.py | 22 +- plugins/model-providers/gemini/__init__.py | 41 +-- plugins/model-providers/gmi/__init__.py | 4 +- .../model-providers/huggingface/__init__.py | 5 +- .../model-providers/kimi-coding/__init__.py | 113 ++------ plugins/model-providers/meta-ai/__init__.py | 100 ++----- plugins/model-providers/minimax/__init__.py | 38 +-- .../nebius-token-factory/__init__.py | 77 ++--- plugins/model-providers/nous/__init__.py | 100 ++----- plugins/model-providers/nvidia/__init__.py | 42 +-- .../model-providers/ollama-cloud/__init__.py | 85 ++---- .../model-providers/opencode-free/__init__.py | 46 ++- .../model-providers/opencode-zen/__init__.py | 176 +++--------- .../model-providers/openrouter/__init__.py | 189 ++++--------- .../model-providers/qwen-oauth/__init__.py | 89 +++--- plugins/model-providers/router/__init__.py | 263 +++++------------- plugins/model-providers/upstage/__init__.py | 108 ++----- plugins/model-providers/vertex/__init__.py | 51 +--- plugins/model-providers/zai/__init__.py | 131 ++------- 32 files changed, 595 insertions(+), 1685 deletions(-) diff --git a/plugins/model-providers/actual/__init__.py b/plugins/model-providers/actual/__init__.py index 0141dd8175..9cce101ef8 100644 --- a/plugins/model-providers/actual/__init__.py +++ b/plugins/model-providers/actual/__init__.py @@ -14,10 +14,11 @@ from providers.base import ProviderProfile, _profile_user_agent logger = logging.getLogger(__name__) DEFAULT_ACTUAL_BASE_URL = "https://api.actual.inc/v1" -DEFAULT_ACTUAL_LOCAL_BASE_URL = "http://127.0.0.1:8080/v1" +_LOCAL_HOSTS = {"localhost", "127.0.0.1", "::1", "0.0.0.0"} def _normalize_actual_base_url(base_url: str) -> str: + """Append /v1 to a bare hosted or local host; pass anything else through.""" url = str(base_url or "").strip().rstrip("/") if not url: return DEFAULT_ACTUAL_BASE_URL @@ -27,42 +28,28 @@ def _normalize_actual_base_url(base_url: str) -> str: path = parsed.path.rstrip("/") except Exception: return url - if host == "api.actual.inc" and path in {"", "/"}: - return url + "/v1" - if host in {"localhost", "127.0.0.1", "::1", "0.0.0.0"} and path in {"", "/"}: + if (host == "api.actual.inc" or host in _LOCAL_HOSTS) and path in {"", "/"}: return url + "/v1" return url class ActualProfile(ProviderProfile): - """Actual Computer provider. - - Hosted inference defaults to api.actual.inc. Local inference is exposed by - the Actual client only when it runs in offline mode, so users opt into it by - setting ACTUAL_BASE_URL to the local API URL. - """ + """Actual Computer: hosted at api.actual.inc; local (offline-mode client) + inference opted into via ACTUAL_BASE_URL.""" def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: + from hermes_cli.urllib_security import open_credentialed_url + base_url = _normalize_actual_base_url( os.getenv("ACTUAL_BASE_URL", "").strip() or base_url or self.base_url ) - if not base_url: - return None - req = urllib.request.Request(base_url + "/models") if api_key: req.add_header("Authorization", f"Bearer {api_key}") req.add_header("Accept", "application/json") req.add_header("User-Agent", _profile_user_agent()) - - from hermes_cli.urllib_security import open_credentialed_url - try: with open_credentialed_url(req, timeout=timeout) as resp: data = json.loads(resp.read().decode()) diff --git a/plugins/model-providers/ai-gateway/__init__.py b/plugins/model-providers/ai-gateway/__init__.py index 9d01ab9824..a89c7c2d33 100644 --- a/plugins/model-providers/ai-gateway/__init__.py +++ b/plugins/model-providers/ai-gateway/__init__.py @@ -1,8 +1,4 @@ -"""Vercel AI Gateway provider profile. - -AI Gateway routes to multiple backends. Hermes sends attribution -headers and full reasoning config passthrough. -""" +"""Vercel AI Gateway provider profile: attribution headers + reasoning passthrough.""" from typing import Any @@ -14,18 +10,12 @@ class VercelAIGatewayProfile(ProviderProfile): """Vercel AI Gateway — attribution headers + reasoning passthrough.""" def build_api_kwargs_extras( - self, - *, - reasoning_config: dict | None = None, - supports_reasoning: bool = True, - **ctx: Any, + self, *, reasoning_config: dict | None = None, supports_reasoning: bool = True, **ctx: Any ) -> tuple[dict[str, Any], dict[str, Any]]: - extra_body: dict[str, Any] = {} - if supports_reasoning and reasoning_config is not None: - extra_body["reasoning"] = dict(reasoning_config) - elif supports_reasoning: - extra_body["reasoning"] = {"enabled": True, "effort": "medium"} - return extra_body, {} + if not supports_reasoning: + return {}, {} + reasoning = dict(reasoning_config) if reasoning_config is not None else {"enabled": True, "effort": "medium"} + return {"reasoning": reasoning}, {} vercel = VercelAIGatewayProfile( diff --git a/plugins/model-providers/alibaba-coding-plan/__init__.py b/plugins/model-providers/alibaba-coding-plan/__init__.py index 4723606d0e..9c19d153c1 100644 --- a/plugins/model-providers/alibaba-coding-plan/__init__.py +++ b/plugins/model-providers/alibaba-coding-plan/__init__.py @@ -1,18 +1,8 @@ -"""Alibaba Cloud Coding Plan provider profiles. +"""Alibaba Cloud Coding Plan provider profiles (intl + CN): a dedicated endpoint +and key tier separate from ``alibaba``. Names match models.dev catalog keys. -Separate from the standard `alibaba` profile because it hits a different -endpoint (coding-intl.dashscope.aliyuncs.com) with a dedicated API key tier. - -Region split, mirroring the base DashScope pair (#73265): - - ``alibaba-coding-plan`` → coding-intl.dashscope.aliyuncs.com (international) - - ``alibaba-coding-plan-cn`` → coding.dashscope.aliyuncs.com (mainland China) - -Profile names match the models.dev catalog keys exactly so model metadata -lines up and ``model.provider: alibaba-coding-plan-cn`` resolves at runtime. - -The CN profile checks its own ``ALIBABA_CODING_PLAN_CN_API_KEY`` first (#101122, -mirroring kimi-coding-cn) and keeps the shared vars as ordered fallbacks so -existing CN users configured with the shared key keep working. +The CN profile checks its own key first and keeps the shared vars as ordered +fallbacks so existing CN users configured with the shared key keep working. """ from providers import register_provider diff --git a/plugins/model-providers/alibaba/__init__.py b/plugins/model-providers/alibaba/__init__.py index a9198d3931..6ed2f58c45 100644 --- a/plugins/model-providers/alibaba/__init__.py +++ b/plugins/model-providers/alibaba/__init__.py @@ -1,19 +1,8 @@ -"""Alibaba Cloud DashScope provider profiles. +"""Alibaba Cloud DashScope provider profiles (intl + CN, plus the Model Studio +Token Plan flat-token tier with its own key/endpoints — one module per vendor). -DashScope has region-split endpoints with the same key type: - - ``alibaba`` → dashscope-intl.aliyuncs.com (international) - - ``alibaba-cn`` → dashscope.aliyuncs.com (mainland China) - -The Model Studio Token Plan (flat-token tier of the SAME vendor/service, -same OpenAI-compatible protocol, its own key + endpoints) registers here -too rather than as a new plugin directory — one module per vendor, matching -how the kimi module carries both of its endpoint variants: - - ``alibaba-token-plan`` → token-plan.ap-southeast-1.maas.aliyuncs.com - - ``alibaba-token-plan-cn`` → token-plan.cn-beijing.maas.aliyuncs.com - -Profile names match the models.dev catalog keys exactly -(``alibaba`` / ``alibaba-cn``) so model metadata lines up and -``model.provider: alibaba-cn`` resolves at runtime (#73265). +Profile names match models.dev catalog keys exactly so model metadata lines up +and ``model.provider: alibaba-cn`` resolves at runtime. """ from providers import register_provider diff --git a/plugins/model-providers/anthropic/__init__.py b/plugins/model-providers/anthropic/__init__.py index 2773476c93..fbe5269f9f 100644 --- a/plugins/model-providers/anthropic/__init__.py +++ b/plugins/model-providers/anthropic/__init__.py @@ -15,11 +15,7 @@ class AnthropicProfile(ProviderProfile): """Native Anthropic — uses x-api-key header, not Bearer.""" def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: """Anthropic uses x-api-key header and anthropic-version.""" if not api_key: @@ -31,11 +27,7 @@ class AnthropicProfile(ProviderProfile): req.add_header("Accept", "application/json") with open_credentialed_url(req, timeout=timeout) as resp: data = json.loads(resp.read().decode()) - return [ - m["id"] - for m in data.get("data", []) - if isinstance(m, dict) and "id" in m - ] + return [m["id"] for m in data.get("data", []) if isinstance(m, dict) and "id" in m] except Exception as exc: logger.debug("fetch_models(anthropic): %s", exc) return None diff --git a/plugins/model-providers/azure-foundry/__init__.py b/plugins/model-providers/azure-foundry/__init__.py index 50968805f5..a0fc7d2d60 100644 --- a/plugins/model-providers/azure-foundry/__init__.py +++ b/plugins/model-providers/azure-foundry/__init__.py @@ -1,8 +1,5 @@ -"""Microsoft Foundry provider profile. - -Azure Foundry exposes an OpenAI-compatible endpoint; users supply their own -base URL at setup since endpoints are per-resource. -""" +"""Microsoft Foundry provider profile: OpenAI-compatible, per-resource base URL +supplied by the user at setup.""" from providers import register_provider from providers.base import ProviderProfile diff --git a/plugins/model-providers/bedrock/__init__.py b/plugins/model-providers/bedrock/__init__.py index d0ee99c58c..0b23f4cbdd 100644 --- a/plugins/model-providers/bedrock/__init__.py +++ b/plugins/model-providers/bedrock/__init__.py @@ -8,11 +8,7 @@ class BedrockProfile(ProviderProfile): """AWS Bedrock — no REST /v1/models endpoint; uses AWS SDK.""" def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: """Bedrock model listing requires AWS SDK, not a REST call.""" return None diff --git a/plugins/model-providers/commandcode/__init__.py b/plugins/model-providers/commandcode/__init__.py index 1eff3d2380..7191d9d4f0 100644 --- a/plugins/model-providers/commandcode/__init__.py +++ b/plugins/model-providers/commandcode/__init__.py @@ -1,28 +1,6 @@ -"""CommandCode provider profile. - -CommandCode provides a unified API that fronts 20+ models from DeepSeek, Qwen, -Kimi, GLM, MiniMax, StepFun, Xiaomi Mimo, Google Gemini, and OpenAI GPT — all -accessible through either OpenAI-compatible chat completions or Anthropic -Messages endpoints from a single base URL and API key. - -Two provider profiles are registered: - -``commandcode`` - ``api_mode=chat_completions`` — standard OpenAI-compatible endpoint. - Model prefix: ``deepseek/deepseek-v4-pro``, ``Qwen/Qwen3.7-Max``, etc. - -``commandcode-anthropic`` - ``api_mode=anthropic_messages`` — Anthropic Messages API-compatible. - Model names: ``claude-sonnet-4-6``, ``claude-opus-4-7``, - ``claude-haiku-4-5-20251001``. - -Both use the same ``COMMANDCODE_API_KEY`` env var and -``https://api.commandcode.ai/provider/v1`` base URL. The -``commandcode-anthropic`` profile relies on ``agent/anthropic_adapter.py`` -recognizing the ``api.commandcode.ai`` hostname for Bearer auth (the -CommandCode /anthropic endpoint uses ``Authorization: Bearer``, not -Anthropic's native ``x-api-key`` header). -""" +"""CommandCode provider profiles: ``commandcode`` (chat_completions) and +``commandcode-anthropic`` (anthropic_messages, Bearer auth — see +``agent/anthropic_adapter.py``). Same key and base URL for both.""" from __future__ import annotations @@ -35,74 +13,63 @@ from providers.base import ProviderProfile, _profile_user_agent logger = logging.getLogger(__name__) -# ── Shared constants ────────────────────────────────────────────────────────── _COMMANDCODE_BASE = "https://api.commandcode.ai/provider/v1" _COMMANDCODE_MODELS_URL = f"{_COMMANDCODE_BASE}/models" -# Both profiles authenticate with the same key; each carries its own base-URL -# override var so each renders its own card on the desktop Keys tab (rows are -# keyed by env var, and the shared API key attributes to the first profile). -_COMMANDCODE_ENV = ("COMMANDCODE_API_KEY", "COMMANDCODE_BASE_URL") -_COMMANDCODE_ANTHROPIC_ENV = ("COMMANDCODE_API_KEY", "COMMANDCODE_ANTHROPIC_BASE_URL") def _fetch_commandcode_models( timeout: float = 10.0, base_url: str | None = None, ) -> list[str] | None: - """Fetch the live model list from the CommandCode /models endpoint. + """Fetch model IDs from the public (unauthenticated) /models endpoint. - Returns a flat list of model IDs or None on failure. - No auth required — the public models endpoint is open. - - ``base_url`` overrides the endpoint only when the caller passed a URL - that differs from the default ``_COMMANDCODE_BASE`` (a user-configured - ``model.base_url`` / ``COMMANDCODE_BASE_URL`` pointing at a proxy or - custom deployment). The picker passes base_url unconditionally, falling - back to the profile default — equality means "not customised". + The picker passes base_url unconditionally, so only a value differing from + the default counts as a customised endpoint. """ - caller_base = (base_url or "").strip() - if caller_base and caller_base.rstrip("/") != _COMMANDCODE_BASE.rstrip("/"): - models_url = caller_base.rstrip("/") + "/models" - else: - models_url = _COMMANDCODE_MODELS_URL + caller_base = (base_url or "").strip().rstrip("/") + custom = caller_base and caller_base != _COMMANDCODE_BASE + models_url = caller_base + "/models" if custom else _COMMANDCODE_MODELS_URL try: req = urllib.request.Request(models_url) req.add_header("Accept", "application/json") req.add_header("User-Agent", _profile_user_agent()) with urllib.request.urlopen(req, timeout=timeout) as resp: data = json.loads(resp.read().decode()) - # Response shape: {"object": "list", "data": [{"id": "..."}, ...]} - return [ - m["id"] - for m in data.get("data", []) - if isinstance(m, dict) and "id" in m - ] + return [m["id"] for m in data.get("data", []) if isinstance(m, dict) and "id" in m] except Exception as exc: logger.debug("fetch_models(commandcode): %s", exc) return None -# ── Chat Completions profile ────────────────────────────────────────────────── - class CommandCodeProfile(ProviderProfile): """CommandCode — OpenAI-compatible chat completions endpoint.""" def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: - """Fetch from the public CommandCode /models endpoint.""" return _fetch_commandcode_models(timeout=timeout, base_url=base_url) +class CommandCodeAnthropicProfile(ProviderProfile): + """CommandCode — Anthropic Messages API-compatible endpoint.""" + + def fetch_models( + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 + ) -> list[str] | None: + """Public /models endpoint, filtered to Anthropic-family models.""" + all_models = _fetch_commandcode_models(timeout=timeout, base_url=base_url) + if all_models is None: + return None + return [m for m in all_models if m.startswith("claude-")] + + commandcode = CommandCodeProfile( name="commandcode", aliases=("commandcode-chat",), api_mode="chat_completions", - env_vars=_COMMANDCODE_ENV, + # Same key as the anthropic profile; distinct base-URL override vars so each + # profile renders its own card on the desktop Keys tab (rows keyed by env var). + env_vars=("COMMANDCODE_API_KEY", "COMMANDCODE_BASE_URL"), display_name="CommandCode", description="CommandCode — 20+ models via OpenAI-compatible API", signup_url="https://commandcode.ai/", @@ -124,53 +91,19 @@ commandcode = CommandCodeProfile( default_aux_model="deepseek/deepseek-v4-flash", ) - -# ── Anthropic Messages profile ──────────────────────────────────────────────── - -class CommandCodeAnthropicProfile(ProviderProfile): - """CommandCode — Anthropic Messages API-compatible endpoint. - - Uses Bearer auth (same API key), not Anthropic's native x-api-key header. - ``agent/anthropic_adapter.py`` must recognize ``api.commandcode.ai`` - as a Bearer-auth domain for this to work. - """ - - def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, - ) -> list[str] | None: - """Fetch from the public CommandCode /models endpoint. - - Filter to Anthropic-family models only (claude-*). - """ - all_models = _fetch_commandcode_models(timeout=timeout, base_url=base_url) - if all_models is None: - return None - return [m for m in all_models if m.startswith("claude-")] - - commandcode_anthropic = CommandCodeAnthropicProfile( name="commandcode-anthropic", aliases=("commandcode-claude",), api_mode="anthropic_messages", - env_vars=_COMMANDCODE_ANTHROPIC_ENV, + env_vars=("COMMANDCODE_API_KEY", "COMMANDCODE_ANTHROPIC_BASE_URL"), display_name="CommandCode (Anthropic)", description="CommandCode — Claude models via Anthropic Messages API", signup_url="https://commandcode.ai/", base_url=_COMMANDCODE_BASE, models_url=_COMMANDCODE_MODELS_URL, - fallback_models=( - "claude-sonnet-4-6", - "claude-opus-4-7", - "claude-haiku-4-5-20251001", - ), + fallback_models=("claude-sonnet-4-6", "claude-opus-4-7", "claude-haiku-4-5-20251001"), default_aux_model="claude-haiku-4-5-20251001", ) - -# ── Registration ────────────────────────────────────────────────────────────── register_provider(commandcode) register_provider(commandcode_anthropic) diff --git a/plugins/model-providers/copilot-acp/__init__.py b/plugins/model-providers/copilot-acp/__init__.py index 89f51b850f..ef7d90ec46 100644 --- a/plugins/model-providers/copilot-acp/__init__.py +++ b/plugins/model-providers/copilot-acp/__init__.py @@ -1,11 +1,9 @@ """GitHub Copilot ACP provider profile. copilot-acp does not speak OpenAI-over-HTTP: it drives an external ACP -subprocess over stdio. The profile therefore supplies its own client through -:meth:`ProviderProfile.create_client` instead of letting the core build an -``openai.OpenAI``. That hook is the registration seam — this profile is its -in-tree consumer, and an out-of-tree ACP provider registered from -``~/.hermes/plugins/model-providers/`` or a pip entry point uses the exact same +subprocess over stdio, so the profile supplies its own client via +:meth:`ProviderProfile.create_client`. An out-of-tree ACP provider registered +from ``~/.hermes/plugins/model-providers/`` or a pip entry point uses the same three lines without touching core. """ @@ -25,11 +23,7 @@ class CopilotACPProfile(ProviderProfile): return CopilotACPClient(**client_kwargs) def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: """Model listing is handled by the ACP subprocess.""" return None diff --git a/plugins/model-providers/copilot/__init__.py b/plugins/model-providers/copilot/__init__.py index f8b5912781..903ef2b3f9 100644 --- a/plugins/model-providers/copilot/__init__.py +++ b/plugins/model-providers/copilot/__init__.py @@ -1,13 +1,8 @@ """Copilot / GitHub Models provider profile. -Copilot uses per-model api_mode routing: - - GPT-5+ / Codex models → codex_responses - - Claude models → anthropic_messages - - Everything else → chat_completions (this profile covers that subset) - -Key quirks for the chat_completions subset: - - Editor attribution headers (via copilot_default_headers()) - - GitHub Models reasoning extra_body (model-catalog gated) +Core routes GPT-5+/Codex -> codex_responses and Claude -> anthropic_messages; +this profile covers the chat_completions remainder: editor attribution headers +(copilot_default_headers()) and catalog-gated GitHub Models reasoning. """ from typing import Any @@ -27,46 +22,28 @@ class CopilotProfile(ProviderProfile): supports_reasoning: bool = False, **ctx, ) -> tuple[dict[str, Any], dict[str, Any]]: - extra_body: dict[str, Any] = {} - if supports_reasoning and model: - try: - from hermes_cli.models import github_model_reasoning_efforts + if not (supports_reasoning and model): + return {}, {} + try: + from hermes_cli.models import clamp_reasoning_effort_to_supported, github_model_reasoning_efforts - supported_efforts = github_model_reasoning_efforts(model) - if supported_efforts and reasoning_config: - effort = reasoning_config.get("effort", "medium") - # Honor the requested level when the live Copilot catalog - # lists it as supported: gpt-5.5/gpt-5.4 DO support - # ``xhigh``. Otherwise clamp to the nearest WEAKER - # supported level via the shared ladder helper — the old - # ad-hoc rules dropped everything unrecognized to - # ``medium``, which inverted the ladder: ``ultra`` (the - # strongest ask) resolved weaker than an explicit - # ``high`` (#74295). - if effort not in supported_efforts: - from hermes_cli.models import ( - clamp_reasoning_effort_to_supported, - ) - - effort = clamp_reasoning_effort_to_supported( - effort, list(supported_efforts) - ) - if effort not in supported_efforts: - # Unrecognized/bespoke level the ladder can't - # place — fall back to medium, then to the - # catalog's first entry. - effort = ( - "medium" - if "medium" in supported_efforts - else supported_efforts[0] - ) - if effort in supported_efforts: - extra_body["reasoning"] = {"effort": effort} - elif supported_efforts: - extra_body["reasoning"] = {"effort": "medium"} - except Exception: - pass - return extra_body, {} + supported = github_model_reasoning_efforts(model) + if not supported: + return {}, {} + if not reasoning_config: + return {"reasoning": {"effort": "medium"}}, {} + effort = reasoning_config.get("effort", "medium") + # Honor a level the live catalog lists; otherwise clamp to the nearest + # WEAKER supported level (never drop straight to medium, which inverted + # the ladder: ultra < high). Bespoke levels the ladder can't place fall + # to medium (or the first supported level). + if effort not in supported: + effort = clamp_reasoning_effort_to_supported(effort, list(supported)) + if effort not in supported: + effort = "medium" if "medium" in supported else supported[0] + return {"reasoning": {"effort": effort}}, {} + except Exception: + return {}, {} copilot = CopilotProfile( diff --git a/plugins/model-providers/custom/__init__.py b/plugins/model-providers/custom/__init__.py index 80ad699dec..7260096833 100644 --- a/plugins/model-providers/custom/__init__.py +++ b/plugins/model-providers/custom/__init__.py @@ -1,127 +1,65 @@ -"""Custom / Ollama (local) provider profile. - -Covers any endpoint registered as provider="custom", including local -Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on -Volcengine ARK, vLLM, llama.cpp). Key quirks: - - ollama_num_ctx → extra_body.options.num_ctx (local context window) - - reasoning_config disabled → top-level reasoning_effort="none" - (Ollama /v1/chat/completions ignores think=False — ollama#14820) - + extra_body.think = False only on Ollama URLs (/api/chat and proxies) - - reasoning_config enabled + effort → top-level reasoning_effort - (the native OpenAI-compatible format GLM/ARK expect; unset omits it - so the endpoint's server default applies) -""" +"""Custom / Ollama (local) provider profile: any endpoint registered as +provider="custom" (Ollama, vLLM, llama.cpp, GLM-5.2 on ARK, …).""" from typing import Any from urllib.parse import urlparse +from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort from providers import register_provider from providers.base import ProviderProfile def _looks_like_ollama_endpoint(base_url: str | None) -> bool: - """True when ``base_url`` is an Ollama host, not a generic OpenAI-compat relay. + """True only for explicit Ollama signatures (port 11434 or an ``ollama`` host label). - ``think`` is an Ollama-native extra_body field. Strict hosts (Mistral - ``extra=forbid``, Groq, …) reject it with HTTP 422. Match only explicit - Ollama signatures — default port 11434, or ``ollama`` as a hostname - label — not arbitrary localhost (llama.cpp / vLLM / LM Studio). + ``think`` is Ollama-native; strict hosts (Mistral, Groq) 422 on it, and + arbitrary localhost may be llama.cpp / vLLM / LM Studio. """ raw = (base_url or "").strip() if not raw: return False parsed = urlparse(raw if "://" in raw else f"//{raw}") - # urlparse raises ValueError for non-integer / out-of-range ports - # ("http://host:99999/v1" parses fine in the OpenAI client, so the URL - # is reachable here). Treat a malformed port as "not Ollama" instead of - # killing the whole kwargs build — same try/except shape the 11434 - # check in hermes_cli/models.py uses, not the same detection logic. + # urlparse raises ValueError on malformed ports ("host:99999"); treat as not-Ollama. try: if parsed.port == 11434: return True except ValueError: return False host = (parsed.hostname or "").lower().rstrip(".") - if not host: - return False - if host == "ollama.com" or host.endswith(".ollama.com"): - return True - return "ollama" in host.split(".") + return bool(host) and (host == "ollama.com" or host.endswith(".ollama.com") or "ollama" in host.split(".")) class CustomProfile(ProviderProfile): """Custom/Ollama local provider — think=false and num_ctx support.""" def build_api_kwargs_extras( - self, - *, - reasoning_config: dict | None = None, - ollama_num_ctx: int | None = None, - **ctx: Any, + self, *, reasoning_config: dict | None = None, ollama_num_ctx: int | None = None, **ctx: Any ) -> tuple[dict[str, Any], dict[str, Any]]: extra_body: dict[str, Any] = {} top_level: dict[str, Any] = {} - - # Ollama context window if ollama_num_ctx: - options = extra_body.get("options", {}) - options["num_ctx"] = ollama_num_ctx - extra_body["options"] = options + extra_body["options"] = {"num_ctx": ollama_num_ctx} - # Reasoning / thinking control for custom OpenAI-compatible endpoints - # (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …). - # - # - disabled → top-level reasoning_effort="none"; extra_body.think - # = False only on Ollama URLs (Ollama's thinking-off flag) - # - enabled + effort set → TOP-LEVEL reasoning_effort string, the - # format GLM-5.2/ARK and other OpenAI-compatible reasoning APIs - # expect (GLM documents "high" and "max"; "max" is its default). - # - enabled + no effort → omit both, so the endpoint applies its own - # server-side default (do NOT force a level the user didn't pick). - # - # We deliberately do NOT emit ``think=True`` on enable: it is an - # Ollama-only flag and thinking is already server-default-on for these - # backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't - # recognize it. Mirrors the DeepSeek/Zai profile precedent. The same - # constraint applies to ``think=False`` on disable — Mistral/Groq - # reject unknown fields (HTTP 422 extra_forbidden) rather than ignoring - # them, so that flag stays Ollama-URL-gated. + # disabled -> top-level reasoning_effort="none" (Ollama's /v1 ignores + # extra_body.think) plus think=False only on Ollama URLs; enabled+effort -> + # top-level reasoning_effort clamped to the OpenAI-compat wire (GLM/ARK, + # vLLM and SGLang all top out at "max"; "ultra" verbatim 400s); enabled + # without effort -> omit so the server default applies. Never emit + # think=True (Ollama-only flag). if reasoning_config and isinstance(reasoning_config, dict): - _effort = (reasoning_config.get("effort") or "").strip().lower() - _enabled = reasoning_config.get("enabled", True) - if _effort == "none" or _enabled is False: - # Ollama's /v1/chat/completions silently ignores - # extra_body.think (only /api/chat honours it — ollama#14820) - # but respects the top-level reasoning_effort field (#25758). - # Always emit reasoning_effort="none"; only add think=False - # when the URL is actually Ollama. + effort = (reasoning_config.get("effort") or "").strip().lower() + if effort == "none" or reasoning_config.get("enabled", True) is False: top_level["reasoning_effort"] = "none" if _looks_like_ollama_endpoint(ctx.get("base_url")): extra_body["think"] = False - elif _effort: - # Clamp the internal ladder onto the widest OpenAI-compatible - # wire vocabulary (shared policy in agent.reasoning_effort) — - # GLM/ARK, vLLM and SGLang all top out at "max"; forwarding - # "ultra" verbatim is a guaranteed 400 (#89503). - from agent.reasoning_effort import ( - OPENAI_COMPAT_WIRE_EFFORTS, - clamp_effort, - ) - - top_level["reasoning_effort"] = clamp_effort( - _effort, OPENAI_COMPAT_WIRE_EFFORTS - ) - + elif effort: + top_level["reasoning_effort"] = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) return extra_body, top_level def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: - """Custom/Ollama: base_url is user-configured; fetch if set.""" + """base_url is user-configured; fetch only if set.""" if not (base_url or self.base_url): return None return super().fetch_models(api_key=api_key, base_url=base_url, timeout=timeout) @@ -129,20 +67,11 @@ class CustomProfile(ProviderProfile): custom = CustomProfile( name="custom", - aliases=( - "ollama", - "local", - "vllm", - "llamacpp", - "llama.cpp", - "llama-cpp", - ), + aliases=("ollama", "local", "vllm", "llamacpp", "llama.cpp", "llama-cpp"), env_vars=(), # No fixed key — custom endpoint base_url="", # User-configured - # Without this, no max_tokens is sent and Ollama falls back to its internal - # num_predict=128, truncating responses after a few tokens (#39281). This is - # only a floor used when the user hasn't set model.max_tokens — they can - # override per-model — so we set it generously rather than lowballing it. + # Floor only (user model.max_tokens overrides); without it Ollama falls + # back to num_predict=128 and truncates. default_max_tokens=65536, ) diff --git a/plugins/model-providers/deepinfra/__init__.py b/plugins/model-providers/deepinfra/__init__.py index 57c6a9c4ce..f57077f43b 100644 --- a/plugins/model-providers/deepinfra/__init__.py +++ b/plugins/model-providers/deepinfra/__init__.py @@ -1,33 +1,20 @@ -"""DeepInfra provider profile. - -DeepInfra is an OpenAI-compatible inference gateway that hosts 100+ open -models (Step, GLM, Kimi, DeepSeek, MiniMax, Nemotron, Mistral, Qwen, …) as -well as image-gen / TTS / STT / embedding endpoints. The chat surface is -wired in through this profile; non-chat surfaces are wired in through -their respective plugin subsystems (``plugins/image_gen/deepinfra`` and -the TTS/STT dispatchers in ``tools/``). -""" +"""DeepInfra provider profile (chat surface; image-gen/TTS/STT are wired via +their own plugin subsystems).""" from providers import register_provider from providers.base import ProviderProfile class _DeepInfraProfile(ProviderProfile): - """DeepInfra profile with live vision-default discovery. - - Owns its own vision default so shared vision resolution in - ``agent/auxiliary_client.py`` stays provider-agnostic (a - ``default_vision_model()`` hook call instead of an ``if provider == - "deepinfra"`` branch reaching into the catalog helpers). - """ + """DeepInfra profile with live vision-default discovery, so shared vision + resolution in ``agent/auxiliary_client.py`` stays provider-agnostic.""" def default_vision_model(self): # type: ignore[override] """First vision-capable *chat* model from the live catalog, or None. - Key-gated so a box without ``DEEPINFRA_API_KEY`` never pays the - catalog round-trip. Requires the ``chat`` surface tag (not just the - ``vision`` capability) so an image-gen/edit model that merely carries - a ``vision`` tag can't be picked as a chat-completions vision backend. + Key-gated so a box without DEEPINFRA_API_KEY never pays the round-trip. + Requires the ``chat`` surface tag so an image-gen model carrying a + ``vision`` tag can't be picked as a chat-completions vision backend. """ from agent.secret_scope import get_secret @@ -41,10 +28,8 @@ class _DeepInfraProfile(ProviderProfile): for item in items or []: metadata = item.get("metadata") or {} tags = metadata.get("tags") if isinstance(metadata, dict) else None - if isinstance(tags, list) and "vision" in tags: - model_id = item.get("id") - if model_id: - return model_id + if isinstance(tags, list) and "vision" in tags and item.get("id"): + return item["id"] return None @@ -57,24 +42,13 @@ deepinfra = _DeepInfraProfile( env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"), base_url="https://api.deepinfra.com/v1/openai", auth_type="api_key", - # The catalog spans models with different output limits. Omitting a - # provider-wide default lets DeepInfra apply its documented per-model cap; - # an explicit user ``agent.max_tokens`` still passes through normally. + # No provider-wide cap: DeepInfra applies its documented per-model limit. default_max_tokens=None, - # Auxiliary model — cheap/fast chat model the same provider uses for - # side tasks (context compression, session search, web extract, - # vision). This is the *only* hardcoded DeepInfra model in the - # integration: aux resolution is synchronous (no time for a catalog - # round-trip on every agent turn), so we need one explicit choice. - # Every other surface (chat picker, image-gen, tts, stt, pricing) - # discovers models live from - # ``api.deepinfra.com/v1/openai/models?filter=true&sort_by=hermes``. + # The only hardcoded DeepInfra model: aux resolution is synchronous, so it + # can't wait on a catalog round-trip. Everything else is discovered live. default_aux_model="deepseek-ai/DeepSeek-V4-Flash", - # ``fallback_models`` deliberately empty — the live catalog at - # ``hermes_cli/models.py::_fetch_deepinfra_models`` is the source of - # truth. When the live fetch fails (network/DNS), the picker shows - # no options, which is preferable to silently routing the user to a - # model that may have been retired upstream. + # Empty on purpose: the live catalog is the source of truth; an empty picker + # beats silently routing to a retired model. fallback_models=(), ) diff --git a/plugins/model-providers/deepseek/__init__.py b/plugins/model-providers/deepseek/__init__.py index 1273f51ed6..b669f66595 100644 --- a/plugins/model-providers/deepseek/__init__.py +++ b/plugins/model-providers/deepseek/__init__.py @@ -1,96 +1,45 @@ """DeepSeek provider profile. -DeepSeek's V4 family defaults to thinking-mode ON when ``extra_body.thinking`` -is unset. The API then returns ``reasoning_content`` and starts enforcing -the contract that subsequent turns echo it back; combined with how Hermes -replays history this lands on the notorious HTTP 400 -``reasoning_content must be passed back`` error after the first tool call -(#15700, #17212, #17825). - -This profile overrides :meth:`build_api_kwargs_extras` to mirror the Kimi / -Moonshot wire shape that DeepSeek's OpenAI-compat endpoint expects: - - {"reasoning_effort": "", - "extra_body": {"thinking": {"type": "enabled" | "disabled"}}} - -Non-thinking models (``deepseek-v3-*`` variants) are left as no-ops so we -don't perturb the V3 wire format. - -The legacy aliases ``deepseek-chat`` / ``deepseek-reasoner`` were retired on -2026-07-24. Use ``deepseek-v4-flash`` or ``deepseek-v4-pro``; Hermes remaps -the retired IDs in ``hermes_cli.model_normalize``. +V4 defaults to thinking ON when ``extra_body.thinking`` is unset, and then +requires ``reasoning_content`` to be echoed back on later turns (HTTP 400 after +the first tool call otherwise). This profile sets ``thinking`` explicitly and +maps effort onto DeepSeek's ``reasoning_effort``; V3 models are left untouched. +Retired ``deepseek-chat``/``deepseek-reasoner`` IDs are remapped in +``hermes_cli.model_normalize`` before reaching here. """ from __future__ import annotations from typing import Any +from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, clamp_effort from providers import register_provider from providers.base import ProviderProfile -def _model_supports_thinking(model: str | None) -> bool: - """DeepSeek thinking-capable model families. - - Currently covers the V4 family (``deepseek-v4-pro``, ``deepseek-v4-flash``, - and any future ``deepseek-v4-*`` variants). Retired aliases are remapped - before requests leave Hermes, so they are not listed here. - """ - m = (model or "").strip().lower() - if not m: - return False - if m.startswith("deepseek-v") and not m.startswith("deepseek-v3"): - # deepseek-v4-*, deepseek-v5-*, etc. — every V4+ generation has - # thinking. v3 explicitly excluded. - return True - return False - - class DeepSeekProfile(ProviderProfile): """DeepSeek — extra_body.thinking + top-level reasoning_effort.""" def build_api_kwargs_extras( self, *, reasoning_config: dict | None = None, model: str | None = None, **context ) -> tuple[dict[str, Any], dict[str, Any]]: - extra_body: dict[str, Any] = {} + m = (model or "").strip().lower() + # deepseek-v4-* and every later generation; v3 explicitly excluded. + if not m.startswith("deepseek-v") or m.startswith("deepseek-v3"): + return {}, {} + rc = reasoning_config if isinstance(reasoning_config, dict) else None + # Always set explicitly (default enabled, matching the API default) to + # avoid the reasoning_content echo trap on subsequent turns. + if rc is not None and rc.get("enabled") is False: + return {"thinking": {"type": "disabled"}}, {} top_level: dict[str, Any] = {} - - if not _model_supports_thinking(model): - # V3 / unknown — leave wire format untouched, current behavior. - return extra_body, top_level - - # Determine enabled/disabled. Default is enabled to match DeepSeek's - # API default; the API requires this to be set explicitly to avoid the - # reasoning_content echo trap on subsequent turns. - enabled = True - if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False: - enabled = False - - extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"} - - if not enabled: - return extra_body, top_level - - # Effort mapping via the shared vocabulary in agent.reasoning_effort - # (DeepSeek V4: low/medium/high/max, xhigh rounds up to max). When no - # effort is set we omit reasoning_effort so DeepSeek applies its - # server default (currently high). - if isinstance(reasoning_config, dict): - from agent.reasoning_effort import ( - DEEPSEEK_V4_EFFORTS, - DEEPSEEK_V4_OVERRIDES, - clamp_effort, - ) - - effort = (reasoning_config.get("effort") or "").strip().lower() - if effort and effort != "none": - clamped = clamp_effort( - effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES - ) - if clamped in DEEPSEEK_V4_EFFORTS: - top_level["reasoning_effort"] = clamped - - return extra_body, top_level + # No effort -> omit reasoning_effort so DeepSeek applies its server default. + effort = (rc.get("effort") or "").strip().lower() if rc is not None else "" + if effort and effort != "none": + clamped = clamp_effort(effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES) + if clamped in DEEPSEEK_V4_EFFORTS: + top_level["reasoning_effort"] = clamped + return {"thinking": {"type": "enabled"}}, top_level deepseek = DeepSeekProfile( @@ -100,10 +49,7 @@ deepseek = DeepSeekProfile( display_name="DeepSeek", description="DeepSeek — native DeepSeek API", signup_url="https://platform.deepseek.com/", - fallback_models=( - "deepseek-v4-pro", - "deepseek-v4-flash", - ), + fallback_models=("deepseek-v4-pro", "deepseek-v4-flash"), base_url="https://api.deepseek.com/v1", default_aux_model="deepseek-v4-flash", ) diff --git a/plugins/model-providers/fireworks/__init__.py b/plugins/model-providers/fireworks/__init__.py index 338058e0bc..92a39c9cd8 100644 --- a/plugins/model-providers/fireworks/__init__.py +++ b/plugins/model-providers/fireworks/__init__.py @@ -1,13 +1,5 @@ -"""Fireworks AI provider profile. - -Fireworks AI serves fast, production-grade inference for open and proprietary -models through an OpenAI-compatible chat-completions endpoint. - -Address models directly by their catalog ID, e.g. -``accounts/fireworks/models/kimi-k2p6`` or ``accounts/fireworks/models/glm-5p2``. -Model IDs here track the canonical Fireworks catalog (fw-ai/fireconnect -``setup-cli``). -""" +"""Fireworks AI provider profile. Models are addressed by full catalog ID +(``accounts/fireworks/models/``), tracking fw-ai/fireconnect ``setup-cli``.""" from hermes_cli import __version__ as _HERMES_VERSION from providers import register_provider @@ -23,19 +15,15 @@ fireworks = ProviderProfile( env_vars=("FIREWORKS_API_KEY",), base_url="https://api.fireworks.ai/inference/v1", auth_type="api_key", - # Attribution headers sent on every Fireworks request. Values match the - # canonical Hermes set in agent/auxiliary_client.py. Applied through the - # generic profile.default_headers path, so they survive switch_model and - # credential rotation. + # Attribution headers (canonical Hermes set); via default_headers so they + # survive switch_model and credential rotation. default_headers={ "HTTP-Referer": "https://hermes-agent.nousresearch.com", "X-Title": "Hermes Agent", "User-Agent": f"HermesAgent/{_HERMES_VERSION}", }, - # Auxiliary model for cheap tasks (compaction, title generation, vision). - # A standard pay-as-you-go catalog ``/models/`` ID. default_aux_model="accounts/fireworks/models/glm-5p2", - # Curated safety net shown in the picker when the live catalog fetch fails. + # Picker safety net when the live catalog fetch fails. fallback_models=( "accounts/fireworks/models/kimi-k2p6", "accounts/fireworks/models/glm-5p2", diff --git a/plugins/model-providers/gemini/__init__.py b/plugins/model-providers/gemini/__init__.py index b136fb2434..baea33c4dd 100644 --- a/plugins/model-providers/gemini/__init__.py +++ b/plugins/model-providers/gemini/__init__.py @@ -1,12 +1,7 @@ -"""Google Gemini provider profiles. +"""Google Gemini (AI Studio) provider profile. -gemini: Google AI Studio (API key) — uses GeminiNativeClient - -Reports api_mode="chat_completions" but uses a custom native client -that bypasses the standard OpenAI transport. The profile captures auth -and endpoint metadata for auth.py / runtime_provider.py migration, and -carries the thinking_config translation hook so the transport's profile -path produces the same extra_body shape the legacy flag path did. +Reports api_mode="chat_completions" but runs on GeminiNativeClient; this +profile carries auth/endpoint metadata and the thinking_config translation hook. """ from typing import Any @@ -18,34 +13,22 @@ from providers.base import ProviderProfile class GeminiProfile(ProviderProfile): """Gemini — translate reasoning_config to thinking_config in extra_body.""" - def build_extra_body( - self, *, session_id: str | None = None, **context: Any - ) -> dict[str, Any]: - """Emit extra_body.thinking_config (native) or extra_body.extra_body.google.thinking_config - (OpenAI-compat /openai subpath), mirroring the legacy path's behavior. - """ + def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]: + """Native: ``thinking_config``; OpenAI-compat /openai subpath: + ``extra_body.google.thinking_config`` (snake_case).""" from agent.transports.chat_completions import ( _build_gemini_thinking_config, _is_gemini_openai_compat_base_url, _snake_case_gemini_thinking_config, ) - model = context.get("model") or "" - reasoning_config = context.get("reasoning_config") - base_url = context.get("base_url") or self.base_url - - raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config) - if not raw_thinking_config: + raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config")) + if not raw: return {} - - body: dict[str, Any] = {} - if self.name == "gemini" and _is_gemini_openai_compat_base_url(base_url): - thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config) - if thinking_config: - body["extra_body"] = {"google": {"thinking_config": thinking_config}} - else: - body["thinking_config"] = raw_thinking_config - return body + if self.name == "gemini" and _is_gemini_openai_compat_base_url(context.get("base_url") or self.base_url): + thinking_config = _snake_case_gemini_thinking_config(raw) + return {"extra_body": {"google": {"thinking_config": thinking_config}}} if thinking_config else {} + return {"thinking_config": raw} gemini = GeminiProfile( diff --git a/plugins/model-providers/gmi/__init__.py b/plugins/model-providers/gmi/__init__.py index bf7dfb9168..307b657a53 100644 --- a/plugins/model-providers/gmi/__init__.py +++ b/plugins/model-providers/gmi/__init__.py @@ -13,9 +13,7 @@ gmi = ProviderProfile( env_vars=("GMI_API_KEY", "GMI_BASE_URL"), base_url="https://api.gmi-serving.com/v1", auth_type="api_key", - # Attribution so GMI can identify traffic from Hermes Agent. - # The generic profile.default_headers fallback in run_agent.py and - # agent/auxiliary_client.py picks this up at client construction time. + # Attribution so GMI can identify Hermes Agent traffic. default_headers={"User-Agent": f"HermesAgent/{_HERMES_VERSION}"}, default_aux_model="google/gemini-3.1-flash-lite-preview", fallback_models=( diff --git a/plugins/model-providers/huggingface/__init__.py b/plugins/model-providers/huggingface/__init__.py index 039d5a1319..7fcc4b32bc 100644 --- a/plugins/model-providers/huggingface/__init__.py +++ b/plugins/model-providers/huggingface/__init__.py @@ -10,10 +10,7 @@ huggingface = ProviderProfile( display_name="HuggingFace", description="HuggingFace Inference API", signup_url="https://huggingface.co/settings/tokens", - fallback_models=( - "Qwen/Qwen3.5-72B-Instruct", - "deepseek-ai/DeepSeek-V3.2", - ), + fallback_models=("Qwen/Qwen3.5-72B-Instruct", "deepseek-ai/DeepSeek-V3.2"), base_url="https://router.huggingface.co/v1", ) diff --git a/plugins/model-providers/kimi-coding/__init__.py b/plugins/model-providers/kimi-coding/__init__.py index 23b33417a2..1e9525cdc1 100644 --- a/plugins/model-providers/kimi-coding/__init__.py +++ b/plugins/model-providers/kimi-coding/__init__.py @@ -1,36 +1,32 @@ -"""Kimi / Moonshot provider profiles. - -Kimi has dual endpoints: - - sk-kimi-* keys → api.kimi.com/coding (Anthropic Messages API) - - legacy keys → api.moonshot.ai/v1 (OpenAI chat completions) - -This module covers the chat_completions path (/v1 endpoint). -""" +"""Kimi / Moonshot provider profiles (chat_completions path; sk-kimi-* keys are +redirected to api.kimi.com/coding by core).""" from typing import Any from urllib.parse import urlparse +from agent.reasoning_effort import KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, clamp_effort, requested_effort from hermes_cli import __version__ as _HERMES_VERSION from providers import register_provider from providers.base import OMIT_TEMPERATURE, ProviderProfile +_HEADERS = { + "HTTP-Referer": "https://hermes-agent.nousresearch.com", + "X-Title": "Hermes Agent", + "User-Agent": f"HermesAgent/{_HERMES_VERSION}", +} + def _is_confirmed_kimi_coding_url(base_url: str) -> bool: """Return True only for Kimi Code's canonical HTTPS API surfaces.""" try: - parsed = urlparse(base_url) - port = parsed.port + p = urlparse(base_url) + port = p.port except ValueError: return False return ( - parsed.scheme.lower() == "https" - and (parsed.hostname or "").lower() == "api.kimi.com" - and port in (None, 443) - and parsed.username is None - and parsed.password is None - and parsed.path.rstrip("/") in {"/coding", "/coding/v1"} - and not parsed.query - and not parsed.fragment + p.scheme.lower() == "https" and (p.hostname or "").lower() == "api.kimi.com" and port in (None, 443) + and p.username is None and p.password is None + and p.path.rstrip("/") in {"/coding", "/coding/v1"} and not p.query and not p.fragment ) @@ -38,22 +34,15 @@ class KimiProfile(ProviderProfile): """Kimi/Moonshot — temperature omitted, thinking xor reasoning_effort.""" def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: - """Use Kimi Code's OpenAI-compatible surface for model discovery.""" + """Use Kimi Code's OpenAI-compatible surface for model discovery; the bare + ``k3`` slug is only served there, so it is filtered off other endpoints.""" effective_base = (base_url or self.base_url or "").rstrip("/") confirmed_coding_endpoint = _is_confirmed_kimi_coding_url(effective_base) if confirmed_coding_endpoint and urlparse(effective_base).path.rstrip("/") == "/coding": effective_base += "/v1" - models = super().fetch_models( - api_key=api_key, - base_url=effective_base or None, - timeout=timeout, - ) + models = super().fetch_models(api_key=api_key, base_url=effective_base or None, timeout=timeout) if models is None or confirmed_coding_endpoint: return models return [model for model in models if model.strip().lower() != "k3"] @@ -61,53 +50,15 @@ class KimiProfile(ProviderProfile): def build_api_kwargs_extras( self, *, reasoning_config: dict | None = None, **context ) -> tuple[dict[str, Any], dict[str, Any]]: - """Kimi reasoning controls. - - Moonshot's wire shape treats ``extra_body.thinking`` (a binary toggle) - and a top-level ``reasoning_effort`` as mutually exclusive — sending - both is at best redundant and risks "cannot specify both 'thinking' and - 'reasoning_effort'" (HTTP 400). This mirrors the kimi-k2 handling on the - opencode-go relay: send effort when one is requested, otherwise fall - back to ``extra_body.thinking`` — never both. - """ - extra_body = {} - top_level = {} - - if not reasoning_config or not isinstance(reasoning_config, dict): - # No config → thinking enabled, let the server pick the depth. - # (Previously also sent reasoning_effort="medium", which paired - # thinking + effort on every default call.) - extra_body["thinking"] = {"type": "enabled"} - return extra_body, top_level - - enabled = reasoning_config.get("enabled", True) - if enabled is False: - extra_body["thinking"] = {"type": "disabled"} - return extra_body, top_level - - # Enabled: prefer an explicit effort; only fall back to extra_body - # thinking when no recognized effort is requested. - # K3's vocabulary (low/high/max, default high) and its documented - # rounding (medium→high, xhigh→max) are declared in - # agent.reasoning_effort — shared with the chat-completions - # transport's Kimi path so both stay in sync. - from agent.reasoning_effort import ( - KIMI_K3_EFFORTS, - KIMI_K3_OVERRIDES, - clamp_effort, - ) - - effort = (reasoning_config.get("effort") or "").strip().lower() - if effort and effort != "none": - k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES) - else: - k3_effort = None + """Moonshot treats extra_body.thinking and reasoning_effort as mutually + exclusive (400 on both): send effort when requested, else the toggle.""" + if isinstance(reasoning_config, dict) and reasoning_config.get("enabled", True) is False: + return {"thinking": {"type": "disabled"}}, {} + effort = requested_effort(reasoning_config) + k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES) if effort != "none" else None if k3_effort in KIMI_K3_EFFORTS: - top_level["reasoning_effort"] = k3_effort - else: - extra_body["thinking"] = {"type": "enabled"} - - return extra_body, top_level + return {}, {"reasoning_effort": k3_effort} + return {"thinking": {"type": "enabled"}}, {} kimi = KimiProfile( @@ -117,11 +68,7 @@ kimi = KimiProfile( base_url="https://api.moonshot.ai/v1", fixed_temperature=OMIT_TEMPERATURE, default_max_tokens=32000, - default_headers={ - "HTTP-Referer": "https://hermes-agent.nousresearch.com", - "X-Title": "Hermes Agent", - "User-Agent": f"HermesAgent/{_HERMES_VERSION}", - }, + default_headers=dict(_HEADERS), default_aux_model="kimi-k2-turbo-preview", ) @@ -132,11 +79,7 @@ kimi_cn = KimiProfile( base_url="https://api.moonshot.cn/v1", fixed_temperature=OMIT_TEMPERATURE, default_max_tokens=32000, - default_headers={ - "HTTP-Referer": "https://hermes-agent.nousresearch.com", - "X-Title": "Hermes Agent", - "User-Agent": f"HermesAgent/{_HERMES_VERSION}", - }, + default_headers=dict(_HEADERS), default_aux_model="kimi-k2-turbo-preview", ) diff --git a/plugins/model-providers/meta-ai/__init__.py b/plugins/model-providers/meta-ai/__init__.py index de2a3847f5..f349f8461f 100644 --- a/plugins/model-providers/meta-ai/__init__.py +++ b/plugins/model-providers/meta-ai/__init__.py @@ -1,29 +1,9 @@ -"""Meta Model API (Muse Spark) provider plugin for Hermes Agent. +"""Meta Model API (Muse Spark) provider profile — https://api.meta.ai/v1. -Provider profile for Meta Superintelligence Labs' Muse Spark family, served -via the OpenAI-compatible Meta Model API at ``https://api.meta.ai/v1``. - -Bundled from https://github.com/albertodepaola/hermes-meta-provider. Hermes' -provider discovery (``providers/__init__.py``) imports it on first -``get_provider_profile()`` / ``list_providers()`` call, and the module-level -``register_provider()`` below wires it into the registry. - -Design notes ------------- -* **Zero core edits.** Everything rides on ``ProviderProfile`` hooks. No changes - to hermes' ``model_metadata.py`` / ``models.py`` / ``run_agent.py`` are needed: - - Context window (1M), reasoning and vision capabilities already resolve from - models.dev for the muse-spark family, so no static ctx table entry is required. - - The reasoning dial is emitted as a **top-level ``reasoning_effort``** kwarg - (returned in the ``top_level`` slot of ``build_api_kwargs_extras``), which the - chat-completions transport merges unconditionally. This deliberately avoids - the ``extra_body.reasoning`` path, whose emission is gated by a hardcoded - host allowlist in core (``AIAgent._supports_reasoning_extra_body``) that a - third-party plugin must not edit. -* **Meta 400 on ``reasoning_effort: "none"``.** Muse rejects ``none``; disabling - reasoning maps to ``"minimal"`` instead. -* **``default_max_tokens=16384``.** Muse spends completion budget on hidden - reasoning tokens first; small caps can finish with empty content. +Bundled from albertodepaola/hermes-meta-provider; rides entirely on +ProviderProfile hooks (zero core edits). The reasoning dial is emitted as a +top-level ``reasoning_effort`` kwarg — not ``extra_body.reasoning``, whose +emission is gated by a core host allowlist a third-party plugin must not edit. """ from __future__ import annotations @@ -31,30 +11,11 @@ from __future__ import annotations import os from typing import Any +from agent.reasoning_effort import META_AI_EFFORTS, clamp_effort from providers import register_provider from providers.base import ProviderProfile -def _resolve_effort(reasoning_config: dict | None) -> str: - """Map Hermes' reasoning_config to a Meta-safe ``reasoning_effort`` value. - - Meta's vocabulary (minimal..xhigh; rejects ``none``) is declared in - agent.reasoning_effort. Disabled/"none" maps to ``minimal`` (the closest - Meta has to off); unset/bespoke levels fall to ``medium``. - """ - rc = reasoning_config or {} - if rc.get("enabled") is False: - return "minimal" - effort = str(rc.get("effort") or "").strip().lower() - if effort in {"", "none"}: - return "minimal" if effort == "none" else "medium" - - from agent.reasoning_effort import META_AI_EFFORTS, clamp_effort - - clamped = clamp_effort(effort, META_AI_EFFORTS) - return clamped if clamped in META_AI_EFFORTS else "medium" - - class MetaAIProfile(ProviderProfile): """Meta Model API — top-level reasoning_effort, self-contained.""" @@ -65,19 +26,20 @@ class MetaAIProfile(ProviderProfile): supports_reasoning: bool = False, # noqa: ARG002 — we self-gate below **context: Any, ) -> tuple[dict[str, Any], dict[str, Any]]: - """Emit ``reasoning_effort`` as a top-level api kwarg. + """Ignores the core ``supports_reasoning`` gate (host-allowlist driven); + Muse Spark always accepts ``reasoning_effort``. - We ignore the core ``supports_reasoning`` gate on purpose: that flag is - driven by a host allowlist in core we cannot (and should not) edit from - an out-of-tree plugin. Muse Spark always accepts ``reasoning_effort``, - so we resolve it from ``reasoning_config`` directly. + Muse 400s on ``none``: disabled/"none" -> ``minimal`` (closest to off); + unset/bespoke levels -> ``medium``. """ - return {}, {"reasoning_effort": _resolve_effort(reasoning_config)} - - -def _base_url() -> str: - """Allow a base-URL override via ``META_BASE_URL`` without editing config.""" - return os.getenv("META_BASE_URL", "").strip() or "https://api.meta.ai/v1" + rc = reasoning_config or {} + effort = str(rc.get("effort") or "").strip().lower() + if rc.get("enabled") is False or effort == "none": + mapped = "minimal" + else: + clamped = clamp_effort(effort, META_AI_EFFORTS) + mapped = clamped if clamped in META_AI_EFFORTS else "medium" + return {}, {"reasoning_effort": mapped} meta_ai = MetaAIProfile( @@ -88,30 +50,18 @@ meta_ai = MetaAIProfile( signup_url="https://developer.meta.com/ai/", # MODEL_API_KEY is Meta's documented env var; the aliases are conveniences. env_vars=("MODEL_API_KEY", "META_API_KEY", "META_MODEL_API_KEY", "META_BASE_URL"), - base_url=_base_url(), + base_url=os.getenv("META_BASE_URL", "").strip() or "https://api.meta.ai/v1", auth_type="api_key", - # Responses API is the wire that engages Muse prompt caching: measured - # 0 cached tokens on /v1/chat/completions vs 93-99% cache hits on - # /v1/responses with prompt_cache_retention (see host_mandated_api_mode - # in hermes_cli/providers.py and the retention hint in - # agent/transports/codex.py). The MetaAIProfile chat-completions hook - # above still covers custom OpenAI-compatible endpoints configured with - # a non-api.meta.ai base URL, which fall through to chat_completions. + # Responses API engages Muse prompt caching (0 cached tokens on + # chat/completions vs 93-99% hits on /v1/responses); the chat-completions + # hook above still covers custom non-api.meta.ai base URLs. api_mode="codex_responses", - # Muse Spark is natively multimodal (image/video/pdf/audio in, text out). supports_vision=True, - # Cheap contributor tier is a good default for auxiliary tasks - # (compaction, title generation, vision) when this is the main provider. default_aux_model="muse-spark-1.2-contributor", - # Muse spends completion budget on hidden reasoning tokens first; a low cap - # can finish with empty content. 16k is a safe floor. + # Muse spends completion budget on hidden reasoning first; low caps can + # finish with empty content. default_max_tokens=16384, - # Curated safety net shown in the picker when the live /v1/models fetch - # fails or no credentials are configured yet. - fallback_models=( - "muse-spark-1.2", - "muse-spark-1.2-contributor", - ), + fallback_models=("muse-spark-1.2", "muse-spark-1.2-contributor"), ) register_provider(meta_ai) diff --git a/plugins/model-providers/minimax/__init__.py b/plugins/model-providers/minimax/__init__.py index 7dbaf4000c..00dcbe2f66 100644 --- a/plugins/model-providers/minimax/__init__.py +++ b/plugins/model-providers/minimax/__init__.py @@ -1,9 +1,8 @@ -"""MiniMax provider profiles (international + China). +"""MiniMax provider profiles (international, China, OAuth). -The default API-key routes use anthropic_messages because their base URLs end -with /anthropic. Users can opt MiniMax-M3 into the OpenAI-compatible endpoint -with base_url=https://api.minimax.io/v1; that route needs MiniMax-specific -reasoning controls in extra_body. +Default routes use anthropic_messages (base URLs end in /anthropic). Users can +opt MiniMax-M3 into the OpenAI-compatible https://api.minimax.io/v1 route, +which needs MiniMax-specific reasoning controls in extra_body. """ from typing import Any @@ -15,15 +14,7 @@ from providers.base import ProviderProfile def _is_minimax_global_openai_base_url(base_url: str | None) -> bool: parsed = urlparse(str(base_url or "").strip()) - if (parsed.hostname or "").lower() != "api.minimax.io": - return False - path = parsed.path.rstrip("/").lower() - return path == "/v1" - - -def _is_minimax_m3(model: str | None) -> bool: - normalized = str(model or "").strip().lower() - return normalized in {"minimax-m3", "minimax/minimax-m3"} + return (parsed.hostname or "").lower() == "api.minimax.io" and parsed.path.rstrip("/").lower() == "/v1" class MiniMaxProfile(ProviderProfile): @@ -37,25 +28,16 @@ class MiniMaxProfile(ProviderProfile): base_url: str | None = None, **context: Any, ) -> tuple[dict[str, Any], dict[str, Any]]: - """Emit M3 reasoning controls for api.minimax.io/v1. - - MiniMax-M3's OpenAI-compatible endpoint keeps thinking inline unless - ``reasoning_split`` is sent, so always request the split format on that - route. ``thinking`` controls the M3 mode; Hermes' effort levels are not - a MiniMax depth knob here, so they only select adaptive vs disabled. - """ - if not _is_minimax_global_openai_base_url(base_url) or not _is_minimax_m3(model): + """M3 on api.minimax.io/v1 keeps thinking inline unless ``reasoning_split`` + is sent; effort levels only select adaptive vs disabled ``thinking``.""" + is_m3 = str(model or "").strip().lower() in {"minimax-m3", "minimax/minimax-m3"} + if not _is_minimax_global_openai_base_url(base_url) or not is_m3: return {}, {} - extra_body: dict[str, Any] = {"reasoning_split": True} - if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False: extra_body["thinking"] = {"type": "disabled"} - return extra_body, {} - - if reasoning_config is not None: + elif reasoning_config is not None: extra_body["thinking"] = {"type": "adaptive"} - return extra_body, {} diff --git a/plugins/model-providers/nebius-token-factory/__init__.py b/plugins/model-providers/nebius-token-factory/__init__.py index d54881d4cb..198dd8d21a 100644 --- a/plugins/model-providers/nebius-token-factory/__init__.py +++ b/plugins/model-providers/nebius-token-factory/__init__.py @@ -4,33 +4,21 @@ from __future__ import annotations from typing import Any +from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort from providers import register_provider from providers.base import ProviderProfile - -def _flat_model_name(model: str | None) -> str: - """Return a lowercase model id, tolerating vendor-prefixed IDs.""" - return (model or "").strip().rsplit("/", 1)[-1].lower() - - -def _model_supports_reasoning_effort(model: str | None) -> bool: - """Conservative allowlist for Nebius models that expose reasoning effort.""" - model_name = _flat_model_name(model) - if not model_name: - return False - return any( - marker in model_name - for marker in ( - "deepseek-r1", - "deepseek-v4", - "deepseek-reasoner", - "gpt-oss", - "glm-5", - "kimi-k2", - "minimax-m2", - "qwen3", - ) - ) +# Conservative allowlist of model families that expose reasoning effort. +_REASONING_MARKERS = ( + "deepseek-r1", + "deepseek-v4", + "deepseek-reasoner", + "gpt-oss", + "glm-5", + "kimi-k2", + "minimax-m2", + "qwen3", +) class NebiusTokenFactoryProfile(ProviderProfile): @@ -44,46 +32,25 @@ class NebiusTokenFactoryProfile(ProviderProfile): supports_reasoning: bool = False, **context: Any, ) -> tuple[dict[str, Any], dict[str, Any]]: - if not supports_reasoning and not _model_supports_reasoning_effort(model): + model_name = (model or "").strip().rsplit("/", 1)[-1].lower() + if not supports_reasoning and not any(marker in model_name for marker in _REASONING_MARKERS): return {}, {} - - if isinstance(reasoning_config, dict): - enabled = reasoning_config.get("enabled", True) - raw_effort = reasoning_config.get("effort", "medium") - else: - enabled = True - raw_effort = "medium" - - effort = str(raw_effort or "medium").strip().lower() - if enabled is False or effort in {"none", "off", "disabled"}: + rc = reasoning_config if isinstance(reasoning_config, dict) else {} + # Unset/blank effort defaults to medium (reasoning ON). + effort = str(rc.get("effort", "medium") or "medium").strip().lower() + if rc.get("enabled", True) is False or effort in {"none", "off", "disabled"}: return {}, {} - # Canonical clamp (nearest weaker supported level, never escalate, - # monotonic) — the hand-rolled map this replaces inverted the ladder: - # ultra fell through to medium while xhigh mapped to high. - from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort - - effort = clamp_effort(effort, NEBIUS_EFFORTS) or "medium" - - return {}, {"reasoning_effort": effort} + # Canonical clamp: nearest weaker supported level, never escalate. + return {}, {"reasoning_effort": clamp_effort(effort, NEBIUS_EFFORTS) or "medium"} nebius_token_factory = NebiusTokenFactoryProfile( name="nebius-token-factory", - aliases=( - "nebius", - "nebius-tokenfactory", - "nebius-tf", - "token-factory", - "tokenfactory", - ), + aliases=("nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory"), display_name="Nebius Token Factory", description="Nebius Token Factory — OpenAI-compatible inference", signup_url="https://tokenfactory.nebius.com/", - env_vars=( - "NEBIUS_API_KEY", - "NEBIUS_TOKEN_FACTORY_API_KEY", - "NEBIUS_BASE_URL", - ), + env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY", "NEBIUS_BASE_URL"), base_url="https://api.tokenfactory.nebius.com/v1", models_url="https://api.tokenfactory.nebius.com/v1/models?verbose=true", auth_type="api_key", diff --git a/plugins/model-providers/nous/__init__.py b/plugins/model-providers/nous/__init__.py index ad7540ad13..24df5f8503 100644 --- a/plugins/model-providers/nous/__init__.py +++ b/plugins/model-providers/nous/__init__.py @@ -16,14 +16,7 @@ class NousProfile(ProviderProfile): """Nous Portal — product tags, reasoning with Nous-specific omission.""" def resolve_aux_model(self, *, vision: bool = False) -> str: - """Ask the Portal which cheap model it currently recommends. - - ``/api/nous/recommended-models`` is the authoritative, tier-aware - source (free vs paid), so the auxiliary fast tier tracks the live - catalog instead of a hardcoded id that 404s the day Nous retires it. - The underlying fetch is memory- and disk-cached with a last-known-good - fallback, so this is cheap to call and safe offline. - """ + """Portal's tier-aware ``/api/nous/recommended-models`` pick (cached, offline-safe).""" try: from hermes_cli.models import get_nous_recommended_aux_model @@ -31,37 +24,12 @@ class NousProfile(ProviderProfile): except Exception: return "" - def build_extra_body( - self, *, session_id: str | None = None, **context - ) -> dict[str, Any]: + def build_extra_body(self, *, session_id: str | None = None, **context) -> dict[str, Any]: body: dict[str, Any] = {"tags": nous_portal_tags(session_id=session_id)} - # Top-level session_id → provider sticky routing key. Pins every - # turn of a session to the same upstream endpoint so explicit - # Anthropic cache_control breakpoints stay warm instead of - # cold-writing a fresh cache on each reroute (Anthropic/Vertex/ - # Bedrock caches are instance-local). Mirrors the OpenRouter - # profile; without it the portal falls back to hashing the opening - # messages, which breaks pinning whenever those shift. - # - # Resolve it exactly like ``nous_portal_tags`` resolves the - # ``conversation=`` tag: ambient context first (the lineage ROOT id - # published by the agent loop), explicit argument as fallback. - # - # The gap this closes is the auxiliary call sites — compression, - # title generation, vision, web_extract, session_search, MoA slots. - # They funnel through ``agent.auxiliary_client`` which has no session - # handle, so they never pass ``session_id``: they carried the - # ``conversation=`` tag but NO sticky key at all, and each one routed - # independently of the conversation it belongs to. Reading the same - # ambient contextvar the tag already uses fixes that with zero - # per-call-site plumbing; a host-declared routing scope (#96811) wins - # over it when one was published for this turn. - # - # For the main loop the two agree anyway under the default - # ``compression.in_place: true`` (#38763), where compaction keeps the - # session id; the ambient root additionally keeps the key stable for - # installs that opt back into rotating compaction, and across - # delegate-subagent trees. + # Top-level session_id = sticky routing key, so Anthropic-style cache + # breakpoints stay warm on one upstream instance. Resolved like the + # ``conversation=`` tag: declared scope, then the ambient lineage ROOT + # (covers aux call sites that pass no session_id), then the explicit argument. sticky_key = _cache_scope_from_session_id( get_affinity_scope() or get_conversation_context() or session_id ) @@ -74,23 +42,13 @@ class NousProfile(ProviderProfile): @staticmethod def _cannot_disable_reasoning(model: str | None) -> bool: - """True when a disable can't safely be sent for *model*. + """True when ``reasoning: {enabled: false}`` would 400 on *model*. - Reasoning-mandatory routes answer ``reasoning: {enabled: false}`` - with HTTP 400 ("Reasoning is mandatory for this model"), so the - catalog decides. Cache-only, and an unknown model (catalog cold, - unlisted, or unreachable) also answers True: a cold first turn errs - toward the old omit-everything behavior rather than risking a 400. - - A route the catalog says takes no reasoning parameter at all is - treated the same way — sending it a disable is sending a parameter - the Portal has told us it doesn't accept. + Cache-only catalog lookup; unknown/cold (warmer kicked) and + no-reasoning-parameter routes both answer True (omit rather than risk a 400). """ try: - from hermes_cli.models import ( - nous_model_reasoning_capabilities, - warm_nous_reasoning_caps_async, - ) + from hermes_cli.models import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async caps = nous_model_reasoning_capabilities(model) if caps is None: @@ -98,9 +56,7 @@ class NousProfile(ProviderProfile): return True except Exception: return True - if not caps.get("supports_reasoning"): - return True - return bool(caps.get("mandatory")) + return not caps.get("supports_reasoning") or bool(caps.get("mandatory")) def build_api_kwargs_extras( self, @@ -110,25 +66,16 @@ class NousProfile(ProviderProfile): model: str | None = None, **context, ) -> tuple[dict[str, Any], dict[str, Any]]: - """Nous: passes the full reasoning_config, disable included. - - The Portal honors ``reasoning: {enabled: false}`` — it is the only - wire shape that does. Sending nothing means the *upstream* default, - which for a thinking-first model like ``deepseek/deepseek-v4-pro`` - (catalog: ``default_effort: high``) is thinking ON, so omitting a - disable silently ignored the user's "thinking off". - """ - extra_body = {} - if supports_reasoning: - if reasoning_config is not None: - rc = dict(reasoning_config) - if rc.get("enabled") is False and self._cannot_disable_reasoning(model): - pass # route rejects a disable — let the model think - else: - extra_body["reasoning"] = rc - else: - extra_body["reasoning"] = {"enabled": True, "effort": "medium"} - return extra_body, {} + """Pass the full reasoning_config, disable included (the Portal honors it; + omitting it means the upstream default, thinking ON for V4-class models).""" + if not supports_reasoning: + return {}, {} + if reasoning_config is None: + return {"reasoning": {"enabled": True, "effort": "medium"}}, {} + rc = dict(reasoning_config) + if rc.get("enabled") is False and self._cannot_disable_reasoning(model): + return {}, {} + return {"reasoning": rc}, {} nous = NousProfile( @@ -138,10 +85,7 @@ nous = NousProfile( display_name="Nous Research", description="Nous Research — Hermes model family", signup_url="https://nousresearch.com/", - fallback_models=( - "hermes-3-405b", - "hermes-3-70b", - ), + fallback_models=("hermes-3-405b", "hermes-3-70b"), base_url="https://inference-api.nousresearch.com/v1", auth_type="oauth_device_code", ) diff --git a/plugins/model-providers/nvidia/__init__.py b/plugins/model-providers/nvidia/__init__.py index 1175693692..abc81f45c0 100644 --- a/plugins/model-providers/nvidia/__init__.py +++ b/plugins/model-providers/nvidia/__init__.py @@ -9,36 +9,23 @@ from providers.base import ProviderProfile class NvidiaProviderProfile(ProviderProfile): """NVIDIA NIM accepts a stricter ToolMessage schema than most OpenAI-compatible APIs.""" - def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]: - needs_sanitize = any( + @staticmethod + def _needs_strip(msg: Any) -> bool: + return ( isinstance(msg, dict) and msg.get("role") == "tool" and ("name" in msg or "tool_name" in msg) - for msg in messages ) - if not needs_sanitize: - return messages - # Copy-on-write: shallow outer-list copy, then a shallow dict copy - # only for the role:"tool" messages that actually need a field - # dropped. Avoids recursively deep-copying every message's content - # (including large tool outputs and attachments) for a turn that - # only ever needs to touch two top-level keys on a handful of - # messages. Matches the pattern already used by the shared - # sanitizer in agent/transports/chat_completions.py and by - # QwenProfile.prepare_messages(). - sanitized = list(messages) - for idx, msg in enumerate(messages): - if ( - isinstance(msg, dict) - and msg.get("role") == "tool" - and ("name" in msg or "tool_name" in msg) - ): - msg_copy = dict(msg) - msg_copy.pop("name", None) - msg_copy.pop("tool_name", None) - sanitized[idx] = msg_copy - return sanitized + def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]: + """Copy-on-write: only tool messages that lose a field are copied + (no deep copy of large tool outputs); untouched input returned as-is.""" + if not any(self._needs_strip(msg) for msg in messages): + return messages + return [ + {k: v for k, v in msg.items() if k not in ("name", "tool_name")} if self._needs_strip(msg) else msg + for msg in messages + ] nvidia = NvidiaProviderProfile( @@ -48,10 +35,7 @@ nvidia = NvidiaProviderProfile( display_name="NVIDIA NIM", description="NVIDIA NIM — accelerated inference", signup_url="https://build.nvidia.com/", - fallback_models=( - "nvidia/llama-3.1-nemotron-70b-instruct", - "nvidia/llama-3.3-70b-instruct", - ), + fallback_models=("nvidia/llama-3.1-nemotron-70b-instruct", "nvidia/llama-3.3-70b-instruct"), base_url="https://integrate.api.nvidia.com/v1", default_max_tokens=16384, ) diff --git a/plugins/model-providers/ollama-cloud/__init__.py b/plugins/model-providers/ollama-cloud/__init__.py index ccc1eb19b3..a9e467586e 100644 --- a/plugins/model-providers/ollama-cloud/__init__.py +++ b/plugins/model-providers/ollama-cloud/__init__.py @@ -1,24 +1,15 @@ """Ollama Cloud provider profile. -Ollama Cloud's OpenAI-compatible ``/v1/chat/completions`` endpoint -supports top-level ``reasoning_effort`` with values ``none``, ``low``, -``medium``, ``high``, and ``max`` (the last being undocumented but -empirically confirmed for DeepSeek V4 — ``max`` produces ~2.5× more -thinking tokens than ``high``). - -This profile maps Hermes's ``xhigh`` → ``max`` to unlock DeepSeek V4's -"Max thinking" tier through Ollama Cloud. ``low`` / ``medium`` / ``high`` -pass through unchanged. - -When reasoning is explicitly disabled (``enabled: false`` or -``effort: "none"``), ``reasoning_effort`` is omitted entirely so the -model runs in non-thinking mode. +Top-level ``reasoning_effort`` on /v1/chat/completions accepts none|low|medium| +high|max (``max`` is undocumented but real — ~2.5x more thinking tokens on +DeepSeek V4); Hermes' ``xhigh`` maps to ``max``. """ from __future__ import annotations from typing import Any +from agent.reasoning_effort import OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES, clamp_effort from providers import register_provider from providers.base import ProviderProfile @@ -27,61 +18,23 @@ class OllamaCloudProfile(ProviderProfile): """Ollama Cloud — maps xhigh→max via top-level reasoning_effort.""" def build_api_kwargs_extras( - self, - *, - reasoning_config: dict | None = None, - supports_reasoning: bool = False, - **ctx: Any, + self, *, reasoning_config: dict | None = None, supports_reasoning: bool = False, **ctx: Any ) -> tuple[dict[str, Any], dict[str, Any]]: - """Emit top-level ``reasoning_effort`` for Ollama Cloud thinking models. - - Gated on ``supports_reasoning``, which the transport resolves from the - model's native ``/api/show`` ``capabilities`` (``thinking``). Models - without the thinking capability (e.g. ``gemma3``, ``qwen3-coder``) get - no ``reasoning_effort`` at all — emitting it there is a no-op the API - ignores, and gating avoids sending a meaningless field. - """ - top_level: dict[str, Any] = {} - - if not supports_reasoning: + """Gated on ``supports_reasoning`` (resolved from the model's /api/show + ``thinking`` capability) so non-thinking models get no meaningless field.""" + if not supports_reasoning or not reasoning_config or not isinstance(reasoning_config, dict): return {}, {} - - if reasoning_config and isinstance(reasoning_config, dict): - enabled = reasoning_config.get("enabled", True) - if enabled is False: - # Ollama Cloud defaults to thinking ON, and ignores the - # extra_body.thinking:{type:disabled} shape (verified live). - # The ONLY way to actually suppress thinking on its - # /v1/chat/completions endpoint is top-level - # reasoning_effort:"none" — omitting the field leaves - # thinking on. - return {}, {"reasoning_effort": "none"} - - effort = (reasoning_config.get("effort") or "").strip().lower() - if not effort: - # No explicit effort requested — let the model decide - # (Ollama Cloud's server default is thinking ON). - return {}, {} - if effort == "none": - return {}, {"reasoning_effort": "none"} # explicit off switch - # Accepted set {none, low, medium, high, max} is declared in - # agent.reasoning_effort ("minimal" is rejected with HTTP 400 → - # clamps to low; xhigh rounds up to max). Bespoke levels outside - # the ladder are omitted so the model applies its own default - # rather than triggering a hard 400. - from agent.reasoning_effort import ( - OLLAMA_CLOUD_EFFORTS, - OLLAMA_CLOUD_OVERRIDES, - clamp_effort, - ) - - clamped = clamp_effort( - effort, OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES - ) - if clamped in OLLAMA_CLOUD_EFFORTS: - top_level["reasoning_effort"] = clamped - - return {}, top_level + # Ollama Cloud defaults to thinking ON and ignores extra_body.thinking + # (verified live); top-level reasoning_effort:"none" is the ONLY off switch. + effort = (reasoning_config.get("effort") or "").strip().lower() + if reasoning_config.get("enabled", True) is False or effort == "none": + return {}, {"reasoning_effort": "none"} + if not effort: + return {}, {} # let the server default (thinking ON) apply + # "minimal" 400s -> clamps to low; xhigh rounds up to max. Bespoke + # levels outside the ladder are omitted rather than risking a 400. + clamped = clamp_effort(effort, OLLAMA_CLOUD_EFFORTS, OLLAMA_CLOUD_OVERRIDES) + return {}, {"reasoning_effort": clamped} if clamped in OLLAMA_CLOUD_EFFORTS else {} ollama_cloud = OllamaCloudProfile( diff --git a/plugins/model-providers/opencode-free/__init__.py b/plugins/model-providers/opencode-free/__init__.py index 96d1970cc9..d8251415cc 100644 --- a/plugins/model-providers/opencode-free/__init__.py +++ b/plugins/model-providers/opencode-free/__init__.py @@ -1,12 +1,9 @@ -"""OpenCode Free provider profile. +"""OpenCode Free provider profile: the free tier on the Zen relay (https://opencode.ai/zen/v1). -OpenCode's free model tier on the Zen relay (https://opencode.ai/zen/v1). -KEYLESS: the relay serves free-tier models anonymously and rejects any -Authorization bearer it doesn't recognize with 401 — so this provider -never sends a credential at all (the runtime resolver pins the keyless -placeholder and an empty Authorization header; see -hermes_cli.models.opencode_zen_free_runtime). No OpenCode account needed. -Select via ``hermes model`` or ``/model free``. +KEYLESS: the relay serves free-tier models anonymously and 401s any bearer it +doesn't recognize, so this provider never sends a credential (the runtime +resolver pins the keyless placeholder and an empty Authorization header; see +hermes_cli.models.opencode_zen_free_runtime). Select via ``/model free``. """ from typing import Any @@ -15,25 +12,13 @@ from hermes_cli import __version__ as _HERMES_VERSION from providers import register_provider from providers.base import ProviderProfile -# Attribution headers, same values as the opencode-zen/go profiles, plus the -# empty Authorization override that keeps the SDK's "Bearer " -# off the wire (the free tier 401s any unrecognized bearer). -_KEYLESS_HEADERS = { - "Authorization": "", - "HTTP-Referer": "https://hermes-agent.nousresearch.com", - "X-Title": "Hermes Agent", - "User-Agent": f"HermesAgent/{_HERMES_VERSION}", -} - class OpenCodeFreeProfile(ProviderProfile): """OpenCode Free — keyless, with Ox Alpha reasoning controls. - Ox Alpha (x-preview-f-free) is reachable through this provider as well - as opencode-zen; both share the same wire contract (reasoning_effort - accepts exactly low/high/max — anything else 400s). The translation - lives in the zen plugin; resolve it through the registered zen profile's - module so the two providers can never drift. + Ox Alpha (x-preview-f-free) is also reachable via opencode-zen with the same + wire contract; the translation lives in the zen plugin and is resolved through + the registered zen profile's module so the two providers can never drift. """ def build_api_kwargs_extras( @@ -44,8 +29,7 @@ class OpenCodeFreeProfile(ProviderProfile): from providers import get_provider_profile - zen_profile = get_provider_profile("opencode-zen") - zen_module = sys.modules[type(zen_profile).__module__] + zen_module = sys.modules[type(get_provider_profile("opencode-zen")).__module__] return zen_module._build_ox_alpha_reasoning_extras(reasoning_config, model) except Exception: return {}, {} @@ -58,9 +42,17 @@ opencode_free = OpenCodeFreeProfile( base_url="https://opencode.ai/zen/v1", display_name="OpenCode Free", description="OpenCode free models — keyless, no account needed", - default_headers=dict(_KEYLESS_HEADERS), + # Attribution headers (same values as opencode-zen/go) plus the empty + # Authorization override that keeps the SDK's "Bearer " off the + # wire (the free tier 401s any unrecognized bearer). + default_headers={ + "Authorization": "", + "HTTP-Referer": "https://hermes-agent.nousresearch.com", + "X-Title": "Hermes Agent", + "User-Agent": f"HermesAgent/{_HERMES_VERSION}", + }, # laguna is the fastest non-UA-gated free model; big-pickle 429s every - # client except the opencode CLI's own User-Agent (verified 2026-08-21). + # client except the opencode CLI's own User-Agent. default_aux_model="laguna-s-2.1-free", ) diff --git a/plugins/model-providers/opencode-zen/__init__.py b/plugins/model-providers/opencode-zen/__init__.py index 2d55cae33b..b8601771f6 100644 --- a/plugins/model-providers/opencode-zen/__init__.py +++ b/plugins/model-providers/opencode-zen/__init__.py @@ -1,25 +1,20 @@ """OpenCode provider profiles (Zen + Go). -Both use per-model api_mode routing: - - OpenCode Zen: Claude → anthropic_messages, GPT-5/Codex/Grok → codex_responses, - Muse Spark → codex_responses, everything else → chat_completions (this profile) - - OpenCode Go: GPT / Grok / Muse Spark → codex_responses, MiniMax/Qwen → anthropic_messages, - GLM/Kimi/DeepSeek/MiMo → chat_completions (this profile) +Both route api_mode per model in core; these profiles carry the +chat_completions reasoning translations (GLM-5.2, Kimi K2, DeepSeek, Ox Alpha). """ from __future__ import annotations from typing import Any +from agent import reasoning_effort as re_ from hermes_cli import __version__ as _HERMES_VERSION from providers import register_provider from providers.base import ProviderProfile -# Attribution headers sent on every OpenCode request. Same values we send -# to OpenRouter, Vercel AI Gateway, and Fireworks. Going through -# profile.default_headers means they survive model switches and credential -# rotation. Without them OpenCode only sees the OpenAI SDK's generic -# "OpenAI/Python x.y.z" User-Agent and can't tell the traffic is Hermes Agent. +# Attribution headers (same values as OpenRouter / Vercel / Fireworks); via +# default_headers so they survive model switches and credential rotation. _ATTRIBUTION_HEADERS = { "HTTP-Referer": "https://hermes-agent.nousresearch.com", "X-Title": "Hermes Agent", @@ -32,15 +27,9 @@ def _flat_model_name(model: str | None) -> str: return (model or "").strip().rsplit("/", 1)[-1].lower() -def _is_kimi_k2_model(model: str | None) -> bool: - return _flat_model_name(model).startswith("kimi-k2") - - def _is_deepseek_thinking_model(model: str | None) -> bool: m = _flat_model_name(model) - if m.startswith("deepseek-v") and not m.startswith("deepseek-v3"): - return True - return m == "deepseek-reasoner" + return (m.startswith("deepseek-v") and not m.startswith("deepseek-v3")) or m == "deepseek-reasoner" def _is_glm_5_2_model(model: str | None) -> bool: @@ -49,143 +38,70 @@ def _is_glm_5_2_model(model: str | None) -> bool: return any(token in m for token in ("glm-5.2", "glm-5-2", "glm-5p2")) +def _requested_effort(reasoning_config: dict | None) -> str | None: + """Normalized effort when reasoning is enabled and an effort is set, else None.""" + effort = re_.requested_effort(reasoning_config) + return None if effort == "none" else effort + + +def _thinking_toggle_extras( + reasoning_config: dict | None, efforts, overrides=None +) -> tuple[dict[str, Any], dict[str, Any]]: + """Moonshot/DeepSeek wire shape: extra_body.thinking XOR top-level reasoning_effort + (sending both is an HTTP 400).""" + if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False: + return {"thinking": {"type": "disabled"}}, {} + clamped = re_.clamp_effort(_requested_effort(reasoning_config), efforts, overrides) + if clamped in efforts: + return {}, {"reasoning_effort": clamped} + return {"thinking": {"type": "enabled"}}, {} + + class OpenCodeGoProfile(ProviderProfile): """OpenCode Go - model-specific reasoning controls.""" - # Per-model completion-token cap. The opencode-go relay's default is - # too large for mimo-v2.5-pro — it sends max_tokens=262144 but Xiaomi - # only supports 131072 completion tokens and 400s the request. - # Setting an explicit cap here prevents the relay default from being - # applied. Keys are normalized via _flat_model_name(). + # The relay's default max_tokens (262144) exceeds what Xiaomi accepts for + # mimo-v2.5-pro and 400s; keys are normalized via _flat_model_name(). _MODEL_MAX_TOKENS: dict[str, int] = { "mimo-v2.5-pro": 131072, } def get_max_tokens(self, model: str | None) -> int | None: cap = self._MODEL_MAX_TOKENS.get(_flat_model_name(model)) - if cap is not None: - return cap - return self.default_max_tokens + return self.default_max_tokens if cap is None else cap def build_api_kwargs_extras( self, *, reasoning_config: dict | None = None, model: str | None = None, **context ) -> tuple[dict[str, Any], dict[str, Any]]: - extra_body: dict[str, Any] = {} - top_level: dict[str, Any] = {} - if _is_glm_5_2_model(model): - # GLM-5.2 on OpenCode Go uses its native OpenAI-compatible - # reasoning_effort knob (high/max — declared in - # agent.reasoning_effort, shared with the zai profile); leave the - # server default alone when reasoning is disabled or unset. + # Native reasoning_effort knob (high/max); server default when unset/disabled. + effort = _requested_effort(reasoning_config) + if effort is None: + return {}, {} + clamped = re_.clamp_effort(effort, re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES) + return {}, {"reasoning_effort": clamped if clamped in re_.GLM52_EFFORTS else "high"} + if _flat_model_name(model).startswith("kimi-k2"): if not isinstance(reasoning_config, dict): - return extra_body, top_level - if reasoning_config.get("enabled") is False: - return extra_body, top_level - effort = (reasoning_config.get("effort") or "").strip().lower() - if not effort or effort == "none": - return extra_body, top_level - from agent.reasoning_effort import ( - GLM52_EFFORTS, - GLM52_OVERRIDES, - clamp_effort, + return {}, {} + return _thinking_toggle_extras(reasoning_config, re_.KIMI_K2_EFFORTS) + if _is_deepseek_thinking_model(model): + return _thinking_toggle_extras( + reasoning_config, re_.DEEPSEEK_V4_EFFORTS, re_.DEEPSEEK_V4_OVERRIDES ) - - clamped = clamp_effort(effort, GLM52_EFFORTS, GLM52_OVERRIDES) - top_level["reasoning_effort"] = ( - clamped if clamped in GLM52_EFFORTS else "high" - ) - return extra_body, top_level - - if _is_kimi_k2_model(model): - # Kimi K2 on OpenCode Go uses Moonshot's native wire shape: - # extra_body.thinking (binary toggle) + top-level reasoning_effort - # (low|medium|high). Mirrors the KimiProfile (api.moonshot.ai/v1). - if not isinstance(reasoning_config, dict): - # No config → leave server defaults alone. - return extra_body, top_level - - enabled = reasoning_config.get("enabled") is not False - if not enabled: - extra_body["thinking"] = {"type": "disabled"} - return extra_body, top_level - - effort = (reasoning_config.get("effort") or "").strip().lower() - if effort and effort != "none": - from agent.reasoning_effort import KIMI_K2_EFFORTS, clamp_effort - - clamped = clamp_effort(effort, KIMI_K2_EFFORTS) - if clamped in KIMI_K2_EFFORTS: - top_level["reasoning_effort"] = clamped - - # Avoid "cannot specify both 'thinking' and 'reasoning_effort'" HTTP 400: - # only send extra_body["thinking"] when no reasoning_effort is set. - if "reasoning_effort" not in top_level: - extra_body["thinking"] = {"type": "enabled"} - return extra_body, top_level - - if not _is_deepseek_thinking_model(model): - return extra_body, top_level - - enabled = True - if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False: - enabled = False - - if not enabled: - extra_body["thinking"] = {"type": "disabled"} - return extra_body, top_level - - if isinstance(reasoning_config, dict): - effort = (reasoning_config.get("effort") or "").strip().lower() - if effort and effort != "none": - from agent.reasoning_effort import ( - DEEPSEEK_V4_EFFORTS, - DEEPSEEK_V4_OVERRIDES, - clamp_effort, - ) - - clamped = clamp_effort( - effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES - ) - if clamped in DEEPSEEK_V4_EFFORTS: - top_level["reasoning_effort"] = clamped - - # Avoid "cannot specify both 'thinking' and 'reasoning_effort'" HTTP 400: - # only send extra_body["thinking"] when no reasoning_effort is set. - if "reasoning_effort" not in top_level: - extra_body["thinking"] = {"type": "enabled"} - - return extra_body, top_level + return {}, {} def _build_ox_alpha_reasoning_extras( reasoning_config: dict | None, model: str | None ) -> tuple[dict[str, Any], dict[str, Any]]: - """Shared Ox Alpha (x-preview-f-free) reasoning_effort translation. - - Used by both the opencode-zen profile and the opencode-free keyless - profile — the model is reachable through either provider and the wire - contract is identical (low/high/max only; anything else 400s). - """ + """Ox Alpha (x-preview-f-free) reasoning_effort translation, shared with the + opencode-free profile (low/high/max only; anything else 400s).""" if _flat_model_name(model) != "x-preview-f-free": return {}, {} - if not isinstance(reasoning_config, dict): - return {}, {} - if reasoning_config.get("enabled") is False: - return {}, {} - - effort = (reasoning_config.get("effort") or "").strip().lower() - if not effort or effort == "none": - return {}, {} - - from agent.reasoning_effort import ( - OX_ALPHA_EFFORTS, - OX_ALPHA_OVERRIDES, - clamp_effort, + clamped = re_.clamp_effort( + _requested_effort(reasoning_config), re_.OX_ALPHA_EFFORTS, re_.OX_ALPHA_OVERRIDES ) - - clamped = clamp_effort(effort, OX_ALPHA_EFFORTS, OX_ALPHA_OVERRIDES) - if clamped not in OX_ALPHA_EFFORTS: + if clamped not in re_.OX_ALPHA_EFFORTS: return {}, {} return {}, {"reasoning_effort": clamped} diff --git a/plugins/model-providers/openrouter/__init__.py b/plugins/model-providers/openrouter/__init__.py index b5e6b60997..20f0597ec1 100644 --- a/plugins/model-providers/openrouter/__init__.py +++ b/plugins/model-providers/openrouter/__init__.py @@ -12,14 +12,10 @@ logger = logging.getLogger(__name__) _CACHE: list[str] | None = None -# Anthropic model families that still accept an explicit "disable thinking" -# request (the manual ``thinking: {type: "disabled"}`` form OpenRouter emits -# for ``reasoning: {enabled: false}``). Everything Claude 4.6 and newer — -# including future date-stamped / named models (fable, mythos-class, …) — -# mandates reasoning and returns HTTP 400 on any disable form. We therefore -# default *unknown* Anthropic models to "cannot disable" (the modern contract) -# and keep only this explicit legacy allowlist of models that can. Mirrors the -# default-to-newest philosophy in agent/anthropic_adapter._get_anthropic_max_output. +# Legacy allowlist of Anthropic models that still accept an explicit "disable +# thinking" request. Claude 4.6+ and newer named models mandate reasoning and +# 400 on any disable form, so *unknown* Anthropic models default to "cannot +# disable" (mirrors agent/anthropic_adapter._get_anthropic_max_output). _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS = ( "claude-3", # 3, 3.5, 3.7 "claude-opus-4-0", "claude-opus-4.0", "claude-opus-4-1", "claude-opus-4.1", @@ -32,48 +28,44 @@ _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS = ( def _anthropic_reasoning_is_mandatory(model: str | None) -> bool: - """Return True for Anthropic models that reject any disable-thinking form. - - Claude 4.6+ (adaptive thinking) and newer named models have no "off" - switch — sending ``reasoning: {enabled: false}`` makes OpenRouter emit - ``thinking: {type: "disabled"}``, which these models 400 on. Unknown / - new Anthropic model names default to mandatory so the next un-numbered - release doesn't reintroduce the 400. - """ + """True for Anthropic models that reject any disable-thinking form (unknown -> True).""" m = (model or "").lower() if not m.startswith(("anthropic/", "claude")) and "claude" not in m: return False return not any(sub in m for sub in _ANTHROPIC_REASONING_OPTIONAL_SUBSTRINGS) +def _sticky_key(session_id: str | None) -> str | None: + """Declared routing scope, then ambient conversation, then explicit session_id. + + Aux call sites (compression, titles, vision, MoA…) pass no ``session_id``, + so the ambient lineage ROOT keeps them pinned to their conversation. + """ + return _cache_scope_from_session_id( + get_affinity_scope() or get_conversation_context() or session_id + ) + + class OpenRouterProfile(ProviderProfile): """OpenRouter aggregator — provider preferences, reasoning config passthrough.""" @staticmethod def _clamp_reasoning_to_catalog(cfg: dict[str, Any], model: str | None) -> dict[str, Any]: - """Clamp ``cfg["effort"]`` to the model's catalog-advertised levels. + """Clamp ``cfg["effort"]`` to the nearest LOWER catalog-advertised level. - OpenRouter's /v1/models entries publish ``reasoning.supported_efforts`` - per model (ported from PrimeIntellect-ai/prime-agent#1258). Sending an - unsupported effort (e.g. ``ultra`` to a route that stops at ``high``) - yields provider 4xx errors; clamp to the nearest LOWER supported level - instead. No-op when the catalog is unreachable, the model is unlisted, - or no supported_efforts list is published (None = all levels accepted). + No-op when the catalog is unreachable, the model is unlisted, or no + supported_efforts list is published (None = all levels accepted). """ effort = cfg.get("effort") if not effort or cfg.get("enabled") is False: return cfg try: - from hermes_cli.models import ( - clamp_reasoning_effort_to_supported, - openrouter_model_reasoning_capabilities, - ) + from hermes_cli.models import clamp_reasoning_effort_to_supported, openrouter_model_reasoning_capabilities + caps = openrouter_model_reasoning_capabilities(model) if not caps or not caps.get("supports_reasoning"): return cfg - clamped = clamp_reasoning_effort_to_supported( - effort, caps.get("supported_efforts") - ) + clamped = clamp_reasoning_effort_to_supported(effort, caps.get("supported_efforts")) except Exception: return cfg if clamped and clamped != effort: @@ -82,83 +74,46 @@ class OpenRouterProfile(ProviderProfile): "(catalog supported_efforts=%s)", effort, clamped, model, caps.get("supported_efforts"), ) - cfg = dict(cfg) - cfg["effort"] = clamped + cfg = {**cfg, "effort": clamped} return cfg def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: - """Fetch from public OpenRouter catalog — no auth required. - - Note: Tool-call capability filtering is applied by hermes_cli/models.py - via fetch_openrouter_models() → _openrouter_model_supports_tools(), not - here. The picker early-returns via the dedicated openrouter path before - reaching this method, so filtering here would be unreachable. - """ + """Fetch from the public OpenRouter catalog (no auth). Tool-call filtering + happens in hermes_cli/models.py, which the picker reaches first.""" global _CACHE # noqa: PLW0603 if _CACHE is not None: return _CACHE try: result = super().fetch_models(api_key=None, base_url=base_url, timeout=timeout) - if result is not None: - _CACHE = result - return result except Exception as exc: logger.debug("fetch_models(openrouter): %s", exc) return None + if result is not None: + _CACHE = result + return result - def build_extra_body( - self, *, session_id: str | None = None, **context: Any - ) -> dict[str, Any]: + def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]: body: dict[str, Any] = {} - # Top-level session_id → OpenRouter's sticky routing key. Per their - # prompt-caching docs it is used directly as the routing key instead of - # hashing the opening messages, and it activates stickiness on the - # first successful request rather than only after a cache hit. - # - # Resolve it from the declared routing scope first (set only by a host - # that names its own conversation, #96811), then the ambient conversation - # contextvar, with the explicit argument as fallback. The gap this closes is the auxiliary call sites - # — compression, title generation, vision, web_extract, session_search, - # MoA slots — which funnel through ``agent.auxiliary_client``. That - # module has no session handle and passes no ``session_id``, so those - # calls sent NO sticky key at all and each routed independently of the - # conversation it belonged to (#70820). - # - # Mirrors the Nous Portal profile, which resolves the same way - # (f2f4df064d). The ambient value is the session-lineage ROOT, so it - # also stays stable for installs that opt out of the default - # ``compression.in_place: true`` and across delegate-subagent trees. - sticky_key = _cache_scope_from_session_id( - get_affinity_scope() or get_conversation_context() or session_id - ) + # Top-level session_id is OpenRouter's sticky routing key (used directly, + # not hashed from the opening messages; active from the first request). + sticky_key = _sticky_key(session_id) if sticky_key: body["session_id"] = sticky_key prefs = context.get("provider_preferences") if prefs: body["provider"] = prefs - # Pareto Code router — model-gated. The plugins block is only - # meaningful for openrouter/pareto-code; sending it on any other - # model has no documented effect and would be confusing in logs. - # See: https://openrouter.ai/docs/guides/routing/routers/pareto-router - model = (context.get("model") or "") - if model == "openrouter/pareto-code": - score = context.get("openrouter_min_coding_score") - if score is not None and score != "": - try: - score_f = float(score) - except (TypeError, ValueError): - score_f = None - if score_f is not None and 0.0 <= score_f <= 1.0: - body["plugins"] = [ - {"id": "pareto-router", "min_coding_score": score_f} - ] + # Pareto Code router plugin is only meaningful for openrouter/pareto-code. + score = context.get("openrouter_min_coding_score") + if (context.get("model") or "") == "openrouter/pareto-code" and score is not None and score != "": + try: + score_f = float(score) + except (TypeError, ValueError): + score_f = None + if score_f is not None and 0.0 <= score_f <= 1.0: + body["plugins"] = [{"id": "pareto-router", "min_coding_score": score_f}] return body def build_api_kwargs_extras( @@ -170,69 +125,29 @@ class OpenRouterProfile(ProviderProfile): session_id: str | None = None, **context: Any, ) -> tuple[dict[str, Any], dict[str, Any]]: - """OpenRouter passes the full reasoning_config dict as extra_body.reasoning. - - For xAI Grok models routed through OpenRouter, attach the - ``x-grok-conv-id`` header so that xAI's prompt cache stays pinned to - the same backend server across turns. - """ + """Pass reasoning_config as extra_body.reasoning; pin Grok's cache via x-grok-conv-id.""" extra_body: dict[str, Any] = {} top_level: dict[str, Any] = {} - extra_headers: dict[str, Any] = {} if supports_reasoning: - # Reasoning-mandatory Anthropic models (Claude 4.6+ / fable / - # future named models) use *adaptive* thinking: the model decides - # how much to think, and OpenRouter ignores ``reasoning.effort`` for - # them entirely. Sending any ``reasoning`` field is therefore both - # pointless and actively harmful: - # - ``{enabled: false}`` → OpenRouter emits Anthropic's manual - # ``thinking: {type: "disabled"}``, which these models 400 on. - # - any enabled form, on a tool-continuation turn whose prior - # assistant tool_call carries no thinking block (chat_completions - # never replays signed thinking blocks), ALSO makes OpenRouter - # emit ``thinking: {type: "disabled"}`` → the same 400 on every - # turn after the first tool call. - # The only reliable behavior is to omit ``reasoning`` and let the - # model default to adaptive. See hermes-agent#42991 (disable case) - # and the tool-replay follow-up. - # - # ``reasoning.effort`` being ignored does NOT mean these models have - # no effort lever — OpenRouter honors the requested effort on the - # top-level ``verbosity`` field instead (it maps to Anthropic's - # ``output_config.effort``; ``reasoning.effort`` is accepted but - # ignored — confirmed by OpenRouter's Claude migration docs and a - # live token-spend probe in hermes-agent#43432). Route the existing - # ``reasoning_config["effort"]`` (sourced from - # ``agent.reasoning_effort``) onto ``verbosity`` so the knob the user - # already sets keeps working for these models. We still send NO - # ``reasoning`` field, preserving the #42991 400 fix. + # Reasoning-mandatory Anthropic models use adaptive thinking: any + # ``reasoning`` field (disable, or an enabled form on a tool-continuation + # turn without a replayed thinking block) makes OpenRouter emit + # ``thinking: {type: "disabled"}`` -> 400. Omit it; the user's effort + # still reaches Anthropic's output_config.effort via top-level ``verbosity``. if _anthropic_reasoning_is_mandatory(model): cfg = reasoning_config or {} effort = cfg.get("effort") - # Only emit when effort is actually requested and reasoning - # isn't explicitly disabled. Otherwise omit ``verbosity`` so the - # model keeps its own adaptive default (``high``). if cfg.get("enabled", True) is not False and effort and effort != "none": top_level["verbosity"] = effort elif reasoning_config is not None: - extra_body["reasoning"] = self._clamp_reasoning_to_catalog( - dict(reasoning_config), model - ) + extra_body["reasoning"] = self._clamp_reasoning_to_catalog(dict(reasoning_config), model) else: extra_body["reasoning"] = {"enabled": True, "effort": "medium"} - # Same resolution as build_extra_body: xAI's prompt cache is pinned per - # backend server via this header, and aux calls pass no session_id, so - # reading the ambient conversation keeps compression/vision/MoA traffic - # on the same Grok backend as the conversation it belongs to. - grok_conv_id = _cache_scope_from_session_id( - get_affinity_scope() or get_conversation_context() or session_id - ) + # xAI's prompt cache is pinned per backend server via this header. + grok_conv_id = _sticky_key(session_id) if grok_conv_id and model and model.startswith(("x-ai/grok-", "xai/grok-")): - extra_headers["x-grok-conv-id"] = grok_conv_id - if extra_headers: - top_level["extra_headers"] = extra_headers - + top_level["extra_headers"] = {"x-grok-conv-id": grok_conv_id} return extra_body, top_level diff --git a/plugins/model-providers/qwen-oauth/__init__.py b/plugins/model-providers/qwen-oauth/__init__.py index ddf9120a65..6e1a88f95c 100644 --- a/plugins/model-providers/qwen-oauth/__init__.py +++ b/plugins/model-providers/qwen-oauth/__init__.py @@ -5,31 +5,34 @@ from providers import register_provider from providers.base import ProviderProfile +def _normalize_parts(content: list) -> list | None: + """List content -> list-of-dict parts (str -> text part, image_url dicts copied, + other junk dropped). None when nothing changed (copy-on-write).""" + parts, changed = [], False + for part in content: + if isinstance(part, str): + parts.append({"type": "text", "text": part}) + changed = True + elif isinstance(part, dict): + if isinstance(part.get("image_url"), dict): + part = {**part, "image_url": dict(part["image_url"])} + changed = True + parts.append(part) + else: + changed = True + return parts if parts and changed else None + + class QwenProfile(ProviderProfile): """Qwen Portal — message normalization, vl_high_resolution, metadata top-level.""" - @staticmethod - def _copy_part_if_request_mutable(part: dict[str, Any]) -> tuple[dict[str, Any], bool]: - image_url = part.get("image_url") - if isinstance(image_url, dict): - copied = dict(part) - copied["image_url"] = dict(image_url) - return copied, True - return part, False - def prepare_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]: - """Normalize content to list-of-dicts format. - - Inject cache_control on system message. - - Matches the behavior of run_agent.py:_qwen_prepare_chat_messages(). - """ + """Normalize content to list-of-dicts and inject cache_control on the system + message. Copy-on-write: only touched messages/parts are copied.""" if not messages: return [] - prepared = list(messages) system_idx: int | None = None - for idx, msg in enumerate(messages): if not isinstance(msg, dict): continue @@ -37,49 +40,22 @@ class QwenProfile(ProviderProfile): system_idx = idx content = msg.get("content") if isinstance(content, str): - msg_copy = dict(msg) - msg_copy["content"] = [{"type": "text", "text": content}] - prepared[idx] = msg_copy + prepared[idx] = {**msg, "content": [{"type": "text", "text": content}]} elif isinstance(content, list): - normalized_parts = [] - changed = False - for part in content: - if isinstance(part, str): - normalized_parts.append({"type": "text", "text": part}) - changed = True - elif isinstance(part, dict): - normalized_part, copied = self._copy_part_if_request_mutable(part) - normalized_parts.append(normalized_part) - changed = changed or copied - else: - changed = True - if normalized_parts and changed: - msg_copy = dict(msg) - msg_copy["content"] = normalized_parts - prepared[idx] = msg_copy + parts = _normalize_parts(content) + if parts is not None: + prepared[idx] = {**msg, "content": parts} - # Inject cache_control on the last part of the system message. if system_idx is not None: msg = prepared[system_idx] - if isinstance(msg, dict): - content = msg.get("content") - if ( - isinstance(content, list) - and content - and isinstance(content[-1], dict) - ): - msg_copy = dict(msg) - content_copy = list(content) - content_copy[-1] = dict(content_copy[-1]) - content_copy[-1]["cache_control"] = {"type": "ephemeral"} - msg_copy["content"] = content_copy - prepared[system_idx] = msg_copy - + content = msg.get("content") + if isinstance(content, list) and content and isinstance(content[-1], dict): + content_copy = list(content) + content_copy[-1] = {**content_copy[-1], "cache_control": {"type": "ephemeral"}} + prepared[system_idx] = {**msg, "content": content_copy} return prepared - def build_extra_body( - self, *, session_id: str | None = None, **context - ) -> dict[str, Any]: + def build_extra_body(self, *, session_id: str | None = None, **context) -> dict[str, Any]: return {"vl_high_resolution_images": True} def build_api_kwargs_extras( @@ -90,10 +66,7 @@ class QwenProfile(ProviderProfile): **context, ) -> tuple[dict[str, Any], dict[str, Any]]: """Qwen metadata goes to top-level api_kwargs, not extra_body.""" - top_level = {} - if qwen_session_metadata: - top_level["metadata"] = qwen_session_metadata - return {}, top_level + return {}, {"metadata": qwen_session_metadata} if qwen_session_metadata else {} qwen = QwenProfile( diff --git a/plugins/model-providers/router/__init__.py b/plugins/model-providers/router/__init__.py index 6e5b21dc2d..ee20972fdb 100644 --- a/plugins/model-providers/router/__init__.py +++ b/plugins/model-providers/router/__init__.py @@ -1,45 +1,18 @@ -"""Ramp Router (router.com) provider plugin for Hermes Agent. +"""Ramp Router (router.com) provider profile: Responses-only LLM gateway. -Provider profile for `Ramp Router `_, Ramp's LLM -gateway: one OpenAI Responses-compatible endpoint at -``https://api.router.com/v1`` that routes each request across upstream -providers (OpenAI, Anthropic, xAI, Fireworks, ...) and handles fallbacks and -spend controls server-side. - -Wire notes (verified live against api.router.com, Aug 2026): - -* **Responses API is the native wire.** Router serves ``GET /v1/models`` - and ``POST /v1/responses``; ``POST /v1/chat/completions`` is only a - minimal compatibility shim (added Aug 2026) that translates onto - Responses. Per-model reasoning-effort validation, reasoning summaries, - and prompt caching are Responses-surface features, so - ``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate - in ``hermes_cli/providers.py`` keep every path on the native wire — - the same shape as the ``api.openai.com`` mandate. -* **Account-scoped catalog.** Valid model IDs are whatever the key's - ``GET /v1/models`` returns (BYOK accounts see extra entries), so this - profile ships **no** ``fallback_models`` — the picker relies on the live - fetch, per Router's own guidance to never hardcode model names. -* **Strict reasoning-effort validation.** Router validates - ``reasoning.effort`` against each model's catalog-declared vocabulary and - returns HTTP 400 ``invalid-argument`` on a level the model does not accept - (e.g. ``max`` on grok-4.6), and 400 ``unsupported_parameter`` when a - non-reasoning model (gpt-4.1 family, gpt-4o, ...) receives any reasoning - field. The catalog publishes the vocabulary per model - (``router.capabilities.reasoning``), so ``supported_reasoning_efforts`` - below feeds the codex transport's clamp from a cached copy of it. -* **Everything else passes through.** ``store: false``, ``prompt_cache_key``, - ``include: ["reasoning.encrypted_content"]``, and ``reasoning.summary`` are - accepted on all models (ignored where a backend cannot honor them), tools / - ``parallel_tool_calls`` / streaming SSE work across backends, and encrypted - reasoning replay round-trips on OpenAI-served models — so the generic - Responses transport path needs no Router-specific request surgery. - -The capability cache mirrors the OpenRouter reasoning-caps design in -``hermes_cli/models.py``: cache-only lookups on the per-request hot path -(never HTTP), seeded for free whenever ``fetch_models()`` runs (picker, -setup, doctor), hydrated from a disk mirror across processes, and refreshed -by a background warmer when cold or stale. +Wire notes (verified live against api.router.com): +* Responses API is the native wire; ``/chat/completions`` is only a thin shim. + ``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate in + ``hermes_cli/providers.py`` keep every path on it. +* The catalog is account-scoped (BYOK accounts see extra IDs), so this profile + ships no ``fallback_models`` — the picker relies on ``fetch_models()``. +* Router 400s on ``reasoning.effort`` levels outside a model's published + vocabulary and on any reasoning field for non-reasoning models. The efforts + map from ``GET /v1/models`` is cached (memory + disk mirror, background + warmer; never HTTP on the request hot path) and fed to the codex transport's + clamp via ``supported_reasoning_efforts``. +* ``store: false``, ``prompt_cache_key``, encrypted reasoning replay, tools and + streaming pass through unchanged — no Router-specific request surgery. """ from __future__ import annotations @@ -52,6 +25,7 @@ import time from pathlib import Path from typing import Any, Optional +from agent.reasoning_effort import EFFORT_LADDER from hermes_cli import __version__ as _HERMES_VERSION from providers import register_provider from providers.base import ProviderProfile, _profile_user_agent @@ -60,43 +34,32 @@ logger = logging.getLogger(__name__) ROUTER_DEFAULT_BASE_URL = "https://api.router.com/v1" -#: Efforts-by-model cache: ``model id -> list of accepted effort levels``. -#: ``[]`` means the catalog says the model accepts NO reasoning parameters -#: (``reasoning.supported: false``) — the transport must omit reasoning -#: entirely. A model absent from the dict is unknown (custom/BYOK route or -#: vocabulary not published) and callers fall back to their defaults. +#: model id -> accepted effort levels. ``[]`` = model accepts NO reasoning +#: fields; absent = unknown (callers keep their defaults). _efforts_cache: Optional[dict[str, list[str]]] = None _efforts_lock = threading.Lock() _warm_started = False _disk_checked = False -#: Disk-mirror staleness bound. Vocabularies change rarely; a stale verdict -#: beats no verdict, so a past-TTL mirror is still served while a background -#: refresh runs (same policy as the OpenRouter caps mirror). +# A stale verdict beats no verdict: a past-TTL mirror is still served while a +# background refresh runs. _DISK_TTL_SECONDS = 24 * 60 * 60 def _base_url() -> str: - """Allow a base-URL override via ``RAMP_ROUTER_BASE_URL``.""" return os.getenv("RAMP_ROUTER_BASE_URL", "").strip().rstrip("/") or ROUTER_DEFAULT_BASE_URL def _resolve_api_key() -> str: - """Resolve the Router key from .env / environment, preferring dotenv. - - ``RAMP_ROUTER_API_KEY`` is Router's documented variable; - ``ROUTER_API_KEY`` is accepted as a convenience alias. Falls back to the - raw environment when the hermes_cli helper is unavailable (e.g. stripped - test environments). - """ - resolvers = [] + """Resolve the Router key (documented var, then alias), preferring dotenv; + plain os.environ is the fallback when the dotenv resolver is unavailable or raises.""" + resolvers: list = [lambda var: os.environ.get(var, "")] try: from hermes_cli.config import get_env_value_prefer_dotenv - resolvers.append(get_env_value_prefer_dotenv) + resolvers.insert(0, get_env_value_prefer_dotenv) except Exception: pass - resolvers.append(lambda var: os.environ.get(var, "")) for resolve in resolvers: for var in ("RAMP_ROUTER_API_KEY", "ROUTER_API_KEY"): try: @@ -108,63 +71,45 @@ def _resolve_api_key() -> str: return "" -def _parse_efforts(items: Any) -> Optional[dict[str, list[str]]]: - """Parse a Router ``/v1/models`` ``data`` array into the efforts map. +def _dig(obj: Any, *keys: str) -> Any: + """Nested dict lookup; None as soon as a level is missing or not a dict.""" + for key in keys: + obj = obj.get(key) if isinstance(obj, dict) else None + return obj - Returns None when the array has no usable entries, which callers treat - as a failed fetch rather than caching an empty verdict. + +def _parse_efforts(items: Any) -> Optional[dict[str, list[str]]]: + """Parse a ``/v1/models`` ``data`` array into the efforts map (None if unusable). + + Ladder-unknown levels are dropped: clamp_effort ignores them, so an + all-unknown vocabulary would pass the effort through unclamped to a Router + 400. ``supported=True`` with no recognized level leaves the model out + (unknown) so the transport keeps its default clamp behavior. """ if not isinstance(items, list): return None - try: - from agent.reasoning_effort import EFFORT_LADDER - - known_levels = set(EFFORT_LADDER) - except Exception: - known_levels = None efforts_by_id: dict[str, list[str]] = {} for item in items: - if not isinstance(item, dict): - continue - mid = str(item.get("id") or "").strip() - if not mid: - continue - router_meta = item.get("router") - reasoning = None - if isinstance(router_meta, dict): - capabilities = router_meta.get("capabilities") - if isinstance(capabilities, dict): - reasoning = capabilities.get("reasoning") - if not isinstance(reasoning, dict): + mid = str(item.get("id") or "").strip() if isinstance(item, dict) else "" + reasoning = _dig(item, "router", "capabilities", "reasoning") + if not mid or not isinstance(reasoning, dict): continue if reasoning.get("supported") is False: - # Definitive negative: any reasoning field 400s on this model. efforts_by_id[mid] = [] continue - levels = [ - str(entry.get("value") or "").strip() - for entry in reasoning.get("efforts") or [] - if isinstance(entry, dict) and str(entry.get("value") or "").strip() - ] - if known_levels is not None: - # clamp_effort silently ignores ladder-unknown levels, and an - # all-unknown vocabulary would pass the requested effort through - # unclamped straight to a Router 400 — so a new vendor tier is - # dropped at ingest and fails loudly here instead. - unknown = [level for level in levels if level not in known_levels] - if unknown: - logger.info( - "router: model %s publishes unrecognized reasoning effort " - "level(s) %s; ignoring them (update agent/reasoning_effort " - "EFFORT_LADDER to adopt new vendor tiers)", - mid, - unknown, - ) - levels = [level for level in levels if level in known_levels] + values = [str(e.get("value") or "").strip() for e in reasoning.get("efforts") or [] if isinstance(e, dict)] + levels = [v for v in values if v] + unknown = [level for level in levels if level not in EFFORT_LADDER] + if unknown: + logger.info( + "router: model %s publishes unrecognized reasoning effort " + "level(s) %s; ignoring them (update agent/reasoning_effort " + "EFFORT_LADDER to adopt new vendor tiers)", + mid, unknown, + ) + levels = [level for level in levels if level in EFFORT_LADDER] if levels: efforts_by_id[mid] = levels - # supported=True with no (recognized) vocabulary -> leave the model - # out (unknown), so the transport keeps its default clamp behavior. return efforts_by_id or None @@ -184,16 +129,14 @@ def _save_disk(efforts_by_id: dict[str, list[str]]) -> None: try: path.parent.mkdir(parents=True, exist_ok=True) tmp = path.with_suffix(".tmp") - tmp.write_text( - json.dumps({"ts": time.time(), "efforts": efforts_by_id}), - encoding="utf-8", - ) + tmp.write_text(json.dumps({"ts": time.time(), "efforts": efforts_by_id}), encoding="utf-8") tmp.replace(path) except Exception as exc: logger.debug("router: caps disk mirror write failed: %s", exc) def _load_disk() -> tuple[Optional[dict[str, list[str]]], float]: + """Disk mirror -> (efforts map or None, age in seconds; TTL when ``ts`` is unparseable).""" path = _disk_path() if path is None: return None, 0.0 @@ -202,11 +145,7 @@ def _load_disk() -> tuple[Optional[dict[str, list[str]]], float]: efforts = data.get("efforts") if not isinstance(efforts, dict) or not efforts: return None, 0.0 - parsed = { - str(mid): [str(level) for level in levels] - for mid, levels in efforts.items() - if isinstance(levels, list) - } + parsed = {str(mid): [str(lv) for lv in levels] for mid, levels in efforts.items() if isinstance(levels, list)} try: age = max(0.0, time.time() - float(data.get("ts") or 0)) except (TypeError, ValueError): @@ -220,11 +159,10 @@ def _seed_efforts(items: Any) -> Optional[dict[str, list[str]]]: """Seed memory + disk caches from a ``/v1/models`` payload.""" global _efforts_cache parsed = _parse_efforts(items) - if parsed is None: - return None - with _efforts_lock: - _efforts_cache = parsed - _save_disk(parsed) + if parsed is not None: + with _efforts_lock: + _efforts_cache = parsed + _save_disk(parsed) return parsed @@ -232,17 +170,16 @@ def _fetch_catalog_items( *, api_key: str = "", base_url: str = "", timeout: float = 8.0 ) -> Optional[list]: """Fetch the raw ``/v1/models`` ``data`` array. None on any failure.""" - url = (base_url or _base_url()).rstrip("/") + "/models" import urllib.request from hermes_cli.urllib_security import open_credentialed_url - req = urllib.request.Request(url) + req = urllib.request.Request((base_url or _base_url()).rstrip("/") + "/models") key = api_key or _resolve_api_key() if key: req.add_header("Authorization", f"Bearer {key}") req.add_header("Accept", "application/json") - # Router sits behind a WAF that rejects the default Python-urllib UA. + # Router's WAF rejects the default Python-urllib UA. req.add_header("User-Agent", _profile_user_agent()) try: with open_credentialed_url(req, timeout=timeout) as resp: @@ -255,14 +192,12 @@ def _fetch_catalog_items( def _efforts_cache_only() -> Optional[dict[str, list[str]]]: - """Memory, else the disk mirror. Never HTTP (hot-path safe).""" + """Memory, else the disk mirror (checked once per process). Never HTTP (hot-path safe).""" global _efforts_cache, _disk_checked with _efforts_lock: cached = _efforts_cache - if cached is not None: + if cached is not None or _disk_checked: return cached - if _disk_checked: - return None _disk_checked = True parsed, age = _load_disk() if parsed is None: @@ -277,19 +212,19 @@ def _efforts_cache_only() -> Optional[dict[str, list[str]]]: def _warm_efforts_async() -> None: - """Refresh the efforts cache in the background, at most once per process.""" + """Refresh the efforts cache in the background, at most once per process. + + Skipped under pytest (a mid-suite fetch makes cache state timing-dependent) + and without a key (it would 401; the first authenticated fetch_models() seeds). + """ global _warm_started if os.environ.get("PYTEST_CURRENT_TEST"): - # Match the canonical caps warmer (hermes_cli/models.py): a mid-suite - # background fetch would make cache state timing-dependent in tests. return with _efforts_lock: if _warm_started: return _warm_started = True if not _resolve_api_key(): - # Without a key the fetch would 401; the first authenticated - # fetch_models() (picker/setup/doctor) seeds the cache instead. return def _refresh() -> None: @@ -298,9 +233,7 @@ def _warm_efforts_async() -> None: _seed_efforts(items) try: - threading.Thread( - target=_refresh, name="router-caps-warm", daemon=True - ).start() + threading.Thread(target=_refresh, name="router-caps-warm", daemon=True).start() except Exception as exc: logger.debug("router: caps warmer failed to start: %s", exc) @@ -309,46 +242,19 @@ class RouterProfile(ProviderProfile): """Ramp Router — Responses-only gateway with catalog-declared efforts.""" def fetch_models( - self, - *, - api_key: Optional[str] = None, - base_url: Optional[str] = None, - timeout: float = 8.0, + self, *, api_key: Optional[str] = None, base_url: Optional[str] = None, timeout: float = 8.0 ) -> Optional[list[str]]: - """Fetch the live, key-scoped catalog and seed the caps cache. - - One request serves both consumers: the picker gets the model IDs and - the reasoning-vocabulary mirror is left warm at no extra network - cost (the same document carries both). - """ - items = _fetch_catalog_items( - api_key=api_key or "", base_url=base_url or "", timeout=timeout - ) + """Fetch the live, key-scoped catalog; the same payload seeds the caps cache. + Deduped but not sorted: Router's listing order is deliberate presentation.""" + items = _fetch_catalog_items(api_key=api_key or "", base_url=base_url or "", timeout=timeout) if items is None: return None _seed_efforts(items) - # Deduped but not sorted: Router's listing order is deliberate - # presentation (featured/current models first), so the picker keeps it. - ids = list( - dict.fromkeys( - str(item["id"]) - for item in items - if isinstance(item, dict) and item.get("id") - ) - ) + ids = list(dict.fromkeys(str(i["id"]) for i in items if isinstance(i, dict) and i.get("id"))) return ids or None - def supported_reasoning_efforts( - self, model: Optional[str] - ) -> Optional[tuple[str, ...]]: - """Catalog-declared effort vocabulary for *model* (cache-only). - - Router 400s on efforts outside a model's published set and on any - reasoning field for non-reasoning models, so the codex transport - clamps (or suppresses) from this verdict. Cold cache returns None — - the transport keeps its defaults — and kicks a background warmer so - the next turn is covered. - """ + def supported_reasoning_efforts(self, model: Optional[str]) -> Optional[tuple[str, ...]]: + """Catalog-declared effort vocabulary (cache-only; cold cache -> None + warm).""" mid = str(model or "").strip() if not mid: return None @@ -356,10 +262,7 @@ class RouterProfile(ProviderProfile): if efforts_by_id is None: _warm_efforts_async() return None - levels = efforts_by_id.get(mid) - if levels is None: - return None - return tuple(levels) + return None if mid not in efforts_by_id else tuple(efforts_by_id[mid]) router = RouterProfile( @@ -369,26 +272,14 @@ router = RouterProfile( display_name="Ramp Router", description="Ramp Router (router.com) — routes each request to the cheapest model that clears your quality bar", signup_url="https://app.router.com/keys", - # RAMP_ROUTER_API_KEY is Router's documented variable; ROUTER_API_KEY is - # a convenience alias. RAMP_ROUTER_BASE_URL overrides the endpoint - # (auth.py picks it up as the registry's base_url_env_var). env_vars=("RAMP_ROUTER_API_KEY", "ROUTER_API_KEY", "RAMP_ROUTER_BASE_URL"), base_url=_base_url(), auth_type="api_key", - # Identify Hermes traffic to the gateway (Router attributes coding-agent - # clients by User-Agent prefix, the way it already recognizes OpenCode's - # versioned UA) — and Router's WAF rejects blank/default client UAs. + # Router attributes coding-agent clients by UA prefix; its WAF rejects default UAs. default_headers={"User-Agent": f"Hermes-Agent/{_HERMES_VERSION}"}, - # Most of the catalog's frontier routes accept image input; capability is - # still model-dependent and governed by the live catalog. supports_vision=True, - # Cheap, reasoning-capable, and vision-capable — safe for auxiliary tasks - # (compaction, titles, vision) when Router is the main provider. Also the - # model Router's own docs use as their example. default_aux_model="gpt-5.4-mini", - # Deliberately empty: model IDs are account-scoped (BYOK accounts see - # extra entries) and Router's docs say to read the catalog at runtime - # rather than hardcode names. The picker uses fetch_models() above. + # Empty on purpose: model IDs are account-scoped; the picker uses fetch_models(). fallback_models=(), ) diff --git a/plugins/model-providers/upstage/__init__.py b/plugins/model-providers/upstage/__init__.py index 1d6f5749f0..186671e164 100644 --- a/plugins/model-providers/upstage/__init__.py +++ b/plugins/model-providers/upstage/__init__.py @@ -1,102 +1,49 @@ -"""Upstage Solar provider profile.""" +"""Upstage Solar provider profile: top-level ``reasoning_effort`` (low|medium|high). + +Solar's server default is ``minimal`` (reasoning off) — wrong for agentic work — +so an unset reasoning_config defaults reasoning ON at ``medium``, matching the +"medium (default)" the /reasoning panel shows. Explicit settings always win. +""" from typing import Any +from agent.reasoning_effort import EFFORT_LADDER, SOLAR_EFFORTS, clamp_effort from providers import register_provider from providers.base import ProviderProfile - -# Model-name markers for Solar families that do NOT accept ``reasoning_effort``. -# Deny-list on purpose: newly released Solar models are assumed -# reasoning-capable by default, so only the known non-reasoning families are -# listed here. Substring match (not startswith) so dated variants like -# ``solar-mini-250127`` are covered too. +# Deny-list on purpose: new Solar models are assumed reasoning-capable; only +# these known non-reasoning families ignore reasoning_effort. Substring match +# so dated variants (``solar-mini-250127``) are covered. _NON_REASONING_MODEL_MARKERS = ("solar-mini", "syn-pro") -# When the user hasn't picked a reasoning effort, Hermes passes -# reasoning_config=None. Solar's own server default is "minimal" (reasoning -# off), which is the wrong default for an agentic workload. We default reasoning -# ON at this effort — matching the "medium (default)" that Hermes' /reasoning -# panel shows for an unset config, so the displayed default and the real wire -# value agree. An explicit saved setting or a `/reasoning ` change is -# always honored over this default; `/reasoning none` disables it. -_DEFAULT_REASONING_EFFORT = "medium" - - -def _model_supports_reasoning(model: str | None) -> bool: - """Solar reasoning-capable models — True unless the model is deny-listed. - - The Solar Pro family (``solar-pro``, ``solar-pro2``, ``solar-pro3`` and - dated variants like ``solar-pro3-250127``) and the Solar Open family - (``solar-open*``) accept ``reasoning_effort``; only ``solar-mini`` / - ``syn-pro`` ignore the parameter, so we deny-list those and treat every - other (incl. future) Solar model as reasoning-capable. - - ``None``/empty model → True: the provider default (``fallback_models[0]``, - ``solar-pro3``) is reasoning-capable, so an unset model gets the same - default-on behaviour. - """ - m = (model or "").strip().lower() - return not any(marker in m for marker in _NON_REASONING_MODEL_MARKERS) - class UpstageProfile(ProviderProfile): - """Upstage Solar — top-level ``reasoning_effort`` control. - - Solar Pro/Open expose reasoning through a top-level ``reasoning_effort`` - field (``minimal`` | ``low`` | ``medium`` | ``high``), mirroring OpenAI's - shape. Unlike DeepSeek/Kimi it does NOT require echoing ``reasoning_content`` - back on later turns, so only the request field needs wiring. We emit at most - ``low`` | ``medium`` | ``high`` — the explicit values both Solar Pro 2 and - Pro 3 accept. - - Default-on: Solar's own server default is ``minimal`` (off), but for an - agentic workload we default reasoning ON (``_DEFAULT_REASONING_EFFORT``) - when the user hasn't picked an effort. The user can still set any level or - turn it off with ``/reasoning none``. - """ + """Upstage Solar — top-level ``reasoning_effort`` control (no reasoning_content echo needed).""" def build_api_kwargs_extras( self, *, reasoning_config: dict | None = None, model: str | None = None, **context ) -> tuple[dict[str, Any], dict[str, Any]]: - top_level: dict[str, Any] = {} - - # solar-mini / syn-pro (the deny-list) ignore reasoning_effort — send - # nothing. Everything else, including future Solar models, gets it. - if not _model_supports_reasoning(model): - return {}, top_level - - # Unset (reasoning_config is None) → default reasoning ON for agents. + m = (model or "").strip().lower() + if any(marker in m for marker in _NON_REASONING_MODEL_MARKERS): + return {}, {} + # Unset -> default reasoning ON for agents. if not reasoning_config or not isinstance(reasoning_config, dict): - return {}, {"reasoning_effort": _DEFAULT_REASONING_EFFORT} - - # Explicitly disabled (`/reasoning none`) → omit the field so Solar - # applies its own default (minimal = off). + return {}, {"reasoning_effort": "medium"} + # Explicitly disabled -> omit so Solar applies its own default (minimal = off). if reasoning_config.get("enabled") is False: - return {}, top_level - - # Map Hermes' effort vocabulary onto Solar's accepted set via the - # shared clamp (agent.reasoning_effort). minimal → omit (Solar's - # minimal means off); unknown-but-enabled bespoke levels collapse to - # high rather than silently downgrading (#62650 precedent). + return {}, {} effort = (reasoning_config.get("effort") or "").strip().lower() if not effort: - top_level["reasoning_effort"] = _DEFAULT_REASONING_EFFORT - return {}, top_level + return {}, {"reasoning_effort": "medium"} if effort == "minimal": - return {}, top_level - - from agent.reasoning_effort import EFFORT_LADDER, SOLAR_EFFORTS, clamp_effort - + return {}, {} mapped = clamp_effort(effort, SOLAR_EFFORTS) if mapped not in SOLAR_EFFORTS: - # Bespoke level outside the ladder — Solar precedent is to run - # at full strength rather than quietly fall to the default. + # Bespoke level outside the ladder runs at full strength rather + # than quietly falling to the default; ladder levels that still + # don't map are omitted. mapped = "high" if effort not in EFFORT_LADDER else None - - if mapped: - top_level["reasoning_effort"] = mapped - return {}, top_level + return {}, {"reasoning_effort": mapped} if mapped else {} upstage = UpstageProfile( @@ -108,11 +55,8 @@ upstage = UpstageProfile( env_vars=("UPSTAGE_API_KEY", "UPSTAGE_BASE_URL"), base_url="https://api.upstage.ai/v1", auth_type="api_key", - # default_aux_model left empty → auxiliary side tasks use the main model. - # entry [0] is the setup default — solar-pro3, the current Solar Pro flagship. - fallback_models=( - "solar-pro3", - ), + # No default_aux_model: auxiliary tasks use the main model. [0] is the setup default. + fallback_models=("solar-pro3",), ) register_provider(upstage) diff --git a/plugins/model-providers/vertex/__init__.py b/plugins/model-providers/vertex/__init__.py index a63aa2b68c..ef761c2656 100644 --- a/plugins/model-providers/vertex/__init__.py +++ b/plugins/model-providers/vertex/__init__.py @@ -1,20 +1,10 @@ -"""Google Vertex AI provider profile. +"""Google Vertex AI provider profile: Gemini via Google Cloud's OpenAI-compatible +endpoint. -vertex: Gemini models via Google Cloud's OpenAI-compatible endpoint. - -Auth is OAuth2 — short-lived access tokens minted from a service-account JSON -or Application Default Credentials (ADC), NOT a static API key. Token -resolution and refresh live in ``agent/vertex_adapter.py``; runtime_provider.py -calls it to obtain a fresh ``(token, base_url)`` pair, then hands the token to -the standard OpenAI client as ``api_key``. Because the wire format is the -OpenAI-compatible chat/completions surface, no message translation is needed — -the only Gemini-specific concern is the ``thinking_config`` reasoning hook, -which is emitted here exactly as the ``gemini`` provider does for its -OpenAI-compat subpath (``extra_body.google.thinking_config``). - -``auth_type="vertex"`` marks this as an OAuth-token provider (resolved -specially, like bedrock's ``aws_sdk``) so it is never treated as an -api_key provider that would mistake a credentials-file path for a key. +Auth is OAuth2 (service-account JSON or ADC), not a static key: ``agent/ +vertex_adapter.py`` mints ``(token, base_url)`` and the token is passed as +``api_key``. ``auth_type="vertex"`` keeps it out of the api_key provider path so +a credentials-file path is never mistaken for a key. """ from typing import Any @@ -26,39 +16,24 @@ from providers.base import ProviderProfile class VertexProfile(ProviderProfile): """Vertex AI — reuse Gemini's thinking_config translation for extra_body.""" - def build_extra_body( - self, *, session_id: str | None = None, **context: Any - ) -> dict[str, Any]: - """Emit ``extra_body.google.thinking_config`` for the OpenAI-compat - Vertex surface, mirroring the ``gemini`` provider's behavior. - """ + def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]: + """Emit ``extra_body.google.thinking_config`` like the ``gemini`` provider's + OpenAI-compat subpath.""" from agent.transports.chat_completions import ( _build_gemini_thinking_config, _snake_case_gemini_thinking_config, ) - model = context.get("model") or "" - reasoning_config = context.get("reasoning_config") - - raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config) - if not raw_thinking_config: - return {} - - thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config) + raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config")) + thinking_config = _snake_case_gemini_thinking_config(raw) if raw else None if not thinking_config: return {} return {"extra_body": {"google": {"thinking_config": thinking_config}}} def fetch_models( - self, - *, - api_key: str | None = None, - base_url: str | None = None, - timeout: float = 8.0, + self, *, api_key: str | None = None, base_url: str | None = None, timeout: float = 8.0 ) -> list[str] | None: - """Vertex's OpenAI-compat endpoint has no ``/models`` listing route; - model discovery is not available. The setup wizard ships a curated list. - """ + """No ``/models`` route on the OpenAI-compat endpoint; setup ships a curated list.""" return None diff --git a/plugins/model-providers/zai/__init__.py b/plugins/model-providers/zai/__init__.py index 5038e55503..11a57c3dc0 100644 --- a/plugins/model-providers/zai/__init__.py +++ b/plugins/model-providers/zai/__init__.py @@ -1,27 +1,7 @@ """ZAI / GLM provider profile. -Z.AI's GLM-4.5-and-later chat models default to thinking-mode ON when the -request omits ``thinking``. Hermes' ``reasoning_config = {"enabled": False}`` -was previously a silent no-op on this route — the base profile emits nothing, -so users who turned thinking off (desktop toggle, ``/reasoning none``, -``reasoning_effort: none``/``false`` in config.yaml) kept burning thinking -tokens on every turn. - -:meth:`ZaiProfile.build_api_kwargs_extras` translates the Hermes reasoning -config into the wire shape Z.AI's OpenAI-compat endpoint expects: - - {"extra_body": {"thinking": {"type": "enabled" | "disabled"}}} - -When no reasoning preference is set (``reasoning_config is None``) the field -is omitted so the server default applies, matching prior behavior. GLM -models before 4.5 (e.g. ``glm-4-9b``) don't accept ``thinking`` and are left -untouched. - -GLM-5.2 additionally exposes a native ``reasoning_effort`` knob with exactly -two enabled levels — ``high`` and ``max`` — on the OpenAI-compatible endpoint -(per Z.AI / BigModel docs). Hermes' richer effort scale is collapsed onto -those two so the user's effort preference actually reaches the model instead -of being silently dropped. +GLM-4.5+ defaults to thinking ON, so ``reasoning_config`` is translated to +``extra_body.thinking``; GLM-5.2/5.3 also take a native ``reasoning_effort``. """ from __future__ import annotations @@ -29,93 +9,40 @@ from __future__ import annotations import re from typing import Any +from agent import reasoning_effort as re_ from providers import register_provider from providers.base import ProviderProfile _GLM_VERSION_RE = re.compile(r"^glm-(\d+)(?:\.(\d+))?") +# Alias spellings seen on relays (Fireworks ``glm-5p2``, ``zai-org-glm-5-2``…). +_GLM_5_3_TOKENS = ("glm-5.3", "glm-5-3", "glm-5p3") +_GLM_5_2_TOKENS = ("glm-5.2", "glm-5-2", "glm-5p2") + _GLM_5_3_TOKENS def _model_supports_thinking(model: str | None) -> bool: """GLM thinking-capable model families: glm-4.5 and later (4.5, 4.6, 5…).""" + match = _GLM_VERSION_RE.match((model or "").strip().lower()) + return bool(match) and (int(match.group(1)), int(match.group(2) or 0)) >= (4, 5) + + +def _has_token(model: str | None, tokens: tuple[str, ...]) -> bool: m = (model or "").strip().lower() - match = _GLM_VERSION_RE.match(m) - if not match: - return False - major = int(match.group(1)) - minor = int(match.group(2) or 0) - return (major, minor) >= (4, 5) + return bool(m) and any(token in m for token in tokens) -def _is_glm_5_2(model: str | None) -> bool: - """Detect GLM-5.2/5.3 (reasoning_effort-capable) across alias spellings. +def _glm_5_2_reasoning_effort(reasoning_config: dict | None, *, model: str | None = None) -> str | None: + """Map Hermes effort onto GLM's vocabulary (5.2: high/max; 5.3: low..max). - Covers the canonical ``glm-5.2``/``glm-5.3`` plus the ``glm-5-2`` / - ``glm-5p2`` variants seen on relays (Fireworks ``glm-5p2``, etc.) and any - vendor-prefixed form (``z-ai/glm-5.2``, ``zai-org-glm-5-2``). GLM-5.3 - uses the same base model as 5.2 (post-training gains only) and exposes - the same ``reasoning_effort`` knob (verified live 2026-08-14: the - coding-plan endpoint accepts ``reasoning_effort: high`` for glm-5.3). + Below-floor efforts clamp to the floor; disabled/unset leaves the server default. """ - m = (model or "").strip().lower() - if not m: - return False - return any( - token in m - for token in ("glm-5.2", "glm-5-2", "glm-5p2", "glm-5.3", "glm-5-3", "glm-5p3") - ) - - -def _is_glm_5_3(model: str | None) -> bool: - """Detect GLM-5.3 specifically — it has a wider effort vocabulary. - - 5.2 accepts only ``high``/``max``; 5.3 accepts a graded - ``low``/``medium``/``high``/``max`` scale (verified live, issue #91789), - so effort mapping must pick the vocabulary per model. - """ - m = (model or "").strip().lower() - if not m: - return False - return any(token in m for token in ("glm-5.3", "glm-5-3", "glm-5p3")) - - -def _glm_5_2_reasoning_effort( - reasoning_config: dict | None, *, model: str | None = None -) -> str | None: - """Map Hermes reasoning effort onto GLM's native vocabulary. - - GLM-5.2 supports two enabled effort levels (``high``/``max``); - GLM-5.3 supports the graded ``low``/``medium``/``high``/``max`` scale. - ``xhigh``/``max``/``ultra`` request the top tier; anything below the - model's floor clamps to that floor. When reasoning is explicitly - disabled, or no effort preference is supplied, the server default is - left untouched. - """ - if not isinstance(reasoning_config, dict): + effort = re_.requested_effort(reasoning_config) + if effort is None or effort == "none": return None - if reasoning_config.get("enabled") is False: - return None - - effort = (reasoning_config.get("effort") or "").strip().lower() - if not effort or effort == "none": - return None - - # Per-model vocabulary declared in agent.reasoning_effort; xhigh rounds - # up to max on both. 5.2 cannot think less than high; 5.3 accepts a - # graded scale down to low (issue #91789). - from agent.reasoning_effort import ( - GLM52_EFFORTS, - GLM52_OVERRIDES, - GLM53_EFFORTS, - GLM53_OVERRIDES, - clamp_effort, - ) - - if _is_glm_5_3(model): - efforts, overrides, floor = GLM53_EFFORTS, GLM53_OVERRIDES, "low" + if _has_token(model, _GLM_5_3_TOKENS): + efforts, overrides, floor = re_.GLM53_EFFORTS, re_.GLM53_OVERRIDES, "low" else: - efforts, overrides, floor = GLM52_EFFORTS, GLM52_OVERRIDES, "high" - - clamped = clamp_effort(effort, efforts, overrides) + efforts, overrides, floor = re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES, "high" + clamped = re_.clamp_effort(effort, efforts, overrides) return clamped if clamped in efforts else floor @@ -127,21 +54,19 @@ class ZaiProfile(ProviderProfile): ) -> tuple[dict[str, Any], dict[str, Any]]: extra_body: dict[str, Any] = {} top_level: dict[str, Any] = {} - - if not _model_supports_thinking(model) and not _is_glm_5_2(model): + is_5_2 = _has_token(model, _GLM_5_2_TOKENS) + if not _model_supports_thinking(model) and not is_5_2: return extra_body, top_level - # Only emit when the user expressed a preference; omitting the field - # keeps the server default (enabled) exactly as before. + # Only emit when the user expressed a preference (server default = enabled). if isinstance(reasoning_config, dict): enabled = reasoning_config.get("enabled") is not False extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"} - if _is_glm_5_2(model): + if is_5_2: effort = _glm_5_2_reasoning_effort(reasoning_config, model=model) if effort is not None: top_level["reasoning_effort"] = effort - return extra_body, top_level @@ -152,11 +77,7 @@ zai = ZaiProfile( display_name="Z.AI (GLM)", description="Z.AI / GLM — Zhipu AI models", signup_url="https://z.ai/", - fallback_models=( - "glm-5.2", - "glm-5", - "glm-4-9b", - ), + fallback_models=("glm-5.2", "glm-5", "glm-4-9b"), base_url="https://api.z.ai/api/paas/v4", default_aux_model="glm-4.5-flash", )