Files
hermes-agent/plugins/model-providers/deepseek/__init__.py
T
teknium1 f678ed8299 refactor(model-providers): thinking-toggle XOR effort translation lives in agent.reasoning_effort; opencode-free imports it instead of borrowing via sys.modules
Four chat_completions profiles (kimi-coding, deepseek, opencode-go's Kimi K2 and DeepSeek branches, actual) each hand-rolled the same extra_body.thinking / top-level reasoning_effort translation, and the copies had already drifted in small ways (kimi's `.get("enabled", True)`, deepseek's separate effort parsing). agent.reasoning_effort.thinking_toggle_extras is now the single implementation: the Moonshot default emits effort XOR toggle (both is an HTTP 400), and always_emit_toggle=True covers DeepSeek's contract where the toggle must ride on every request to dodge the reasoning_content echo trap. actual keeps its two contract-specific lines (reasoning_config None -> nothing; effort "none" -> disabled toggle plus reasoning_effort="none", which the relay accepts as a real level) and delegates the rest. ox_alpha_reasoning_extras moves alongside so opencode-free imports it like any other helper instead of reaching into the zen plugin's module through sys.modules and swallowing every exception into ({}, {}) - a failure there previously silently dropped the user's effort setting. No wire behavior changes; tests/plugins/model_providers/test_thinking_toggle_parity.py pins the XOR invariant across the matrix and zen/free parity.
2026-09-13 05:19:48 -07:00

54 lines
2.6 KiB
Python

"""DeepSeek provider profile.
V4 defaults to thinking ON when ``extra_body.thinking`` is unset, and then
requires ``reasoning_content`` to be echoed back on later turns (HTTP 400 after
the first tool call otherwise). This profile sets ``thinking`` explicitly and
maps effort onto DeepSeek's ``reasoning_effort``; V3 models are left untouched.
Retired ``deepseek-chat``/``deepseek-reasoner`` IDs are remapped in
``hermes_cli.model_normalize`` before reaching here.
"""
from typing import Any
from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, thinking_toggle_extras
from providers import register_provider
from providers.base import ProviderProfile
# Version-less canonical ids for thinking-capable DeepSeek models. The 2026-09 Flash
# refresh dropped the ``v<N>`` marker from the public id: ``GET /v1/models`` reports
# ``deepseek-flash`` and the API accepts it directly, so the generation check in
# ``build_api_kwargs_extras`` cannot recognise it.
_THINKING_CAPABLE_IDS: frozenset[str] = frozenset({"deepseek-flash"})
class DeepSeekProfile(ProviderProfile):
"""DeepSeek — extra_body.thinking + top-level reasoning_effort."""
def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
) -> tuple[dict[str, Any], dict[str, Any]]:
m = (model or "").strip().lower()
# v4+ only; v3 excluded. Version-less canonicals (``deepseek-flash``) carry the
# same thinking-mode contract but no ``v<N>`` prefix, so consult the id set too —
# missing them makes Hermes omit ``thinking``, so the server defaults to on and
# the user's thinking toggle / effort setting is silently ignored.
versioned_v4_plus = m.startswith("deepseek-v") and not m.startswith("deepseek-v3")
if not versioned_v4_plus and m not in _THINKING_CAPABLE_IDS:
return {}, {}
# Always set thinking explicitly (default enabled, matching the API default)
# to avoid the reasoning_content echo trap on subsequent turns.
return thinking_toggle_extras(
reasoning_config, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, always_emit_toggle=True
)
deepseek = DeepSeekProfile(
name="deepseek", aliases=("deepseek-chat",), env_vars=("DEEPSEEK_API_KEY",), display_name="DeepSeek",
description="DeepSeek — native DeepSeek API", signup_url="https://platform.deepseek.com/",
fallback_models=("deepseek-v4-pro", "deepseek-flash"), base_url="https://api.deepseek.com/v1",
default_aux_model="deepseek-flash",
)
register_provider(deepseek)