From 139396995a8dcfd767dbdb64e15ed591e3c66d37 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:27:38 -0700 Subject: [PATCH] feat(opencode): send x-opencode-session on every OpenCode request for backend affinity MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit OpenCode pins requests sharing an x-opencode-session value to one upstream backend, which is what keeps its prompt cache warm across a conversation. Hermes never sent it, so cache ratios on OpenCode traffic were poor. - agent/opencode_affinity.py: single owner of the header — target detection (built-in zen/go/free, custom opencode-* providers, any opencode.ai URL) and the key (affinity scope → conversation root → session id, cron timestamp stripped), same resolution as OpenRouter/xAI affinity hints. - build_api_kwargs: merged once after the per-mode builder, so chat_completions, codex_responses and anthropic_messages all carry it. - auxiliary _build_call_kwargs: same key from the runtime-main session so compression/title/vision calls stay on the conversation's backend; the aux Codex and Anthropic adapters now forward extra_headers. Closes #81584, #81832 (deepseek-v4-flash 400 without the header). --- agent/auxiliary_client.py | 19 ++++- agent/chat_completion_helpers.py | 20 ++++- agent/opencode_affinity.py | 84 +++++++++++++++++++ tests/agent/test_opencode_session_affinity.py | 62 ++++++++++++++ website/docs/integrations/providers.md | 2 + 5 files changed, 185 insertions(+), 2 deletions(-) create mode 100644 agent/opencode_affinity.py create mode 100644 tests/agent/test_opencode_session_affinity.py diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 2840868b47..b7b482e918 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1800,6 +1800,10 @@ class _CodexCompletionsAdapter: timeout = kwargs.get("timeout") if timeout is not None: resp_kwargs["timeout"] = timeout + # Per-request HTTP headers (OpenCode session affinity, Copilot + # x-initiator) map to real headers via the SDK kwarg — forward them. + if isinstance(kwargs.get("extra_headers"), dict) and kwargs["extra_headers"]: + resp_kwargs["extra_headers"] = dict(kwargs["extra_headers"]) # Note: the Codex endpoint (chatgpt.com/backend-api/codex) does NOT # support max_output_tokens or temperature — omit to avoid 400 errors. @@ -2553,6 +2557,13 @@ class _AnthropicCompletionsAdapter: from agent.anthropic_adapter import _forbids_sampling_params if not _forbids_sampling_params(model): anthropic_kwargs["temperature"] = temperature + # Per-request HTTP headers (OpenCode session affinity) — the Anthropic + # SDK accepts ``extra_headers`` on messages.create/stream too. + if isinstance(kwargs.get("extra_headers"), dict) and kwargs["extra_headers"]: + anthropic_kwargs["extra_headers"] = { + **(anthropic_kwargs.get("extra_headers") or {}), + **kwargs["extra_headers"], + } # Pass through caller-supplied extra_body so providers behind # Anthropic-compatible gateways receive their per-vendor request @@ -9563,7 +9574,13 @@ def _build_call_kwargs( ): kwargs["_reasoning_config"] = dict(reasoning_config) - return kwargs + # OpenCode relay session affinity — same key as the main turn so + # compression/title/vision calls stay on the conversation's warm backend. + from agent.opencode_affinity import merge_opencode_session_headers + + return merge_opencode_session_headers( + kwargs, provider, base_url, _runtime_main_value("session_id") or None + ) def _validate_llm_response( diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index a8b4098989..8fa0c18f8d 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1980,7 +1980,25 @@ def _reasoning_config_for_wire(agent): def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict: - """Build the keyword arguments dict for the active API mode.""" + """Build the keyword arguments dict for the active API mode. + + Wraps the per-api_mode builder so the OpenCode ``x-opencode-session`` + affinity header rides on every OpenCode request regardless of transport + (chat_completions / codex_responses / anthropic_messages all route + OpenCode models). No-op for every other provider. + """ + from agent.opencode_affinity import merge_opencode_session_headers + + kwargs = _build_api_kwargs_for_mode(agent, api_messages, tools_for_api) + return merge_opencode_session_headers( + kwargs, + getattr(agent, "provider", None), + getattr(agent, "base_url", None), + getattr(agent, "session_id", None), + ) + + +def _build_api_kwargs_for_mode(agent, api_messages: list, tools_for_api: list | None = None) -> dict: # One-shot continuation override — consumed exactly once, on the FIRST # request this call builds (only one api_mode branch runs per invocation). _wire_reasoning_config = _reasoning_config_for_wire(agent) diff --git a/agent/opencode_affinity.py b/agent/opencode_affinity.py new file mode 100644 index 0000000000..4f0cc5522e --- /dev/null +++ b/agent/opencode_affinity.py @@ -0,0 +1,84 @@ +"""``x-opencode-session`` — OpenCode relay session-affinity header. + +OpenCode (opencode.ai Zen/Go/free relay) pins requests that share an +``x-opencode-session`` value to the same upstream backend, which is what +keeps its prompt cache warm across the turns of one conversation. The value +only has to be opaque and consistent per conversation, so it is derived the +same way as the other conversation-affinity hints Hermes already sends +(OpenRouter's sticky ``session_id``, xAI's ``x-grok-conv-id``): the +host-declared routing scope first, then the ambient conversation root, then +the physical session id — normalized through ``_cache_scope_from_session_id`` +so cron fires of one job share a scope. + +Every OpenCode request — main turn on any transport, auxiliary calls +(compression, titles, vision, MoA) — goes through :func:`opencode_session_headers` +so the header cannot drift per code path. +""" + +from __future__ import annotations + +from typing import Any, Optional + +OPENCODE_SESSION_HEADER = "x-opencode-session" + + +def is_opencode_target(provider: Optional[str], base_url: Optional[str]) -> bool: + """True when *provider* or *base_url* addresses the OpenCode relay. + + Matches the built-in opencode-zen/go/free providers, custom + ``opencode--*`` providers, and any base_url hosted on opencode.ai. + """ + try: + from hermes_cli.models import opencode_provider_family + + if opencode_provider_family(provider) is not None: + return True + except Exception: + pass + try: + from agent.anthropic_endpoints import _is_opencode_endpoint + + return _is_opencode_endpoint(str(base_url or "")) + except Exception: + return False + + +def opencode_session_headers( + provider: Optional[str], + base_url: Optional[str], + session_id: Optional[str] = None, +) -> dict[str, str]: + """Return ``{"x-opencode-session": }`` for OpenCode targets, else ``{}``.""" + if not is_opencode_target(provider, base_url): + return {} + try: + from agent.portal_tags import get_affinity_scope, get_conversation_context + from agent.transports.codex import _cache_scope_from_session_id + + key = _cache_scope_from_session_id( + get_affinity_scope() or get_conversation_context() or session_id + ) + except Exception: + key = str(session_id or "") + return {OPENCODE_SESSION_HEADER: key} if key else {} + + +def merge_opencode_session_headers( + kwargs: dict[str, Any], + provider: Optional[str], + base_url: Optional[str], + session_id: Optional[str] = None, +) -> dict[str, Any]: + """Merge the affinity header into ``kwargs["extra_headers"]`` (in place). + + Existing per-request headers win, so a caller-pinned value is preserved. + Non-OpenCode targets are left untouched. + """ + headers = opencode_session_headers(provider, base_url, session_id) + if headers: + existing = kwargs.get("extra_headers") + merged = dict(existing) if isinstance(existing, dict) else {} + for key, value in headers.items(): + merged.setdefault(key, value) + kwargs["extra_headers"] = merged + return kwargs diff --git a/tests/agent/test_opencode_session_affinity.py b/tests/agent/test_opencode_session_affinity.py new file mode 100644 index 0000000000..b2b6cf2536 --- /dev/null +++ b/tests/agent/test_opencode_session_affinity.py @@ -0,0 +1,62 @@ +"""x-opencode-session rides on every OpenCode request, on every transport.""" + +from __future__ import annotations + +import pytest + +from agent import auxiliary_client as aux +from agent.chat_completion_helpers import build_api_kwargs +from run_agent import AIAgent + +_MSGS = [{"role": "user", "content": "hi"}] + + +def _agent(provider, model, base_url, api_mode=None): + agent = AIAgent( + api_key="test-key", + base_url=base_url, + model=model, + provider=provider, + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + session_id="sess-affinity-1", + ) + if api_mode: + agent.api_mode = api_mode + agent._transport = None + agent._anthropic_base_url = base_url + return agent + + +@pytest.mark.parametrize( + "provider, model, base_url, api_mode", + [ + ("opencode-go", "glm-5", "https://opencode.ai/zen/go/v1", None), # chat_completions + ("opencode-go", "gpt-5.6-luna", "https://opencode.ai/zen/go/v1", None), # codex_responses + ("opencode-go", "minimax-m2.7", "https://opencode.ai/zen/go/v1", "anthropic_messages"), + ("opencode-free", "laguna-s-2.1-free", "https://opencode.ai/zen/v1", None), + ("custom", "glm-5", "https://opencode.ai/zen/go/v1", None), # URL-only detection + ], +) +def test_main_turn_sends_stable_session_header_on_every_transport(provider, model, base_url, api_mode): + agent = _agent(provider, model, base_url, api_mode) + first = build_api_kwargs(agent, _MSGS)["extra_headers"]["x-opencode-session"] + second = build_api_kwargs(agent, _MSGS)["extra_headers"]["x-opencode-session"] + assert first == second == "sess-affinity-1" + + other = _agent("openrouter", "anthropic/claude-sonnet-4.6", "https://openrouter.ai/api/v1") + assert "x-opencode-session" not in (build_api_kwargs(other, _MSGS).get("extra_headers") or {}) + + +def test_auxiliary_calls_share_the_main_turn_session_key(): + token = aux.set_runtime_main( + "opencode-go", "glm-5", base_url="https://opencode.ai/zen/go/v1", session_id="sess-affinity-1" + ) + try: + kwargs = aux._build_call_kwargs("opencode-go", "glm-5", _MSGS, base_url="https://opencode.ai/zen/go/v1") + assert kwargs["extra_headers"]["x-opencode-session"] == "sess-affinity-1" + other = aux._build_call_kwargs("openrouter", "x", _MSGS, base_url="https://openrouter.ai/api/v1") + assert "x-opencode-session" not in (other.get("extra_headers") or {}) + finally: + aux._RUNTIME_MAIN_CONTEXT.reset(token) diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 76b9c410a8..20bdbd7f92 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -61,6 +61,8 @@ You need at least one way to connect to an LLM. Use `hermes model` to switch pro | **LM Studio** | `hermes model` → "LM Studio" (provider: `lmstudio`, optional `LM_API_KEY`) | | **Custom Endpoint** | `hermes model` → choose "Custom endpoint" (saved in `config.yaml`) | +All three OpenCode providers send an opaque, per-conversation `x-opencode-session` header on every request (main turns on every transport plus auxiliary calls such as compression and titles). OpenCode uses it to pin a conversation to one backend so its prompt cache stays warm; the value is derived from the Hermes session id and carries no personal data. + For the official API-key path, see the dedicated [Google Gemini guide](/guides/google-gemini). :::tip Model key alias