fix(runtime): per-model api_mode + /v1 healing for custom OpenCode-family providers

The named-custom-provider runtime path returned a static api_mode, so a
providers: entry like opencode-go-bridge -> https://opencode.ai/zen/go/v1
sent responses-only models (grok-4.5, gpt-5.6-luna) to /chat/completions
and got HTTP 503 (#85589 repro). Now: when the provider name is in the
OpenCode family or the base_url is hosted on opencode.ai, derive api_mode
from the effective model and run the symmetric /v1 normalization — unless
the user declared an explicit transport, which stays authoritative.

5 new regression tests against a real temp HERMES_HOME config.
This commit is contained in:
Teknium
2026-08-20 20:11:12 -07:00
parent a83c3915a3
commit 49780391a2
2 changed files with 118 additions and 0 deletions
+39
View File
@@ -1098,6 +1098,7 @@ def _resolve_named_custom_runtime(
requested_provider: str,
explicit_api_key: Optional[str] = None,
explicit_base_url: Optional[str] = None,
target_model: Optional[str] = None,
) -> Optional[Dict[str, Any]]:
# Bare `provider="custom"` with an explicit base_url (e.g. propagated
# from a `model_aliases:` direct-alias resolution) — build a runtime
@@ -1238,6 +1239,43 @@ def _resolve_named_custom_runtime(
request_overrides = _custom_provider_request_overrides(custom_provider)
if request_overrides:
result["request_overrides"] = request_overrides
# Custom providers in the OpenCode family (name extends opencode-go/zen,
# or base_url hosted on opencode.ai) serve models behind different API
# surfaces per model — a static api_mode 503s for /v1/responses-only
# models like grok-4.5 (#85589). Re-derive api_mode from the effective
# model and normalize the /v1 suffix, exactly like the built-in
# opencode-zen/go paths do.
from hermes_cli.models import opencode_provider_family
_oc_family = opencode_provider_family(requested_provider)
if _oc_family is None:
try:
from utils import base_url_hostname
if base_url_hostname(base_url).lower() == "opencode.ai":
_oc_family = (
"opencode-go" if "/zen/go" in base_url.lower() else "opencode-zen"
)
except Exception:
_oc_family = None
if _oc_family is not None and not custom_provider.get("api_mode"):
from hermes_cli.models import (
normalize_opencode_base_url,
opencode_model_api_mode,
)
_effective_model = str(
target_model
or custom_provider.get("model")
or _get_model_config().get("default")
or ""
).strip()
if _effective_model:
result["api_mode"] = opencode_model_api_mode(_oc_family, _effective_model)
result["base_url"] = normalize_opencode_base_url(
_oc_family, result["api_mode"], result["base_url"]
)
return result
@@ -1847,6 +1885,7 @@ def resolve_runtime_provider(
requested_provider=requested_provider,
explicit_api_key=explicit_api_key,
explicit_base_url=explicit_base_url,
target_model=target_model,
)
if custom_runtime:
custom_runtime["requested_provider"] = requested_provider
@@ -0,0 +1,79 @@
"""Custom OpenCode-family provider runtime resolution (#85589).
A user-defined provider whose name extends an OpenCode family slug
(``opencode-go-bridge``) — or whose base_url is hosted on opencode.ai —
must get the same per-model api_mode derivation and /v1 base-url
normalization as the built-in ``opencode-go`` / ``opencode-zen``
providers. Before the fix, such providers resolved with a static
api_mode (chat_completions), so /v1/responses-only models like
``grok-4.5`` failed with HTTP 503 "Endpoint unavailable".
"""
import os
import tempfile
import pytest
@pytest.fixture()
def _bridge_env(monkeypatch, tmp_path):
hermes_home = tmp_path / ".hermes"
hermes_home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(hermes_home))
monkeypatch.setenv("OPENCODE_GO_BRIDGE_API_KEY", "sk-test-bridge")
def write_config(provider_name: str, extra: str = "", default: str = "grok-4.5"):
(hermes_home / "config.yaml").write_text(
f"""
model:
provider: {provider_name}
default: {default}
providers:
{provider_name}:
base_url: https://opencode.ai/zen/go/v1
key_env: OPENCODE_GO_BRIDGE_API_KEY
{extra}"""
)
return write_config
def _resolve(provider, model):
from hermes_cli.runtime_provider import resolve_runtime_provider
return resolve_runtime_provider(requested=provider, target_model=model)
class TestCustomOpencodeFamilyRuntime:
def test_family_provider_routes_grok_to_responses(self, _bridge_env):
"""The #85589 repro: grok-4.5 on opencode-go-bridge → codex_responses."""
_bridge_env("opencode-go-bridge")
r = _resolve("opencode-go-bridge", "grok-4.5")
assert r["api_mode"] == "codex_responses"
assert r["base_url"].rstrip("/").endswith("/zen/go/v1")
def test_family_provider_routes_minimax_to_messages_and_strips_v1(self, _bridge_env):
_bridge_env("opencode-go-bridge")
r = _resolve("opencode-go-bridge", "minimax-m2.7")
assert r["api_mode"] == "anthropic_messages"
# /v1 stripped so the Anthropic SDK's own /v1/messages doesn't double up
assert r["base_url"].rstrip("/").endswith("/zen/go")
def test_family_provider_keeps_chat_models_on_v1(self, _bridge_env):
_bridge_env("opencode-go-bridge")
r = _resolve("opencode-go-bridge", "deepseek-v4-flash")
assert r["api_mode"] == "chat_completions"
assert r["base_url"].rstrip("/").endswith("/zen/go/v1")
def test_explicit_transport_in_config_wins(self, _bridge_env):
"""A user-declared transport is authoritative — no family override."""
_bridge_env("opencode-go-bridge", extra=" transport: chat_completions\n")
r = _resolve("opencode-go-bridge", "grok-4.5")
assert r["api_mode"] == "chat_completions"
def test_arbitrary_name_on_opencode_host_derives_family(self, _bridge_env):
"""A provider with a non-family name pointing at opencode.ai still
gets per-model routing (host match)."""
_bridge_env("my-oc-proxy", default="gpt-5.6-luna")
r = _resolve("my-oc-proxy", "gpt-5.6-luna")
assert r["api_mode"] == "codex_responses"