fix(xai): preserve Grok 4.6 wire capabilities

This commit is contained in:
fangliquan
2026-08-13 04:42:55 +08:00
committed by Teknium
parent d3a8be4630
commit db5e2402c2
5 changed files with 107 additions and 12 deletions
+8
View File
@@ -604,6 +604,14 @@ def grok_supports_reasoning_effort(model: str) -> bool:
return any(name.startswith(prefix) for prefix in _GROK_EFFORT_CAPABLE_PREFIXES)
def is_grok_46_family(model: str) -> bool:
"""Return whether *model* is a Grok 4.6 family identifier."""
name = (model or "").strip().lower().replace("_", "-")
if "/" in name:
name = name.rsplit("/", 1)[-1]
return name == "grok-4.6" or name.startswith("grok-4.6-")
_CONTEXT_LENGTH_KEYS = (
"context_length",
"context_window",
+18 -11
View File
@@ -300,8 +300,13 @@ class ResponsesApiTransport(ProviderTransport):
# Ultra is the Codex product tier; the Responses API wire value is max.
_effort_clamp["ultra"] = "max"
if params.get("is_xai_responses", False):
# xAI Responses tops out at high; keep generic stronger values usable.
_effort_clamp.update({"xhigh": "high", "max": "high", "ultra": "high"})
from agent.model_metadata import is_grok_46_family
# Grok 4.6 accepts xhigh as a wire value. Older Grok models top out
# at high, while max/ultra remain Hermes aliases for every xAI model.
if not is_grok_46_family(model):
_effort_clamp["xhigh"] = "high"
_effort_clamp.update({"max": "high", "ultra": "high"})
if (params.get("provider") or "").strip().lower() == "actual":
# Actual Computer relays to SGLang/vLLM backends that accept only
# none/low/medium/high/max for reasoning effort — a forwarded
@@ -441,16 +446,18 @@ class ResponsesApiTransport(ProviderTransport):
else:
kwargs.pop("prompt_cache_key", None)
# xAI Responses API rejects ``service_tier`` (HTTP 400 "Argument not
# supported: service_tier") — hit when ``/fast`` priority-processing
# mode lingers from a prior model in the same session, or when a
# user explicitly sets ``agent.service_tier`` in config.yaml. The
# main-loop guard (``resolve_fast_mode_overrides`` only returns
# ``service_tier`` for OpenAI fast-eligible models) doesn't cover
# those leak paths, so strip defensively when targeting xAI. See
# #28490 for the original report.
# Older xAI Responses models reject ``service_tier`` (HTTP 400
# "Argument not supported: service_tier"). Grok 4.6 accepts Priority
# Processing, but continue stripping stale or unsupported tier values
# on every other xAI path. See #28490 and #84799.
if is_xai_responses:
kwargs.pop("service_tier", None)
from agent.model_metadata import is_grok_46_family
if not (
is_grok_46_family(model)
and kwargs.get("service_tier") == "priority"
):
kwargs.pop("service_tier", None)
# Forward per-request timeout to the SDK so OpenAI/Anthropic clients
# honor it. Without this, ``providers.<id>.request_timeout_seconds``
+7 -1
View File
@@ -2870,7 +2870,13 @@ def _strip_vendor_prefix(model_id: str) -> str:
def model_supports_fast_mode(model_id: Optional[str]) -> bool:
"""Return whether Hermes should expose the /fast toggle for this model."""
return _is_anthropic_fast_model(model_id) or _is_openai_fast_model(model_id)
from agent.model_metadata import is_grok_46_family
return (
_is_anthropic_fast_model(model_id)
or _is_openai_fast_model(model_id)
or is_grok_46_family(str(model_id or ""))
)
def _is_anthropic_fast_model(model_id: Optional[str]) -> bool:
@@ -689,6 +689,47 @@ class TestCodexTransportTimeout:
class TestCodexTransportXaiReasoningEffort:
@pytest.fixture
def transport(self):
from agent.transports.codex import ResponsesApiTransport
return ResponsesApiTransport()
def test_grok_46_preserves_xhigh(self, transport):
kw = transport.build_kwargs(
model="grok-4.6",
messages=[{"role": "user", "content": "hi"}],
tools=[],
is_xai_responses=True,
reasoning_config={"effort": "xhigh"},
)
assert kw["reasoning"]["effort"] == "xhigh"
@pytest.mark.parametrize("effort", ["max", "ultra"])
def test_grok_46_clamps_hermes_aliases_to_high(self, transport, effort):
kw = transport.build_kwargs(
model="x-ai/grok-4.6-latest",
messages=[{"role": "user", "content": "hi"}],
tools=[],
is_xai_responses=True,
reasoning_config={"effort": effort},
)
assert kw["reasoning"]["effort"] == "high"
def test_older_grok_clamps_xhigh_to_high(self, transport):
kw = transport.build_kwargs(
model="grok-4.5",
messages=[{"role": "user", "content": "hi"}],
tools=[],
is_xai_responses=True,
reasoning_config={"effort": "xhigh"},
)
assert kw["reasoning"]["effort"] == "high"
class TestCodexTransportXaiServiceTierStrip:
"""xAI Responses API rejects ``service_tier`` (#28490).
@@ -721,6 +762,28 @@ class TestCodexTransportXaiServiceTierStrip:
f"got {kw.get('service_tier')!r}"
)
def test_grok_46_preserves_priority_service_tier(self, transport):
kw = transport.build_kwargs(
model="x-ai/grok-4.6-latest",
messages=[{"role": "user", "content": "hi"}],
tools=[],
is_xai_responses=True,
request_overrides={"service_tier": "priority"},
)
assert kw.get("service_tier") == "priority"
def test_grok_46_strips_non_priority_service_tier(self, transport):
kw = transport.build_kwargs(
model="grok-4.6",
messages=[{"role": "user", "content": "hi"}],
tools=[],
is_xai_responses=True,
request_overrides={"service_tier": "unsupported"},
)
assert "service_tier" not in kw
def test_non_xai_codex_preserves_service_tier(self, transport):
"""The strip is xAI-only — native Codex DOES accept
service_tier=priority (OpenAI Priority Processing). Stripping
+11
View File
@@ -123,6 +123,17 @@ class TestPriorityProcessingModels(unittest.TestCase):
def test_grok_46_supports_priority_processing(self):
from hermes_cli.models import (
model_supports_fast_mode,
resolve_fast_mode_overrides,
)
assert model_supports_fast_mode("grok-4.6") is True
assert model_supports_fast_mode("x-ai/grok-4.6-latest") is True
assert model_supports_fast_mode("grok-4.5") is False
assert resolve_fast_mode_overrides("grok-4.6") == {"service_tier": "priority"}
def test_resolve_overrides_returns_service_tier(self):
from hermes_cli.models import resolve_fast_mode_overrides