fix(xai): preserve Grok 4.6 wire capabilities
This commit is contained in:
@@ -604,6 +604,14 @@ def grok_supports_reasoning_effort(model: str) -> bool:
|
||||
return any(name.startswith(prefix) for prefix in _GROK_EFFORT_CAPABLE_PREFIXES)
|
||||
|
||||
|
||||
def is_grok_46_family(model: str) -> bool:
|
||||
"""Return whether *model* is a Grok 4.6 family identifier."""
|
||||
name = (model or "").strip().lower().replace("_", "-")
|
||||
if "/" in name:
|
||||
name = name.rsplit("/", 1)[-1]
|
||||
return name == "grok-4.6" or name.startswith("grok-4.6-")
|
||||
|
||||
|
||||
_CONTEXT_LENGTH_KEYS = (
|
||||
"context_length",
|
||||
"context_window",
|
||||
|
||||
+18
-11
@@ -300,8 +300,13 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
# Ultra is the Codex product tier; the Responses API wire value is max.
|
||||
_effort_clamp["ultra"] = "max"
|
||||
if params.get("is_xai_responses", False):
|
||||
# xAI Responses tops out at high; keep generic stronger values usable.
|
||||
_effort_clamp.update({"xhigh": "high", "max": "high", "ultra": "high"})
|
||||
from agent.model_metadata import is_grok_46_family
|
||||
|
||||
# Grok 4.6 accepts xhigh as a wire value. Older Grok models top out
|
||||
# at high, while max/ultra remain Hermes aliases for every xAI model.
|
||||
if not is_grok_46_family(model):
|
||||
_effort_clamp["xhigh"] = "high"
|
||||
_effort_clamp.update({"max": "high", "ultra": "high"})
|
||||
if (params.get("provider") or "").strip().lower() == "actual":
|
||||
# Actual Computer relays to SGLang/vLLM backends that accept only
|
||||
# none/low/medium/high/max for reasoning effort — a forwarded
|
||||
@@ -441,16 +446,18 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
else:
|
||||
kwargs.pop("prompt_cache_key", None)
|
||||
|
||||
# xAI Responses API rejects ``service_tier`` (HTTP 400 "Argument not
|
||||
# supported: service_tier") — hit when ``/fast`` priority-processing
|
||||
# mode lingers from a prior model in the same session, or when a
|
||||
# user explicitly sets ``agent.service_tier`` in config.yaml. The
|
||||
# main-loop guard (``resolve_fast_mode_overrides`` only returns
|
||||
# ``service_tier`` for OpenAI fast-eligible models) doesn't cover
|
||||
# those leak paths, so strip defensively when targeting xAI. See
|
||||
# #28490 for the original report.
|
||||
# Older xAI Responses models reject ``service_tier`` (HTTP 400
|
||||
# "Argument not supported: service_tier"). Grok 4.6 accepts Priority
|
||||
# Processing, but continue stripping stale or unsupported tier values
|
||||
# on every other xAI path. See #28490 and #84799.
|
||||
if is_xai_responses:
|
||||
kwargs.pop("service_tier", None)
|
||||
from agent.model_metadata import is_grok_46_family
|
||||
|
||||
if not (
|
||||
is_grok_46_family(model)
|
||||
and kwargs.get("service_tier") == "priority"
|
||||
):
|
||||
kwargs.pop("service_tier", None)
|
||||
|
||||
# Forward per-request timeout to the SDK so OpenAI/Anthropic clients
|
||||
# honor it. Without this, ``providers.<id>.request_timeout_seconds``
|
||||
|
||||
@@ -2870,7 +2870,13 @@ def _strip_vendor_prefix(model_id: str) -> str:
|
||||
|
||||
def model_supports_fast_mode(model_id: Optional[str]) -> bool:
|
||||
"""Return whether Hermes should expose the /fast toggle for this model."""
|
||||
return _is_anthropic_fast_model(model_id) or _is_openai_fast_model(model_id)
|
||||
from agent.model_metadata import is_grok_46_family
|
||||
|
||||
return (
|
||||
_is_anthropic_fast_model(model_id)
|
||||
or _is_openai_fast_model(model_id)
|
||||
or is_grok_46_family(str(model_id or ""))
|
||||
)
|
||||
|
||||
|
||||
def _is_anthropic_fast_model(model_id: Optional[str]) -> bool:
|
||||
|
||||
@@ -689,6 +689,47 @@ class TestCodexTransportTimeout:
|
||||
|
||||
|
||||
|
||||
class TestCodexTransportXaiReasoningEffort:
|
||||
@pytest.fixture
|
||||
def transport(self):
|
||||
from agent.transports.codex import ResponsesApiTransport
|
||||
return ResponsesApiTransport()
|
||||
|
||||
def test_grok_46_preserves_xhigh(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
model="grok-4.6",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
tools=[],
|
||||
is_xai_responses=True,
|
||||
reasoning_config={"effort": "xhigh"},
|
||||
)
|
||||
|
||||
assert kw["reasoning"]["effort"] == "xhigh"
|
||||
|
||||
@pytest.mark.parametrize("effort", ["max", "ultra"])
|
||||
def test_grok_46_clamps_hermes_aliases_to_high(self, transport, effort):
|
||||
kw = transport.build_kwargs(
|
||||
model="x-ai/grok-4.6-latest",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
tools=[],
|
||||
is_xai_responses=True,
|
||||
reasoning_config={"effort": effort},
|
||||
)
|
||||
|
||||
assert kw["reasoning"]["effort"] == "high"
|
||||
|
||||
def test_older_grok_clamps_xhigh_to_high(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
model="grok-4.5",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
tools=[],
|
||||
is_xai_responses=True,
|
||||
reasoning_config={"effort": "xhigh"},
|
||||
)
|
||||
|
||||
assert kw["reasoning"]["effort"] == "high"
|
||||
|
||||
|
||||
class TestCodexTransportXaiServiceTierStrip:
|
||||
"""xAI Responses API rejects ``service_tier`` (#28490).
|
||||
|
||||
@@ -721,6 +762,28 @@ class TestCodexTransportXaiServiceTierStrip:
|
||||
f"got {kw.get('service_tier')!r}"
|
||||
)
|
||||
|
||||
def test_grok_46_preserves_priority_service_tier(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
model="x-ai/grok-4.6-latest",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
tools=[],
|
||||
is_xai_responses=True,
|
||||
request_overrides={"service_tier": "priority"},
|
||||
)
|
||||
|
||||
assert kw.get("service_tier") == "priority"
|
||||
|
||||
def test_grok_46_strips_non_priority_service_tier(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
model="grok-4.6",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
tools=[],
|
||||
is_xai_responses=True,
|
||||
request_overrides={"service_tier": "unsupported"},
|
||||
)
|
||||
|
||||
assert "service_tier" not in kw
|
||||
|
||||
def test_non_xai_codex_preserves_service_tier(self, transport):
|
||||
"""The strip is xAI-only — native Codex DOES accept
|
||||
service_tier=priority (OpenAI Priority Processing). Stripping
|
||||
|
||||
@@ -123,6 +123,17 @@ class TestPriorityProcessingModels(unittest.TestCase):
|
||||
|
||||
|
||||
|
||||
def test_grok_46_supports_priority_processing(self):
|
||||
from hermes_cli.models import (
|
||||
model_supports_fast_mode,
|
||||
resolve_fast_mode_overrides,
|
||||
)
|
||||
|
||||
assert model_supports_fast_mode("grok-4.6") is True
|
||||
assert model_supports_fast_mode("x-ai/grok-4.6-latest") is True
|
||||
assert model_supports_fast_mode("grok-4.5") is False
|
||||
assert resolve_fast_mode_overrides("grok-4.6") == {"service_tier": "priority"}
|
||||
|
||||
def test_resolve_overrides_returns_service_tier(self):
|
||||
from hermes_cli.models import resolve_fast_mode_overrides
|
||||
|
||||
|
||||
Reference in New Issue
Block a user