fix(openai): keep Astra rules on eligible routes
(cherry picked from commit f92fb9d9964e6e8ea118035c548aae812054f504)
This commit is contained in:
@@ -1380,8 +1380,13 @@ class _CodexCompletionsAdapter:
|
||||
if isinstance(reasoning_cfg, dict) and reasoning_cfg.get("enabled") is not False:
|
||||
# Truthy-only: Codex 400s on e.g. {"effort": null}, so falsy → default. Shared
|
||||
# per-model clamp with the main transport ("max" is gpt-5.6-only; "minimal"/"ultra" rejected).
|
||||
from agent.reasoning_effort import clamp_effort, codex_supported_efforts
|
||||
effort = clamp_effort(reasoning_cfg.get("effort") or "medium", codex_supported_efforts(model))
|
||||
from agent.reasoning_effort import clamp_effort
|
||||
from agent.transports.codex import _codex_efforts_for_route
|
||||
is_codex_backend = base_url_host_matches(host, "chatgpt.com") and "/backend-api/codex" in host.lower()
|
||||
effort = clamp_effort(
|
||||
reasoning_cfg.get("effort") or "medium",
|
||||
_codex_efforts_for_route(model, host, is_codex_backend=is_codex_backend),
|
||||
)
|
||||
resp_kwargs["reasoning"] = {"effort": effort, "summary": "auto"}
|
||||
resp_kwargs["include"] = ["reasoning.encrypted_content"]
|
||||
tools = kwargs.get("tools")
|
||||
|
||||
@@ -11,7 +11,8 @@ import re
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from agent.reasoning_effort import (
|
||||
ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
|
||||
ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, CODEX_LEGACY_EFFORTS,
|
||||
XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
|
||||
# Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort):
|
||||
# per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365).
|
||||
codex_supported_efforts,
|
||||
@@ -226,7 +227,9 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
|
||||
declared = _profile_declared_efforts(params.get("provider"), model, params.get("base_url"))
|
||||
if declared is not None and not declared:
|
||||
reasoning_enabled = False
|
||||
supported = declared or codex_supported_efforts(model)
|
||||
supported = declared or _codex_efforts_for_route(
|
||||
model, params.get("base_url"), is_codex_backend=params.get("is_codex_backend") is True
|
||||
)
|
||||
return clamp_effort(reasoning_effort, supported), reasoning_enabled
|
||||
|
||||
|
||||
@@ -275,6 +278,15 @@ def _is_official_openai_responses_route(model: Any, base_url: Any) -> bool:
|
||||
return base_url_host_matches(str(base_url or ""), "api.openai.com")
|
||||
|
||||
|
||||
def _codex_efforts_for_route(model: Any, base_url: Any, *, is_codex_backend: bool = False) -> tuple[str, ...]:
|
||||
"""Keep Astra's new vocabulary off unrelated Responses-compatible endpoints."""
|
||||
if _is_astra_model(model) and not (
|
||||
is_codex_backend or _is_official_openai_responses_route(model, base_url)
|
||||
):
|
||||
return CODEX_LEGACY_EFFORTS
|
||||
return codex_supported_efforts(str(model or ""))
|
||||
|
||||
|
||||
def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url: Any) -> None:
|
||||
"""Apply Astra's model-specific restrictions after all request overrides are merged."""
|
||||
if not _is_astra_model(model):
|
||||
|
||||
@@ -3197,6 +3197,18 @@ class TestCodexAdapterPromptCacheKey:
|
||||
assert captured["prompt_cache_options"] == {"ttl": "30m"}
|
||||
assert "prompt_cache_retention" not in captured
|
||||
|
||||
def test_astra_auxiliary_proxy_keeps_legacy_effort_contract(self):
|
||||
adapter, captured = self._build_adapter(
|
||||
base_url="https://responses.example.com/v1",
|
||||
model="gpt-6-astra",
|
||||
)
|
||||
adapter.create(
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
extra_body={"reasoning": {"effort": "none"}},
|
||||
)
|
||||
assert captured["reasoning"]["effort"] == "none"
|
||||
assert "prompt_cache_options" not in captured
|
||||
|
||||
def test_codex_backend_forwards_auxiliary_service_tier(self):
|
||||
adapter, captured = self._build_adapter(
|
||||
base_url="https://chatgpt.com/backend-api/codex",
|
||||
|
||||
@@ -96,10 +96,12 @@ class TestCodexBuildKwargs:
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
tools=[],
|
||||
base_url="https://responses.example.com/v1",
|
||||
reasoning_config={"enabled": True, "effort": "none"},
|
||||
request_overrides={"prompt_cache_options": {"ttl": "30m"}},
|
||||
)
|
||||
|
||||
assert "prompt_cache_options" not in kw
|
||||
assert kw["reasoning"]["effort"] == "none"
|
||||
|
||||
def test_900k_context_variant_suffix_stripped_on_wire(self, transport):
|
||||
"""``-900k`` large-context picker variants are Hermes-side aliases —
|
||||
|
||||
Reference in New Issue
Block a user