From 170a73637d503365847711b7cda9e09010c98d36 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 7 Sep 2026 21:35:51 +0530 Subject: [PATCH] =?UTF-8?q?fix(openai):=20drop=20prompt=5Fcache=5Foptions?= =?UTF-8?q?=20from=20Astra=20requests=20=E2=80=94=20not=20an=20SDK=20kwarg?= =?UTF-8?q?,=2030m=20is=20the=20server=20default?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every direct-API (api.openai.com) Astra request raised ``TypeError: Responses.create() got an unexpected keyword argument 'prompt_cache_options'`` before reaching the network: openai 2.24.0's Responses.create has no such parameter and no **kwargs, and neither send path relocates it into extra_body. The PR's tests stopped at build_kwargs/preflight so the SDK boundary was never crossed. OpenAI's prompt-caching guide states ``prompt_cache_options.ttl`` accepts only ``30m`` and that ``30m`` is the default, so the field carried no information: sending nothing yields the same cache lifetime. The sanitizer now only removes what the API rejects (none/minimal effort, sampling/logprob knobs, the pre-5.6 ``prompt_cache_retention``) and never adds a field, which also keeps the request body byte-stable for the cache prefix. Also: none/minimal→low no longer needs a bespoke {"", "none", "disabled", "off"} set — ``clamp_effort`` against CODEX_ASTRA_EFFORTS already resolves to the floor (``low``); and the auxiliary adapter derives ``is_codex_backend`` from ``classify_responses_route`` (the declared single owner of that predicate) instead of re-implementing the host test inline. Tests reshaped to contracts: the two proxy/subdomain cases collapse into one parametrised "exact host only" test asserting effort and temperature pass through untouched. --- agent/auxiliary_client.py | 7 ++-- agent/codex_responses_adapter.py | 2 +- agent/transports/codex.py | 38 +++++++------------ tests/agent/test_astra_baseline_runtime.py | 10 ++--- tests/agent/test_auxiliary_client.py | 4 +- .../agent/transports/test_codex_transport.py | 28 ++++---------- 6 files changed, 31 insertions(+), 58 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 0caabcf0b4..9f77e631a1 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1380,9 +1380,10 @@ class _CodexCompletionsAdapter: if isinstance(reasoning_cfg, dict) and reasoning_cfg.get("enabled") is not False: # Truthy-only: Codex 400s on e.g. {"effort": null}, so falsy → default. Shared # per-model clamp with the main transport ("max" is gpt-5.6-only; "minimal"/"ultra" rejected). + from agent.codex_responses_adapter import classify_responses_route from agent.reasoning_effort import clamp_effort from agent.transports.codex import _codex_efforts_for_route - is_codex_backend = base_url_host_matches(host, "chatgpt.com") and "/backend-api/codex" in host.lower() + is_codex_backend = classify_responses_route(SimpleNamespace(base_url=host)).is_codex_backend effort = clamp_effort( reasoning_cfg.get("effort") or "medium", _codex_efforts_for_route(model, host, is_codex_backend=is_codex_backend), @@ -1439,9 +1440,7 @@ class _CodexCompletionsAdapter: resp_kwargs["prompt_cache_retention"] = cache_retention except Exception: logger.debug("Codex auxiliary: prompt_cache_key derivation skipped", exc_info=True) - # Keep auxiliary Responses calls on the same exact-model contract as main - # turns. This runs last so caller overrides cannot reintroduce unsupported - # Astra reasoning/sampling fields or replace the official cache TTL. + # Last, like the main transport: caller extra_body must not put a rejected Astra field back. from agent.transports.codex import _sanitize_astra_request_kwargs _sanitize_astra_request_kwargs(resp_kwargs, model, host) return resp_kwargs, model, timeout diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 1d114af397..0dadad59b9 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -770,7 +770,7 @@ _PREFLIGHT_OPTIONAL_FIELDS: tuple[tuple[str, Callable[[Any], bool], Optional[Cal # Cache routing/retention and tool-dispatch hints pass through as-is. *( (key, lambda v: v is not None, None) - for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention", "prompt_cache_options") + for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention") ), # Native compaction directive; eligibility is resolved in agent/native_compaction.py. ("context_management", lambda v: isinstance(v, list) and bool(v), None), diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 9534b7452d..1c4a87dbfa 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -12,7 +12,7 @@ from typing import Any, Callable, Optional from agent.reasoning_effort import ( ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, CODEX_LEGACY_EFFORTS, - XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, + XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, is_astra_model, # Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort): # per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365). codex_supported_efforts, @@ -265,24 +265,19 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op return "24h" if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized) else None -def _is_astra_model(model: Any) -> bool: - return str(model or "").strip().lower().rsplit("/", 1)[-1] in ("gpt-6-astra", "gpt-6-astra-900k") - - def _is_official_openai_responses_route(model: Any, base_url: Any) -> bool: - """Astra cache semantics apply only to the official OpenAI API host.""" - if not _is_astra_model(model): + """Astra on the canonical API origin only — exact host, so a Responses-compatible proxy or a + lookalike subdomain keeps the generic contract.""" + if not is_astra_model(model): return False from utils import base_url_hostname - # Astra's account-gated cache contract is for the canonical API origin only. - # Do not extend it to subdomains (or lookalike hosts) by suffix matching. return base_url_hostname(str(base_url or "")).lower() == "api.openai.com" def _codex_efforts_for_route(model: Any, base_url: Any, *, is_codex_backend: bool = False) -> tuple[str, ...]: """Keep Astra's new vocabulary off unrelated Responses-compatible endpoints.""" - if _is_astra_model(model) and not ( + if is_astra_model(model) and not ( is_codex_backend or _is_official_openai_responses_route(model, base_url) ): return CODEX_LEGACY_EFFORTS @@ -290,27 +285,22 @@ def _codex_efforts_for_route(model: Any, base_url: Any, *, is_codex_backend: boo def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url: Any) -> None: - """Apply Astra's model-specific restrictions after all request overrides are merged.""" - if not _is_astra_model(model): - return + """Astra's official-API contract, applied AFTER ``request_overrides`` so an override can't put a + rejected field back on the wire: ``reasoning.effort`` is ``low..max`` only (``none``/``minimal`` + 400), sampling and logprob knobs are rejected, and cache lifetime is fixed server-side + (``prompt_cache_options.ttl`` accepts only its ``30m`` default, so nothing is sent for it and the + pre-5.6 ``prompt_cache_retention`` knob is dropped).""" if not _is_official_openai_responses_route(model, base_url): - kwargs.pop("prompt_cache_options", None) return reasoning = kwargs.get("reasoning") - requested = reasoning.get("effort") if isinstance(reasoning, dict) else None - normalized = str(requested or "").strip().lower() - effort = "low" if normalized in {"", "none", "minimal", "disabled", "off"} else clamp_effort( - requested, CODEX_ASTRA_EFFORTS - ) - kwargs["reasoning"] = {**(reasoning if isinstance(reasoning, dict) else {}), "effort": effort} - kwargs["reasoning"].setdefault("summary", "auto") - for key in ("temperature", "top_p", "top_logprobs", "logprobs"): + if isinstance(reasoning, dict): + requested = str(reasoning.get("effort") or "").strip().lower() + reasoning["effort"] = clamp_effort(requested, CODEX_ASTRA_EFFORTS) if requested else "low" + for key in ("temperature", "top_p", "top_logprobs", "logprobs", "prompt_cache_retention"): kwargs.pop(key, None) include = kwargs.get("include") if isinstance(include, list): kwargs["include"] = [item for item in include if "logprob" not in str(item).lower()] - kwargs["prompt_cache_options"] = {"ttl": "30m"} - kwargs.pop("prompt_cache_retention", None) def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], scope_id: str = "") -> Optional[str]: diff --git a/tests/agent/test_astra_baseline_runtime.py b/tests/agent/test_astra_baseline_runtime.py index 4a6e108336..b3591ad11a 100644 --- a/tests/agent/test_astra_baseline_runtime.py +++ b/tests/agent/test_astra_baseline_runtime.py @@ -33,10 +33,9 @@ def test_explicit_astra_resolves_and_uses_official_responses(monkeypatch, tmp_pa tools=[], provider=agent.provider, base_url=agent.base_url, - reasoning_config={"enabled": False, "effort": "none"}, + reasoning_config={"enabled": True, "effort": "none"}, ) - assert kwargs["reasoning"]["effort"] == "low" - assert kwargs["prompt_cache_options"] == {"ttl": "30m"} + assert kwargs["reasoning"]["effort"] == "low" # Astra has no ``none`` wire level def test_astra_codex_oauth_fallback_uses_backend_context_limit(): @@ -46,8 +45,9 @@ def test_astra_codex_oauth_fallback_uses_backend_context_limit(): _resolve_codex_oauth_context_length_with_source, ) - assert DEFAULT_CONTEXT_LENGTHS["gpt-6-astra"] == 1_050_000 - assert _resolve_codex_oauth_context_length_with_source("gpt-6-astra") == (272_000, "fallback") + codex_ctx, source = _resolve_codex_oauth_context_length_with_source("gpt-6-astra") + assert source == "fallback" + assert codex_ctx < DEFAULT_CONTEXT_LENGTHS["gpt-6-astra"] # Codex caps below the direct API window @pytest.mark.parametrize("advertised,expected", [(272_000, 900_000), (200_000, 200_000), (1_050_000, 1_050_000)]) diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index 22d5390a59..c1d5398e9b 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -3191,10 +3191,9 @@ class TestCodexAdapterPromptCacheKey: ) adapter.create( messages=[{"role": "user", "content": "hi"}], - extra_body={"reasoning": {"enabled": False, "effort": "none"}}, + extra_body={"reasoning": {"effort": "none"}}, ) assert captured["reasoning"]["effort"] == "low" - assert captured["prompt_cache_options"] == {"ttl": "30m"} assert "prompt_cache_retention" not in captured def test_astra_auxiliary_proxy_keeps_legacy_effort_contract(self): @@ -3207,7 +3206,6 @@ class TestCodexAdapterPromptCacheKey: extra_body={"reasoning": {"effort": "none"}}, ) assert captured["reasoning"]["effort"] == "none" - assert "prompt_cache_options" not in captured def test_codex_backend_forwards_auxiliary_service_tier(self): adapter, captured = self._build_adapter( diff --git a/tests/agent/transports/test_codex_transport.py b/tests/agent/transports/test_codex_transport.py index 4ff7a3859e..4da859ccfa 100644 --- a/tests/agent/transports/test_codex_transport.py +++ b/tests/agent/transports/test_codex_transport.py @@ -54,17 +54,13 @@ class TestCodexBuildKwargs: "reasoning": {"effort": "none"}, "include": ["reasoning.encrypted_content", "message.output_text.logprobs"], "prompt_cache_retention": "24h", - "prompt_cache_options": {"ttl": "1h"}, }, ) assert kw["reasoning"]["effort"] == "low" - assert kw["prompt_cache_options"] == {"ttl": "30m"} - assert "prompt_cache_retention" not in kw assert kw["include"] == ["reasoning.encrypted_content"] - for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs"): + for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs", "prompt_cache_retention"): assert unsupported not in kw - assert transport.preflight_kwargs(kw)["prompt_cache_options"] == {"ttl": "30m"} @pytest.mark.parametrize("effort", ["none", "minimal"]) def test_astra_normalizes_unsupported_low_efforts_after_overrides(self, transport, effort): @@ -90,30 +86,20 @@ class TestCodexBuildKwargs: assert kw["reasoning"]["effort"] == effort - def test_astra_proxy_does_not_receive_official_cache_options(self, transport): + @pytest.mark.parametrize("base_url", ["https://responses.example.com/v1", "https://evil.api.openai.com/v1"]) + def test_astra_contract_is_exact_host_only(self, transport, base_url): + """Proxies and lookalike subdomains keep the generic Responses contract (effort passes through).""" kw = transport.build_kwargs( model="gpt-6-astra", messages=[{"role": "user", "content": "Hi"}], tools=[], - base_url="https://responses.example.com/v1", + base_url=base_url, reasoning_config={"enabled": True, "effort": "none"}, - request_overrides={"prompt_cache_options": {"ttl": "30m"}}, + request_overrides={"temperature": 0.4}, ) - assert "prompt_cache_options" not in kw assert kw["reasoning"]["effort"] == "none" - - def test_astra_openai_subdomain_is_not_official_route(self, transport): - """Only api.openai.com gets Astra's official cache contract.""" - kw = transport.build_kwargs( - model="gpt-6-astra", - messages=[{"role": "user", "content": "Hi"}], - tools=[], - base_url="https://evil.api.openai.com/v1", - request_overrides={"prompt_cache_options": {"ttl": "30m"}}, - ) - - assert "prompt_cache_options" not in kw + assert kw["temperature"] == 0.4 def test_900k_context_variant_suffix_stripped_on_wire(self, transport): """``-900k`` large-context picker variants are Hermes-side aliases —