test(openai): close Astra baseline review gaps
(cherry picked from commit 8c27b7c9316ba675e4d9ebcc7a659754f37f18ef)
This commit is contained in:
@@ -1434,6 +1434,11 @@ class _CodexCompletionsAdapter:
|
||||
resp_kwargs["prompt_cache_retention"] = cache_retention
|
||||
except Exception:
|
||||
logger.debug("Codex auxiliary: prompt_cache_key derivation skipped", exc_info=True)
|
||||
# Keep auxiliary Responses calls on the same exact-model contract as main
|
||||
# turns. This runs last so caller overrides cannot reintroduce unsupported
|
||||
# Astra reasoning/sampling fields or replace the official cache TTL.
|
||||
from agent.transports.codex import _sanitize_astra_request_kwargs
|
||||
_sanitize_astra_request_kwargs(resp_kwargs, model, host)
|
||||
return resp_kwargs, model, timeout
|
||||
|
||||
def create(self, **kwargs) -> Any:
|
||||
|
||||
+1
-1
@@ -107,7 +107,7 @@ class ModelCapabilities:
|
||||
# Hermes provider names → models.dev provider IDs
|
||||
PROVIDER_TO_MODELS_DEV: Dict[str, str] = {
|
||||
"openrouter": "openrouter", "novita": "novita-ai", "anthropic": "anthropic",
|
||||
"openai": "openai", "openai-codex": "openai", "zai": "zai",
|
||||
"openai": "openai", "openai-api": "openai", "openai-codex": "openai", "zai": "zai",
|
||||
"kimi": "kimi-for-coding", "kimi-coding": "kimi-for-coding",
|
||||
"moonshot": "kimi-for-coding", "stepfun": "stepfun",
|
||||
"kimi-coding-cn": "kimi-for-coding", "minimax": "minimax",
|
||||
|
||||
+13
-16
@@ -211,15 +211,6 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
|
||||
elif reasoning_config.get("effort"):
|
||||
reasoning_effort = reasoning_config["effort"]
|
||||
|
||||
# Astra has no wire-level disable/minimal setting. Preserve Hermes' user-facing
|
||||
# controls by sending the documented lowest enabled level instead; this also keeps
|
||||
# an explicit ``enabled: false`` request valid on the Responses API.
|
||||
if model.strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra":
|
||||
requested = str(reasoning_effort or "").strip().lower()
|
||||
if not reasoning_enabled or requested in {"", "none", "minimal", "disabled", "off"}:
|
||||
return "low", True
|
||||
return clamp_effort(reasoning_effort, CODEX_ASTRA_EFFORTS), True
|
||||
|
||||
# Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker
|
||||
# supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that
|
||||
# repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a
|
||||
@@ -288,18 +279,24 @@ def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url:
|
||||
"""Apply Astra's model-specific restrictions after all request overrides are merged."""
|
||||
if not _is_astra_model(model):
|
||||
return
|
||||
if not _is_official_openai_responses_route(model, base_url):
|
||||
kwargs.pop("prompt_cache_options", None)
|
||||
return
|
||||
reasoning = kwargs.get("reasoning")
|
||||
requested = reasoning.get("effort") if isinstance(reasoning, dict) else None
|
||||
normalized = str(requested or "").strip().lower()
|
||||
effort = "low" if normalized in {"", "none", "minimal", "disabled", "off"} else clamp_effort(
|
||||
requested, CODEX_ASTRA_EFFORTS
|
||||
)
|
||||
kwargs["reasoning"] = {**(reasoning if isinstance(reasoning, dict) else {}), "effort": effort}
|
||||
kwargs["reasoning"].setdefault("summary", "auto")
|
||||
for key in ("temperature", "top_p", "top_logprobs", "logprobs"):
|
||||
kwargs.pop(key, None)
|
||||
include = kwargs.get("include")
|
||||
if isinstance(include, list):
|
||||
kwargs["include"] = [item for item in include if "logprob" not in str(item).lower()]
|
||||
if _is_official_openai_responses_route(model, base_url):
|
||||
kwargs["prompt_cache_options"] = {"ttl": "30m"}
|
||||
kwargs.pop("prompt_cache_retention", None)
|
||||
else:
|
||||
# A caller override must not accidentally send official-only cache syntax to a
|
||||
# proxy or another provider that happens to accept the Astra model name.
|
||||
kwargs.pop("prompt_cache_options", None)
|
||||
kwargs["prompt_cache_options"] = {"ttl": "30m"}
|
||||
kwargs.pop("prompt_cache_retention", None)
|
||||
|
||||
|
||||
def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], scope_id: str = "") -> Optional[str]:
|
||||
|
||||
@@ -96,7 +96,7 @@ def _finalize_codex_models(model_ids: List[str], *, allow_astra: bool = False) -
|
||||
"""
|
||||
finalized = _add_forward_compat_models(model_ids)
|
||||
if not allow_astra:
|
||||
finalized = [model for model in finalized if model.lower() != "gpt-6-astra"]
|
||||
finalized = [model for model in finalized if model.lower().rsplit("/", 1)[-1] != "gpt-6-astra"]
|
||||
return _add_context_variants(finalized)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Focused real-path coverage for the GPT-6 Astra baseline contract."""
|
||||
|
||||
from types import SimpleNamespace
|
||||
|
||||
|
||||
def test_explicit_astra_resolves_and_uses_official_responses(monkeypatch, tmp_path):
|
||||
"""A fresh profile resolves metadata and routes the official endpoint without live I/O."""
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda *args, **kwargs: {})
|
||||
monkeypatch.setattr("agent.process_bootstrap.OpenAI", lambda **_kwargs: SimpleNamespace())
|
||||
monkeypatch.setattr("model_tools.get_tool_definitions", lambda *args, **kwargs: [])
|
||||
|
||||
from run_agent import AIAgent
|
||||
|
||||
agent = AIAgent(
|
||||
model="gpt-6-astra",
|
||||
provider="openai",
|
||||
api_key="test-key",
|
||||
base_url="https://api.openai.com/v1",
|
||||
platform="cli",
|
||||
max_iterations=2,
|
||||
quiet_mode=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
|
||||
assert agent.api_mode == "codex_responses"
|
||||
assert agent.context_compressor.context_length == 1_050_000
|
||||
kwargs = agent._get_transport().build_kwargs(
|
||||
model=agent.model,
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
tools=[],
|
||||
provider=agent.provider,
|
||||
base_url=agent.base_url,
|
||||
reasoning_config={"enabled": False, "effort": "none"},
|
||||
)
|
||||
assert kwargs["reasoning"]["effort"] == "low"
|
||||
assert kwargs["prompt_cache_options"] == {"ttl": "30m"}
|
||||
@@ -3184,6 +3184,19 @@ class TestCodexAdapterPromptCacheKey:
|
||||
])
|
||||
assert "prompt_cache_retention" not in captured
|
||||
|
||||
def test_astra_auxiliary_request_uses_official_contract(self):
|
||||
adapter, captured = self._build_adapter(
|
||||
base_url="https://api.openai.com/v1",
|
||||
model="gpt-6-astra",
|
||||
)
|
||||
adapter.create(
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
extra_body={"reasoning": {"enabled": False, "effort": "none"}},
|
||||
)
|
||||
assert captured["reasoning"]["effort"] == "low"
|
||||
assert captured["prompt_cache_options"] == {"ttl": "30m"}
|
||||
assert "prompt_cache_retention" not in captured
|
||||
|
||||
def test_codex_backend_forwards_auxiliary_service_tier(self):
|
||||
adapter, captured = self._build_adapter(
|
||||
base_url="https://chatgpt.com/backend-api/codex",
|
||||
|
||||
@@ -837,6 +837,9 @@ class TestGetModelCapabilities:
|
||||
assert caps.supports_vision is True
|
||||
assert caps.supports_reasoning is True
|
||||
|
||||
api_caps = get_model_capabilities("openai-api", "gpt-6-astra")
|
||||
assert api_caps == caps
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Per-model metadata overrides (model_overrides config)
|
||||
|
||||
@@ -51,6 +51,7 @@ class TestCodexBuildKwargs:
|
||||
"top_p": 0.9,
|
||||
"top_logprobs": 5,
|
||||
"logprobs": True,
|
||||
"reasoning": {"effort": "none"},
|
||||
"include": ["reasoning.encrypted_content", "message.output_text.logprobs"],
|
||||
"prompt_cache_retention": "24h",
|
||||
"prompt_cache_options": {"ttl": "1h"},
|
||||
@@ -63,6 +64,31 @@ class TestCodexBuildKwargs:
|
||||
assert kw["include"] == ["reasoning.encrypted_content"]
|
||||
for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs"):
|
||||
assert unsupported not in kw
|
||||
assert transport.preflight_kwargs(kw)["prompt_cache_options"] == {"ttl": "30m"}
|
||||
|
||||
@pytest.mark.parametrize("effort", ["none", "minimal"])
|
||||
def test_astra_normalizes_unsupported_low_efforts_after_overrides(self, transport, effort):
|
||||
kw = transport.build_kwargs(
|
||||
model="gpt-6-astra",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
tools=[],
|
||||
base_url="https://api.openai.com/v1",
|
||||
request_overrides={"reasoning": {"effort": effort}},
|
||||
)
|
||||
|
||||
assert kw["reasoning"]["effort"] == "low"
|
||||
|
||||
@pytest.mark.parametrize("effort", ["low", "medium", "high", "xhigh", "max"])
|
||||
def test_astra_accepts_complete_effort_ladder(self, transport, effort):
|
||||
kw = transport.build_kwargs(
|
||||
model="gpt-6-astra",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
tools=[],
|
||||
base_url="https://api.openai.com/v1",
|
||||
reasoning_config={"enabled": True, "effort": effort},
|
||||
)
|
||||
|
||||
assert kw["reasoning"]["effort"] == effort
|
||||
|
||||
def test_astra_proxy_does_not_receive_official_cache_options(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
|
||||
@@ -118,13 +118,17 @@ def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path):
|
||||
|
||||
(tmp_path / "config.toml").write_text('model = "gpt-6-astra"\n', encoding="utf-8")
|
||||
(tmp_path / "models_cache.json").write_text(
|
||||
json.dumps({"models": [{"slug": "gpt-6-astra", "priority": 0}]}),
|
||||
json.dumps({"models": [
|
||||
{"slug": "gpt-6-astra", "priority": 0},
|
||||
{"slug": "openai/gpt-6-astra", "priority": 1},
|
||||
]}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
monkeypatch.setenv("CODEX_HOME", str(tmp_path))
|
||||
monkeypatch.setattr(codex_models, "_fetch_models_from_api", lambda _token: [])
|
||||
|
||||
assert "gpt-6-astra" not in get_codex_model_ids(access_token="stale-token")
|
||||
assert "openai/gpt-6-astra" not in get_codex_model_ids(access_token="stale-token")
|
||||
|
||||
monkeypatch.setattr(
|
||||
codex_models,
|
||||
|
||||
@@ -79,7 +79,9 @@ def test_astra_is_offered_only_by_successful_account_discovery(monkeypatch):
|
||||
discovered = M.provider_model_ids("openai-api", force_refresh=True)
|
||||
with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol"]):
|
||||
not_entitled = M.provider_model_ids("openai-api", force_refresh=True)
|
||||
with patch.object(M, "fetch_api_models", side_effect=RuntimeError("discovery unavailable")):
|
||||
discovery_failed = M.provider_model_ids("openai-api", force_refresh=True)
|
||||
|
||||
assert "gpt-6-astra" in discovered
|
||||
assert "gpt-6-astra" not in not_entitled
|
||||
|
||||
assert "gpt-6-astra" not in discovery_failed
|
||||
|
||||
Reference in New Issue
Block a user