test(openai): close Astra baseline review gaps

(cherry picked from commit 8c27b7c9316ba675e4d9ebcc7a659754f37f18ef)
This commit is contained in:
Eva
2026-09-04 23:53:18 +07:00
committed by kshitij
parent 2c315ff59b
commit c990017482
10 changed files with 107 additions and 20 deletions
+5
View File
@@ -1434,6 +1434,11 @@ class _CodexCompletionsAdapter:
resp_kwargs["prompt_cache_retention"] = cache_retention
except Exception:
logger.debug("Codex auxiliary: prompt_cache_key derivation skipped", exc_info=True)
# Keep auxiliary Responses calls on the same exact-model contract as main
# turns. This runs last so caller overrides cannot reintroduce unsupported
# Astra reasoning/sampling fields or replace the official cache TTL.
from agent.transports.codex import _sanitize_astra_request_kwargs
_sanitize_astra_request_kwargs(resp_kwargs, model, host)
return resp_kwargs, model, timeout
def create(self, **kwargs) -> Any:
+1 -1
View File
@@ -107,7 +107,7 @@ class ModelCapabilities:
# Hermes provider names → models.dev provider IDs
PROVIDER_TO_MODELS_DEV: Dict[str, str] = {
"openrouter": "openrouter", "novita": "novita-ai", "anthropic": "anthropic",
"openai": "openai", "openai-codex": "openai", "zai": "zai",
"openai": "openai", "openai-api": "openai", "openai-codex": "openai", "zai": "zai",
"kimi": "kimi-for-coding", "kimi-coding": "kimi-for-coding",
"moonshot": "kimi-for-coding", "stepfun": "stepfun",
"kimi-coding-cn": "kimi-for-coding", "minimax": "minimax",
+13 -16
View File
@@ -211,15 +211,6 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
elif reasoning_config.get("effort"):
reasoning_effort = reasoning_config["effort"]
# Astra has no wire-level disable/minimal setting. Preserve Hermes' user-facing
# controls by sending the documented lowest enabled level instead; this also keeps
# an explicit ``enabled: false`` request valid on the Responses API.
if model.strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra":
requested = str(reasoning_effort or "").strip().lower()
if not reasoning_enabled or requested in {"", "none", "minimal", "disabled", "off"}:
return "low", True
return clamp_effort(reasoning_effort, CODEX_ASTRA_EFFORTS), True
# Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker
# supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that
# repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a
@@ -288,18 +279,24 @@ def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url:
"""Apply Astra's model-specific restrictions after all request overrides are merged."""
if not _is_astra_model(model):
return
if not _is_official_openai_responses_route(model, base_url):
kwargs.pop("prompt_cache_options", None)
return
reasoning = kwargs.get("reasoning")
requested = reasoning.get("effort") if isinstance(reasoning, dict) else None
normalized = str(requested or "").strip().lower()
effort = "low" if normalized in {"", "none", "minimal", "disabled", "off"} else clamp_effort(
requested, CODEX_ASTRA_EFFORTS
)
kwargs["reasoning"] = {**(reasoning if isinstance(reasoning, dict) else {}), "effort": effort}
kwargs["reasoning"].setdefault("summary", "auto")
for key in ("temperature", "top_p", "top_logprobs", "logprobs"):
kwargs.pop(key, None)
include = kwargs.get("include")
if isinstance(include, list):
kwargs["include"] = [item for item in include if "logprob" not in str(item).lower()]
if _is_official_openai_responses_route(model, base_url):
kwargs["prompt_cache_options"] = {"ttl": "30m"}
kwargs.pop("prompt_cache_retention", None)
else:
# A caller override must not accidentally send official-only cache syntax to a
# proxy or another provider that happens to accept the Astra model name.
kwargs.pop("prompt_cache_options", None)
kwargs["prompt_cache_options"] = {"ttl": "30m"}
kwargs.pop("prompt_cache_retention", None)
def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], scope_id: str = "") -> Optional[str]:
+1 -1
View File
@@ -96,7 +96,7 @@ def _finalize_codex_models(model_ids: List[str], *, allow_astra: bool = False) -
"""
finalized = _add_forward_compat_models(model_ids)
if not allow_astra:
finalized = [model for model in finalized if model.lower() != "gpt-6-astra"]
finalized = [model for model in finalized if model.lower().rsplit("/", 1)[-1] != "gpt-6-astra"]
return _add_context_variants(finalized)
@@ -0,0 +1,37 @@
"""Focused real-path coverage for the GPT-6 Astra baseline contract."""
from types import SimpleNamespace
def test_explicit_astra_resolves_and_uses_official_responses(monkeypatch, tmp_path):
"""A fresh profile resolves metadata and routes the official endpoint without live I/O."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda *args, **kwargs: {})
monkeypatch.setattr("agent.process_bootstrap.OpenAI", lambda **_kwargs: SimpleNamespace())
monkeypatch.setattr("model_tools.get_tool_definitions", lambda *args, **kwargs: [])
from run_agent import AIAgent
agent = AIAgent(
model="gpt-6-astra",
provider="openai",
api_key="test-key",
base_url="https://api.openai.com/v1",
platform="cli",
max_iterations=2,
quiet_mode=True,
skip_memory=True,
)
assert agent.api_mode == "codex_responses"
assert agent.context_compressor.context_length == 1_050_000
kwargs = agent._get_transport().build_kwargs(
model=agent.model,
messages=[{"role": "user", "content": "Hi"}],
tools=[],
provider=agent.provider,
base_url=agent.base_url,
reasoning_config={"enabled": False, "effort": "none"},
)
assert kwargs["reasoning"]["effort"] == "low"
assert kwargs["prompt_cache_options"] == {"ttl": "30m"}
+13
View File
@@ -3184,6 +3184,19 @@ class TestCodexAdapterPromptCacheKey:
])
assert "prompt_cache_retention" not in captured
def test_astra_auxiliary_request_uses_official_contract(self):
adapter, captured = self._build_adapter(
base_url="https://api.openai.com/v1",
model="gpt-6-astra",
)
adapter.create(
messages=[{"role": "user", "content": "hi"}],
extra_body={"reasoning": {"enabled": False, "effort": "none"}},
)
assert captured["reasoning"]["effort"] == "low"
assert captured["prompt_cache_options"] == {"ttl": "30m"}
assert "prompt_cache_retention" not in captured
def test_codex_backend_forwards_auxiliary_service_tier(self):
adapter, captured = self._build_adapter(
base_url="https://chatgpt.com/backend-api/codex",
+3
View File
@@ -837,6 +837,9 @@ class TestGetModelCapabilities:
assert caps.supports_vision is True
assert caps.supports_reasoning is True
api_caps = get_model_capabilities("openai-api", "gpt-6-astra")
assert api_caps == caps
# ---------------------------------------------------------------------------
# Per-model metadata overrides (model_overrides config)
@@ -51,6 +51,7 @@ class TestCodexBuildKwargs:
"top_p": 0.9,
"top_logprobs": 5,
"logprobs": True,
"reasoning": {"effort": "none"},
"include": ["reasoning.encrypted_content", "message.output_text.logprobs"],
"prompt_cache_retention": "24h",
"prompt_cache_options": {"ttl": "1h"},
@@ -63,6 +64,31 @@ class TestCodexBuildKwargs:
assert kw["include"] == ["reasoning.encrypted_content"]
for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs"):
assert unsupported not in kw
assert transport.preflight_kwargs(kw)["prompt_cache_options"] == {"ttl": "30m"}
@pytest.mark.parametrize("effort", ["none", "minimal"])
def test_astra_normalizes_unsupported_low_efforts_after_overrides(self, transport, effort):
kw = transport.build_kwargs(
model="gpt-6-astra",
messages=[{"role": "user", "content": "Hi"}],
tools=[],
base_url="https://api.openai.com/v1",
request_overrides={"reasoning": {"effort": effort}},
)
assert kw["reasoning"]["effort"] == "low"
@pytest.mark.parametrize("effort", ["low", "medium", "high", "xhigh", "max"])
def test_astra_accepts_complete_effort_ladder(self, transport, effort):
kw = transport.build_kwargs(
model="gpt-6-astra",
messages=[{"role": "user", "content": "Hi"}],
tools=[],
base_url="https://api.openai.com/v1",
reasoning_config={"enabled": True, "effort": effort},
)
assert kw["reasoning"]["effort"] == effort
def test_astra_proxy_does_not_receive_official_cache_options(self, transport):
kw = transport.build_kwargs(
+5 -1
View File
@@ -118,13 +118,17 @@ def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path):
(tmp_path / "config.toml").write_text('model = "gpt-6-astra"\n', encoding="utf-8")
(tmp_path / "models_cache.json").write_text(
json.dumps({"models": [{"slug": "gpt-6-astra", "priority": 0}]}),
json.dumps({"models": [
{"slug": "gpt-6-astra", "priority": 0},
{"slug": "openai/gpt-6-astra", "priority": 1},
]}),
encoding="utf-8",
)
monkeypatch.setenv("CODEX_HOME", str(tmp_path))
monkeypatch.setattr(codex_models, "_fetch_models_from_api", lambda _token: [])
assert "gpt-6-astra" not in get_codex_model_ids(access_token="stale-token")
assert "openai/gpt-6-astra" not in get_codex_model_ids(access_token="stale-token")
monkeypatch.setattr(
codex_models,
@@ -79,7 +79,9 @@ def test_astra_is_offered_only_by_successful_account_discovery(monkeypatch):
discovered = M.provider_model_ids("openai-api", force_refresh=True)
with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol"]):
not_entitled = M.provider_model_ids("openai-api", force_refresh=True)
with patch.object(M, "fetch_api_models", side_effect=RuntimeError("discovery unavailable")):
discovery_failed = M.provider_model_ids("openai-api", force_refresh=True)
assert "gpt-6-astra" in discovered
assert "gpt-6-astra" not in not_entitled
assert "gpt-6-astra" not in discovery_failed