diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index e84c59b1ba..2d614828ae 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -83,7 +83,7 @@ META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh") def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]: """Supported effort set for an OpenAI/Codex Responses model.""" - if (model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra": + if (model or "").strip().lower().rsplit("/", 1)[-1] in ("gpt-6-astra", "gpt-6-astra-900k"): return CODEX_ASTRA_EFFORTS return CODEX_GPT56_EFFORTS if "gpt-5.6" in (model or "").lower() else CODEX_LEGACY_EFFORTS diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 6da9038172..9534b7452d 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -266,7 +266,7 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op def _is_astra_model(model: Any) -> bool: - return str(model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra" + return str(model or "").strip().lower().rsplit("/", 1)[-1] in ("gpt-6-astra", "gpt-6-astra-900k") def _is_official_openai_responses_route(model: Any, base_url: Any) -> bool: diff --git a/hermes_cli/codex_models.py b/hermes_cli/codex_models.py index 0700e11ad0..859780630b 100644 --- a/hermes_cli/codex_models.py +++ b/hermes_cli/codex_models.py @@ -94,9 +94,12 @@ def _finalize_codex_models(model_ids: List[str], *, allow_astra: bool = False) - Cached/configured model names are useful compatibility hints, but only the account-scoped Codex endpoint is authoritative for current Astra access. """ + from agent.model_metadata import strip_codex_context_variant_suffix + finalized = _add_forward_compat_models(model_ids) if not allow_astra: - finalized = [model for model in finalized if model.lower().rsplit("/", 1)[-1] != "gpt-6-astra"] + finalized = [model for model in finalized + if strip_codex_context_variant_suffix(model.lower()).rsplit("/", 1)[-1] != "gpt-6-astra"] return _add_context_variants(finalized) diff --git a/tests/agent/test_astra_baseline_runtime.py b/tests/agent/test_astra_baseline_runtime.py index c26b25786b..6768406dae 100644 --- a/tests/agent/test_astra_baseline_runtime.py +++ b/tests/agent/test_astra_baseline_runtime.py @@ -2,6 +2,8 @@ from types import SimpleNamespace +import pytest + def test_explicit_astra_resolves_and_uses_official_responses(monkeypatch, tmp_path): """A fresh profile resolves metadata and routes the official endpoint without live I/O.""" @@ -46,3 +48,33 @@ def test_astra_codex_oauth_fallback_uses_backend_context_limit(): assert DEFAULT_CONTEXT_LENGTHS["gpt-6-astra"] == 1_050_000 assert _resolve_codex_oauth_context_length_with_source("gpt-6-astra") == (272_000, "fallback") + + +@pytest.mark.parametrize("advertised,expected", [(272_000, 900_000), (200_000, 200_000), (1_050_000, 1_050_000)]) +def test_astra_900k_opt_in_preserves_live_limits_and_wire_contract(monkeypatch, tmp_path, advertised, expected): + """Only the known stale advertisement is lifted; the alias never reaches the wire.""" + from agent import model_metadata as metadata + from agent.reasoning_effort import CODEX_ASTRA_EFFORTS, codex_supported_efforts + from agent.transports.codex import ResponsesApiTransport + + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + monkeypatch.setattr(metadata, "_codex_oauth_context_cache", {}) + monkeypatch.setattr(metadata.requests, "get", lambda *args, **kwargs: SimpleNamespace( + status_code=200, + json=lambda: {"models": [{"slug": "gpt-6-astra", "context_window": advertised}]}, + )) + route = {"base_url": "https://chatgpt.com/backend-api/codex", "provider": "openai-codex"} + assert metadata.get_model_context_length("gpt-6-astra-900k", api_key="test-token", **route) == expected + assert metadata.get_model_context_length("gpt-6-astra", api_key="test-token", **route) == advertised + assert codex_supported_efforts("gpt-6-astra-900k") == CODEX_ASTRA_EFFORTS + + for config, expected_effort in [({"effort": "max"}, "max"), ({"enabled": False}, "low")]: + kwargs = ResponsesApiTransport().build_kwargs( + model="gpt-6-astra-900k", messages=[{"role": "user", "content": "Hi"}], tools=[], + is_codex_backend=True, reasoning_config=config, + request_overrides={"temperature": 0.5, "logprobs": True}, **route, + ) + assert kwargs["model"] == "gpt-6-astra" + assert kwargs["reasoning"]["effort"] == expected_effort + assert "temperature" not in kwargs and "logprobs" not in kwargs + assert "prompt_cache_options" not in kwargs diff --git a/tests/hermes_cli/test_codex_models.py b/tests/hermes_cli/test_codex_models.py index 0694fa4eea..00d7458edb 100644 --- a/tests/hermes_cli/test_codex_models.py +++ b/tests/hermes_cli/test_codex_models.py @@ -121,6 +121,8 @@ def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path): json.dumps({"models": [ {"slug": "gpt-6-astra", "priority": 0}, {"slug": "openai/gpt-6-astra", "priority": 1}, + {"slug": "gpt-6-astra-900k", "priority": 2}, + {"slug": "openai/gpt-6-astra-900k", "priority": 3}, ]}), encoding="utf-8", ) @@ -129,13 +131,16 @@ def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path): assert "gpt-6-astra" not in get_codex_model_ids(access_token="stale-token") assert "openai/gpt-6-astra" not in get_codex_model_ids(access_token="stale-token") + assert "gpt-6-astra-900k" not in get_codex_model_ids(access_token="stale-token") + assert "openai/gpt-6-astra-900k" not in get_codex_model_ids(access_token="stale-token") monkeypatch.setattr( codex_models, "_fetch_models_from_api", lambda _token: codex_models._finalize_codex_models(["gpt-6-astra"], allow_astra=True), ) - assert "gpt-6-astra" in get_codex_model_ids(access_token="entitled-token") + entitled = get_codex_model_ids(access_token="entitled-token") + assert entitled[entitled.index("gpt-6-astra") + 1] == "gpt-6-astra-900k"