fix(agent): fail closed when Codex account-model entitlement 400s exhaust the chain
A Codex ChatGPT-account 400 ('The X model is not supported when using Codex
with a ChatGPT account.') names the model, so with a single credential the
slug is dead for that account. The fallback walk still re-selected it and
restore_primary_runtime switched back to the primary at the start of every
turn, announcing an unverified 'Primary model restored' — the two warnings
alternated forever with zero delivered answers (#106475).
Record the rejected (provider, model) pair on the non-retryable client-error
path (only when no multi-credential pool exists — rotation covers that case,
#71970), skip rejected entries during the fallback walk, and gate
restore_primary_runtime on the primary's slug so the session fails closed
with the terminal entitlement error instead of oscillating. Fixes #106475.
(cherry picked from commit 471435b10288f15387b2549a8b02551d82c0f670)
This commit is contained in:
@@ -1124,6 +1124,13 @@ def restore_primary_runtime(agent) -> bool:
|
||||
return False # primary still in rate-limit cooldown, stay on fallback
|
||||
rt = agent._primary_runtime
|
||||
primary_provider = str((rt or {}).get("provider") or "").strip().lower()
|
||||
primary_model = str((rt or {}).get("model") or "").strip()
|
||||
from agent.chat_completion_helpers import _is_entitlement_rejected
|
||||
if primary_model and _is_entitlement_rejected(agent, primary_provider, primary_model):
|
||||
# The primary slug was rejected as unentitled for this account (#106475): restoring
|
||||
# here would announce a recovery that was never verified and re-fail every turn.
|
||||
# Stay on the fallback; the user sees the terminal entitlement error instead.
|
||||
return False
|
||||
primary_runtime_base_url = str((rt or {}).get("base_url") or "")
|
||||
|
||||
def _matches_primary(candidate) -> bool:
|
||||
|
||||
@@ -1712,6 +1712,68 @@ def _rebind_fallback_credential_pool(agent, fb_provider: str, fb_model: str) ->
|
||||
logger.debug("Fallback to %s/%s: could not attach credential pool: %s", fb_provider, fb_model, exc)
|
||||
|
||||
|
||||
# Codex ChatGPT-account entitlement 400 — the account can never use the named slug, so with
|
||||
# nothing to rotate it is a config error, not a transient failure (#106475).
|
||||
_CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER = "model is not supported when using codex with a chatgpt account"
|
||||
|
||||
|
||||
def _mark_entitlement_rejected_model(agent, api_error) -> bool:
|
||||
"""Record a Codex ChatGPT-account 400 that rejects the current model for this account.
|
||||
|
||||
With a single credential there is no pool to rotate (#71970 covers that case), so the
|
||||
(provider, model) pair is treated as dead for the session: the fallback walk skips it and
|
||||
restore_primary_runtime stops switching back — otherwise every turn re-fails on the primary,
|
||||
announces an unverified "Primary model restored", and oscillates forever (#106475).
|
||||
"""
|
||||
if getattr(api_error, "status_code", None) != 400:
|
||||
return False
|
||||
pool = getattr(agent, "_credential_pool", None)
|
||||
if pool is not None:
|
||||
try:
|
||||
if len(pool.entries()) > 1:
|
||||
return False # another account in the pool may be entitled; leave rotation to it
|
||||
except Exception:
|
||||
return False
|
||||
haystack = str(getattr(api_error, "message", "") or api_error).lower()
|
||||
if _CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER not in haystack:
|
||||
return False
|
||||
provider = str(getattr(agent, "provider", "") or "").strip().lower()
|
||||
model = str(getattr(agent, "model", "") or "").strip()
|
||||
if not provider or not model:
|
||||
return False
|
||||
rejected = getattr(agent, "_entitlement_rejected_models", None)
|
||||
if rejected is None:
|
||||
rejected = agent._entitlement_rejected_models = set()
|
||||
if (provider, model) in rejected:
|
||||
return True
|
||||
rejected.add((provider, model))
|
||||
logger.warning(
|
||||
"Model entitlement rejection: this account is not entitled to %s via %s; "
|
||||
"treating it as unavailable for this session",
|
||||
model, provider,
|
||||
)
|
||||
agent._buffer_status(
|
||||
f"🚫 This account is not entitled to {model} via {provider}; it will be skipped "
|
||||
"until restart. Switch to an entitled model via /model or `hermes model`."
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
def _is_entitlement_rejected(agent, provider: str, model: str) -> bool:
|
||||
"""True when (provider, model) — as configured or normalized — was rejected as unentitled
|
||||
for this account (see _mark_entitlement_rejected_model)."""
|
||||
rejected = getattr(agent, "_entitlement_rejected_models", None) or ()
|
||||
if not rejected:
|
||||
return False
|
||||
if (provider, model) in rejected:
|
||||
return True
|
||||
try:
|
||||
from hermes_cli.model_normalize import normalize_model_for_provider
|
||||
return (provider, normalize_model_for_provider(model, provider)) in rejected
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _fallback_chain_exhausted(agent, reason: "FailoverReason | None") -> bool:
|
||||
"""Chain exhausted (always False). A non-empty chain walked on a non-rate-limit failure arms a
|
||||
short cooldown so next turn's restore_primary_runtime stays gated instead of replaying the whole
|
||||
@@ -1731,6 +1793,9 @@ def _should_skip_fallback_candidate(agent, fb: dict, fb_key: tuple, fb_provider:
|
||||
return True
|
||||
if not fb_provider or not fb_model:
|
||||
return True
|
||||
if _is_entitlement_rejected(agent, fb_provider, fb_model):
|
||||
logger.info("Fallback skip: %s/%s was rejected as unentitled for this account", fb_provider, fb_model)
|
||||
return True
|
||||
local_skip_reason = _fallback_entry_unavailable_without_network(agent, fb)
|
||||
if local_skip_reason:
|
||||
unavailable.add(fb_key)
|
||||
|
||||
@@ -282,6 +282,10 @@ def settle_unrecovered_error(
|
||||
) and not is_context_length_error
|
||||
|
||||
if is_client_error:
|
||||
# A Codex ChatGPT-account entitlement 400 names the model: with nothing to rotate the
|
||||
# slug is dead for this account, so record it before the fallback walk runs (#106475).
|
||||
from agent.chat_completion_helpers import _mark_entitlement_rejected_model
|
||||
_mark_entitlement_rejected_model(agent, api_error)
|
||||
# Copilot self-heal BEFORE fallback: a stale credential yields a 400
|
||||
# ``model_not_available_for_integrator`` / ``model_not_supported``, not a 401.
|
||||
# Fresh token + client rebuild, one retry, SAME provider.
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
"""#106475: with a single credential, a Codex ChatGPT-account entitlement 400 must fail
|
||||
closed — the rejected slug is skipped by the fallback walk and restore_primary_runtime
|
||||
must not switch back and announce an unverified "Primary model restored" that would
|
||||
oscillate forever.
|
||||
"""
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from agent import chat_completion_helpers
|
||||
|
||||
from run_agent import AIAgent
|
||||
|
||||
# Assembled at runtime so no credential-shaped literal sits in source (scanner guard).
|
||||
_TEST_KEY = "test-" + "key-12345678"
|
||||
_FALLBACK_KEY = "fallback-" + "key-1234"
|
||||
|
||||
|
||||
def _make_tool_defs(*names: str) -> list:
|
||||
return [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": n,
|
||||
"description": f"{n} tool",
|
||||
"parameters": {"type": "object", "properties": {}},
|
||||
},
|
||||
}
|
||||
for n in names
|
||||
]
|
||||
|
||||
|
||||
def _make_agent(fallback_model=None, provider="custom", base_url="https://my-llm.example.com/v1"):
|
||||
with (
|
||||
patch("model_tools.get_tool_definitions", return_value=_make_tool_defs("web_search")),
|
||||
patch("model_tools.check_toolset_requirements", return_value={}),
|
||||
patch("agent.process_bootstrap.OpenAI"),
|
||||
patch("agent.context_compressor.get_model_context_length", return_value=200_000),
|
||||
patch("agent.anthropic_adapter.build_anthropic_client", return_value=MagicMock()),
|
||||
):
|
||||
agent = AIAgent(
|
||||
api_key=_TEST_KEY,
|
||||
base_url=base_url,
|
||||
provider=provider,
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
fallback_model=fallback_model,
|
||||
)
|
||||
agent.client = MagicMock()
|
||||
return agent
|
||||
|
||||
|
||||
def _entitlement_error(model: str) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
status_code=400,
|
||||
message=(
|
||||
f"The '{model}' model is not supported when using Codex with a ChatGPT account."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _mock_resolve(base_url="https://fallback.example.com/v1", api_key=_FALLBACK_KEY):
|
||||
mock_client = MagicMock()
|
||||
mock_client.api_key = api_key
|
||||
mock_client.base_url = base_url
|
||||
return mock_client
|
||||
|
||||
|
||||
class TestMarkEntitlementRejectedModel:
|
||||
def test_marks_codex_entitlement_400(self):
|
||||
agent = _make_agent()
|
||||
agent.provider = "openai-codex"
|
||||
agent.model = "gpt-5.6-sol"
|
||||
assert chat_completion_helpers._mark_entitlement_rejected_model(
|
||||
agent, _entitlement_error("gpt-5.6-sol")
|
||||
) is True
|
||||
assert agent._entitlement_rejected_models == {("openai-codex", "gpt-5.6-sol")}
|
||||
|
||||
def test_ignores_other_statuses_and_bodies(self):
|
||||
agent = _make_agent()
|
||||
agent.provider = "openai-codex"
|
||||
agent.model = "gpt-5.6-sol"
|
||||
for err in (
|
||||
SimpleNamespace(status_code=500, message="server error"),
|
||||
SimpleNamespace(status_code=429, message="rate limited"),
|
||||
SimpleNamespace(status_code=400, message="invalid request body"),
|
||||
SimpleNamespace(status_code=403, message="model is not supported when using Codex with a ChatGPT account."),
|
||||
):
|
||||
assert chat_completion_helpers._mark_entitlement_rejected_model(agent, err) is False
|
||||
assert getattr(agent, "_entitlement_rejected_models", None) is None
|
||||
|
||||
def test_multi_credential_pool_is_left_to_rotation(self):
|
||||
agent = _make_agent()
|
||||
agent.provider = "openai-codex"
|
||||
agent.model = "gpt-5.6-sol"
|
||||
pool = MagicMock()
|
||||
pool.entries.return_value = [object(), object()]
|
||||
agent._credential_pool = pool
|
||||
assert chat_completion_helpers._mark_entitlement_rejected_model(
|
||||
agent, _entitlement_error("gpt-5.6-sol")
|
||||
) is False
|
||||
assert getattr(agent, "_entitlement_rejected_models", None) is None
|
||||
|
||||
def test_single_credential_pool_still_marks(self):
|
||||
agent = _make_agent()
|
||||
agent.provider = "openai-codex"
|
||||
agent.model = "gpt-5.6-sol"
|
||||
pool = MagicMock()
|
||||
pool.entries.return_value = [object()]
|
||||
agent._credential_pool = pool
|
||||
assert chat_completion_helpers._mark_entitlement_rejected_model(
|
||||
agent, _entitlement_error("gpt-5.6-sol")
|
||||
) is True
|
||||
|
||||
|
||||
class TestFallbackWalkSkipsRejectedSlug:
|
||||
def test_rejected_entry_is_skipped_and_chain_exhausts(self):
|
||||
agent = _make_agent(
|
||||
fallback_model={"provider": "openai-codex", "model": "gpt-5.6-sol"},
|
||||
)
|
||||
agent._entitlement_rejected_models = {("openai-codex", "gpt-5.6-sol")}
|
||||
with patch("agent.auxiliary_client.resolve_provider_client") as resolve:
|
||||
assert agent._try_activate_fallback() is False
|
||||
resolve.assert_not_called()
|
||||
|
||||
def test_normalized_slug_form_is_also_skipped(self):
|
||||
agent = _make_agent(
|
||||
fallback_model={"provider": "openai-codex", "model": "openai/gpt-5.6-sol"},
|
||||
)
|
||||
# The runtime (and the marker) recorded the post-normalization slug.
|
||||
agent._entitlement_rejected_models = {("openai-codex", "gpt-5.6-sol")}
|
||||
with patch("agent.auxiliary_client.resolve_provider_client") as resolve:
|
||||
assert agent._try_activate_fallback() is False
|
||||
resolve.assert_not_called()
|
||||
|
||||
|
||||
class TestRestoreGate:
|
||||
def test_restore_does_not_switch_back_to_rejected_primary(self):
|
||||
agent = _make_agent(
|
||||
fallback_model={"provider": "zai", "model": "glm-5.2"},
|
||||
)
|
||||
agent._primary_runtime["model"] = "gpt-5.6-sol"
|
||||
agent._primary_runtime["provider"] = "openai-codex"
|
||||
agent._entitlement_rejected_models = {("openai-codex", "gpt-5.6-sol")}
|
||||
with patch("agent.auxiliary_client.resolve_provider_client", return_value=(_mock_resolve(), None)):
|
||||
assert agent._try_activate_fallback() is True
|
||||
assert agent._fallback_activated is True
|
||||
|
||||
emitted = []
|
||||
agent._emit_status = emitted.append
|
||||
with patch("agent.process_bootstrap.OpenAI", return_value=MagicMock()):
|
||||
assert agent._restore_primary_runtime() is False
|
||||
|
||||
# Still on the fallback; no unverified "Primary model restored" claim.
|
||||
assert agent.provider == "zai"
|
||||
assert agent.model == "glm-5.2"
|
||||
assert agent._fallback_activated is True
|
||||
assert emitted == []
|
||||
|
||||
def test_unrejected_primary_restores_normally(self):
|
||||
agent = _make_agent(
|
||||
fallback_model={"provider": "zai", "model": "glm-5.2"},
|
||||
)
|
||||
agent._entitlement_rejected_models = {("openai-codex", "some-other-slug")}
|
||||
with patch("agent.auxiliary_client.resolve_provider_client", return_value=(_mock_resolve(), None)):
|
||||
assert agent._try_activate_fallback() is True
|
||||
|
||||
emitted = []
|
||||
agent._emit_status = emitted.append
|
||||
with patch("agent.process_bootstrap.OpenAI", return_value=MagicMock()):
|
||||
assert agent._restore_primary_runtime() is True
|
||||
assert any("Primary model restored" in notice for notice in emitted)
|
||||
Reference in New Issue
Block a user