fix(agent): fail closed when Codex account-model entitlement 400s exhaust the chain

A Codex ChatGPT-account 400 ('The X model is not supported when using Codex
with a ChatGPT account.') names the model, so with a single credential the
slug is dead for that account. The fallback walk still re-selected it and
restore_primary_runtime switched back to the primary at the start of every
turn, announcing an unverified 'Primary model restored' — the two warnings
alternated forever with zero delivered answers (#106475).

Record the rejected (provider, model) pair on the non-retryable client-error
path (only when no multi-credential pool exists — rotation covers that case,
#71970), skip rejected entries during the fallback walk, and gate
restore_primary_runtime on the primary's slug so the session fails closed
with the terminal entitlement error instead of oscillating. Fixes #106475.

(cherry picked from commit 471435b10288f15387b2549a8b02551d82c0f670)
This commit is contained in:
liuhao1024
2026-09-09 18:56:56 +08:00
committed by Teknium
parent 9678d59c23
commit 808c5b30aa
4 changed files with 249 additions and 0 deletions
+7
View File
@@ -1124,6 +1124,13 @@ def restore_primary_runtime(agent) -> bool:
return False # primary still in rate-limit cooldown, stay on fallback
rt = agent._primary_runtime
primary_provider = str((rt or {}).get("provider") or "").strip().lower()
primary_model = str((rt or {}).get("model") or "").strip()
from agent.chat_completion_helpers import _is_entitlement_rejected
if primary_model and _is_entitlement_rejected(agent, primary_provider, primary_model):
# The primary slug was rejected as unentitled for this account (#106475): restoring
# here would announce a recovery that was never verified and re-fail every turn.
# Stay on the fallback; the user sees the terminal entitlement error instead.
return False
primary_runtime_base_url = str((rt or {}).get("base_url") or "")
def _matches_primary(candidate) -> bool:
+65
View File
@@ -1712,6 +1712,68 @@ def _rebind_fallback_credential_pool(agent, fb_provider: str, fb_model: str) ->
logger.debug("Fallback to %s/%s: could not attach credential pool: %s", fb_provider, fb_model, exc)
# Codex ChatGPT-account entitlement 400 — the account can never use the named slug, so with
# nothing to rotate it is a config error, not a transient failure (#106475).
_CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER = "model is not supported when using codex with a chatgpt account"
def _mark_entitlement_rejected_model(agent, api_error) -> bool:
"""Record a Codex ChatGPT-account 400 that rejects the current model for this account.
With a single credential there is no pool to rotate (#71970 covers that case), so the
(provider, model) pair is treated as dead for the session: the fallback walk skips it and
restore_primary_runtime stops switching back — otherwise every turn re-fails on the primary,
announces an unverified "Primary model restored", and oscillates forever (#106475).
"""
if getattr(api_error, "status_code", None) != 400:
return False
pool = getattr(agent, "_credential_pool", None)
if pool is not None:
try:
if len(pool.entries()) > 1:
return False # another account in the pool may be entitled; leave rotation to it
except Exception:
return False
haystack = str(getattr(api_error, "message", "") or api_error).lower()
if _CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER not in haystack:
return False
provider = str(getattr(agent, "provider", "") or "").strip().lower()
model = str(getattr(agent, "model", "") or "").strip()
if not provider or not model:
return False
rejected = getattr(agent, "_entitlement_rejected_models", None)
if rejected is None:
rejected = agent._entitlement_rejected_models = set()
if (provider, model) in rejected:
return True
rejected.add((provider, model))
logger.warning(
"Model entitlement rejection: this account is not entitled to %s via %s; "
"treating it as unavailable for this session",
model, provider,
)
agent._buffer_status(
f"🚫 This account is not entitled to {model} via {provider}; it will be skipped "
"until restart. Switch to an entitled model via /model or `hermes model`."
)
return True
def _is_entitlement_rejected(agent, provider: str, model: str) -> bool:
"""True when (provider, model) — as configured or normalized — was rejected as unentitled
for this account (see _mark_entitlement_rejected_model)."""
rejected = getattr(agent, "_entitlement_rejected_models", None) or ()
if not rejected:
return False
if (provider, model) in rejected:
return True
try:
from hermes_cli.model_normalize import normalize_model_for_provider
return (provider, normalize_model_for_provider(model, provider)) in rejected
except Exception:
return False
def _fallback_chain_exhausted(agent, reason: "FailoverReason | None") -> bool:
"""Chain exhausted (always False). A non-empty chain walked on a non-rate-limit failure arms a
short cooldown so next turn's restore_primary_runtime stays gated instead of replaying the whole
@@ -1731,6 +1793,9 @@ def _should_skip_fallback_candidate(agent, fb: dict, fb_key: tuple, fb_provider:
return True
if not fb_provider or not fb_model:
return True
if _is_entitlement_rejected(agent, fb_provider, fb_model):
logger.info("Fallback skip: %s/%s was rejected as unentitled for this account", fb_provider, fb_model)
return True
local_skip_reason = _fallback_entry_unavailable_without_network(agent, fb)
if local_skip_reason:
unavailable.add(fb_key)
+4
View File
@@ -282,6 +282,10 @@ def settle_unrecovered_error(
) and not is_context_length_error
if is_client_error:
# A Codex ChatGPT-account entitlement 400 names the model: with nothing to rotate the
# slug is dead for this account, so record it before the fallback walk runs (#106475).
from agent.chat_completion_helpers import _mark_entitlement_rejected_model
_mark_entitlement_rejected_model(agent, api_error)
# Copilot self-heal BEFORE fallback: a stale credential yields a 400
# ``model_not_available_for_integrator`` / ``model_not_supported``, not a 401.
# Fresh token + client rebuild, one retry, SAME provider.
@@ -0,0 +1,173 @@
"""#106475: with a single credential, a Codex ChatGPT-account entitlement 400 must fail
closed — the rejected slug is skipped by the fallback walk and restore_primary_runtime
must not switch back and announce an unverified "Primary model restored" that would
oscillate forever.
"""
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
from agent import chat_completion_helpers
from run_agent import AIAgent
# Assembled at runtime so no credential-shaped literal sits in source (scanner guard).
_TEST_KEY = "test-" + "key-12345678"
_FALLBACK_KEY = "fallback-" + "key-1234"
def _make_tool_defs(*names: str) -> list:
return [
{
"type": "function",
"function": {
"name": n,
"description": f"{n} tool",
"parameters": {"type": "object", "properties": {}},
},
}
for n in names
]
def _make_agent(fallback_model=None, provider="custom", base_url="https://my-llm.example.com/v1"):
with (
patch("model_tools.get_tool_definitions", return_value=_make_tool_defs("web_search")),
patch("model_tools.check_toolset_requirements", return_value={}),
patch("agent.process_bootstrap.OpenAI"),
patch("agent.context_compressor.get_model_context_length", return_value=200_000),
patch("agent.anthropic_adapter.build_anthropic_client", return_value=MagicMock()),
):
agent = AIAgent(
api_key=_TEST_KEY,
base_url=base_url,
provider=provider,
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
fallback_model=fallback_model,
)
agent.client = MagicMock()
return agent
def _entitlement_error(model: str) -> SimpleNamespace:
return SimpleNamespace(
status_code=400,
message=(
f"The '{model}' model is not supported when using Codex with a ChatGPT account."
),
)
def _mock_resolve(base_url="https://fallback.example.com/v1", api_key=_FALLBACK_KEY):
mock_client = MagicMock()
mock_client.api_key = api_key
mock_client.base_url = base_url
return mock_client
class TestMarkEntitlementRejectedModel:
def test_marks_codex_entitlement_400(self):
agent = _make_agent()
agent.provider = "openai-codex"
agent.model = "gpt-5.6-sol"
assert chat_completion_helpers._mark_entitlement_rejected_model(
agent, _entitlement_error("gpt-5.6-sol")
) is True
assert agent._entitlement_rejected_models == {("openai-codex", "gpt-5.6-sol")}
def test_ignores_other_statuses_and_bodies(self):
agent = _make_agent()
agent.provider = "openai-codex"
agent.model = "gpt-5.6-sol"
for err in (
SimpleNamespace(status_code=500, message="server error"),
SimpleNamespace(status_code=429, message="rate limited"),
SimpleNamespace(status_code=400, message="invalid request body"),
SimpleNamespace(status_code=403, message="model is not supported when using Codex with a ChatGPT account."),
):
assert chat_completion_helpers._mark_entitlement_rejected_model(agent, err) is False
assert getattr(agent, "_entitlement_rejected_models", None) is None
def test_multi_credential_pool_is_left_to_rotation(self):
agent = _make_agent()
agent.provider = "openai-codex"
agent.model = "gpt-5.6-sol"
pool = MagicMock()
pool.entries.return_value = [object(), object()]
agent._credential_pool = pool
assert chat_completion_helpers._mark_entitlement_rejected_model(
agent, _entitlement_error("gpt-5.6-sol")
) is False
assert getattr(agent, "_entitlement_rejected_models", None) is None
def test_single_credential_pool_still_marks(self):
agent = _make_agent()
agent.provider = "openai-codex"
agent.model = "gpt-5.6-sol"
pool = MagicMock()
pool.entries.return_value = [object()]
agent._credential_pool = pool
assert chat_completion_helpers._mark_entitlement_rejected_model(
agent, _entitlement_error("gpt-5.6-sol")
) is True
class TestFallbackWalkSkipsRejectedSlug:
def test_rejected_entry_is_skipped_and_chain_exhausts(self):
agent = _make_agent(
fallback_model={"provider": "openai-codex", "model": "gpt-5.6-sol"},
)
agent._entitlement_rejected_models = {("openai-codex", "gpt-5.6-sol")}
with patch("agent.auxiliary_client.resolve_provider_client") as resolve:
assert agent._try_activate_fallback() is False
resolve.assert_not_called()
def test_normalized_slug_form_is_also_skipped(self):
agent = _make_agent(
fallback_model={"provider": "openai-codex", "model": "openai/gpt-5.6-sol"},
)
# The runtime (and the marker) recorded the post-normalization slug.
agent._entitlement_rejected_models = {("openai-codex", "gpt-5.6-sol")}
with patch("agent.auxiliary_client.resolve_provider_client") as resolve:
assert agent._try_activate_fallback() is False
resolve.assert_not_called()
class TestRestoreGate:
def test_restore_does_not_switch_back_to_rejected_primary(self):
agent = _make_agent(
fallback_model={"provider": "zai", "model": "glm-5.2"},
)
agent._primary_runtime["model"] = "gpt-5.6-sol"
agent._primary_runtime["provider"] = "openai-codex"
agent._entitlement_rejected_models = {("openai-codex", "gpt-5.6-sol")}
with patch("agent.auxiliary_client.resolve_provider_client", return_value=(_mock_resolve(), None)):
assert agent._try_activate_fallback() is True
assert agent._fallback_activated is True
emitted = []
agent._emit_status = emitted.append
with patch("agent.process_bootstrap.OpenAI", return_value=MagicMock()):
assert agent._restore_primary_runtime() is False
# Still on the fallback; no unverified "Primary model restored" claim.
assert agent.provider == "zai"
assert agent.model == "glm-5.2"
assert agent._fallback_activated is True
assert emitted == []
def test_unrejected_primary_restores_normally(self):
agent = _make_agent(
fallback_model={"provider": "zai", "model": "glm-5.2"},
)
agent._entitlement_rejected_models = {("openai-codex", "some-other-slug")}
with patch("agent.auxiliary_client.resolve_provider_client", return_value=(_mock_resolve(), None)):
assert agent._try_activate_fallback() is True
emitted = []
agent._emit_status = emitted.append
with patch("agent.process_bootstrap.OpenAI", return_value=MagicMock()):
assert agent._restore_primary_runtime() is True
assert any("Primary model restored" in notice for notice in emitted)