fix(agent): route relay-wrapped output-cap 429s into the output-cap handler

Salvage follow-up for #72283: instead of a second pre-retry clamp block
(which bypassed the #55546 clamp+compress path and broke its three
regression tests), parse the output cap ONCE at classification time and:
- exempt parseable wrapped output-cap 429s from the eager rate-limit
  provider fallback (a deterministic request-shape failure that failover
  cannot fix but the clamp fixes in one retry), and
- widen is_context_length_error so they reach the SAME #55546
  clamp+compress recovery as plain output-cap 400s.

Adds both #72283 regression scenarios plus an ordering guard proving a
NON-EMPTY fallback chain does not consume the wrapped 429 (fallback
slot unspent, model unchanged). 119 fallback/rate-limit tests green.
This commit is contained in:
kshitijk4poor
2026-08-20 11:12:00 +05:30
committed by kshitij
parent 99c980f466
commit b7e12decc6
2 changed files with 133 additions and 1 deletions
+21 -1
View File
@@ -5304,6 +5304,21 @@ def run_conversation(
FailoverReason.billing,
FailoverReason.upstream_rate_limit,
}
# Relay-wrapped output-cap errors: some gateways wrap an
# upstream "[400]: max_tokens (...) exceeds model's maximum
# output tokens (...)" as HTTP 429, which classifies as
# rate_limit. The failure is a deterministic request-shape
# problem — falling back to another provider (or burning
# generic retries) can't fix it, but the output-cap clamp
# below can, in one retry (#72281). Parse once here; the
# result gates both the eager-fallback exemption and the
# widened is_context_length_error entry, and is reused as
# available_out inside the handler.
_wrapped_output_cap_budget = (
parse_available_output_tokens_from_error(error_msg)
if classified.reason == FailoverReason.rate_limit
else None
)
_is_transport_failure = classified.reason in {
FailoverReason.timeout,
FailoverReason.overloaded,
@@ -5321,7 +5336,7 @@ def run_conversation(
if _is_zai_coding_overload:
max_retries = max(max_retries, zai_coding_overload_retry_ceiling())
_should_fallback = (
is_rate_limited
(is_rate_limited and _wrapped_output_cap_budget is None)
or (_is_transport_failure and retry_count >= 2)
)
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
@@ -5614,6 +5629,11 @@ def run_conversation(
# server disconnect + large session pattern (#2153).
is_context_length_error = (
classified.reason == FailoverReason.context_overflow
# Relay-wrapped output-cap 429s (parsed once above, where
# the eager-fallback exemption is gated) route into the
# output-cap clamp below instead of provider failover or
# generic retries (#72281).
or _wrapped_output_cap_budget is not None
)
if is_context_length_error:
+112
View File
@@ -4535,6 +4535,118 @@ class TestRunConversation:
assert agent.context_compressor.context_length == 200_000
mock_compress.assert_called_once()
def test_output_cap_retry_before_generic_retry_exhaustion(self, agent):
"""Provider max-output-cap 400s clamp via the output-cap handler, not
the generic retry loop ("failed after 3 retries").
"""
self._setup_agent(agent)
agent.api_mode = "chat_completions"
agent.provider = "deepseek"
agent.base_url = "https://api.deepseek.com/v1"
agent.model = "deepseek-v4-flash"
agent.max_tokens = 98_304
agent.compression_enabled = True
agent.context_compressor.context_length = 200_000
agent.context_compressor.should_compress = MagicMock(return_value=False)
error_msg = (
"[400]: max_tokens (98304) exceeds model's maximum output tokens "
"(65536) for model deepseek-v4-flash "
"(ref: 7735422e-9cb4-4075-a779-dfecb3204a0e)"
)
exc = Exception(error_msg)
exc.status_code = 400
exc.code = 400
ok_resp = _mock_response(content="done", finish_reason="stop")
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
mock_compress = MagicMock(return_value=(
[{"role": "user", "content": "hello"}],
"You are helpful.",
))
with (
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
patch.object(agent, "_cleanup_task_resources"),
patch.object(agent.context_compressor, "update_model"),
patch.object(agent, "_compress_context", mock_compress),
):
result = agent.run_conversation("hello")
assert len(agent.client.chat.completions.create.call_args_list) == 2
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
assert result["completed"] is True
assert second_call["max_tokens"] <= 65_472
assert agent.context_compressor.context_length == 200_000
def test_output_cap_retry_when_gateway_wraps_error_as_rate_limit(self, agent):
"""Some relays wrap the upstream max-output 400 as HTTP 429. The
parseable output cap must still route into the output-cap handler
instead of burning generic rate-limit retries (#72281).
"""
self._run_wrapped_429_output_cap(agent, fallback_chain=[])
def test_wrapped_output_cap_429_not_consumed_by_eager_fallback(self, agent):
"""With a NON-EMPTY fallback chain, the eager rate-limit fallback must
NOT consume the wrapped output-cap 429 — the failure is a deterministic
request-shape problem the clamp fixes in one retry; switching provider
burns a fallback slot for nothing (#72281 ordering guard).
"""
self._run_wrapped_429_output_cap(
agent,
fallback_chain=[{"provider": "openrouter", "model": "anthropic/claude-sonnet-4"}],
)
def _run_wrapped_429_output_cap(self, agent, *, fallback_chain):
self._setup_agent(agent)
agent.api_mode = "chat_completions"
agent.provider = "custom"
agent.base_url = "http://192.168.1.254:20128/v1"
agent.model = "deepseekv4flash"
agent.max_tokens = 98_304
agent.compression_enabled = True
agent._fallback_chain = fallback_chain
agent._fallback_index = 0
agent.context_compressor.context_length = 200_000
agent.context_compressor.should_compress = MagicMock(return_value=False)
error_msg = (
"Error code: 429 - {'error': {'message': \"[400]: max_tokens "
"(98304) exceeds model's maximum output tokens (65536) for model "
"deepseek-v4-flash (ref: 37bde60f-44e7-44e2-b995-4af17fba6d6b)\", "
"'type': 'rate_limit_error', 'code': 'rate_limit_exceeded'}}"
)
exc = Exception(error_msg)
exc.status_code = 429
exc.code = "rate_limit_exceeded"
ok_resp = _mock_response(content="done", finish_reason="stop")
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
mock_compress = MagicMock(return_value=(
[{"role": "user", "content": "hello"}],
"You are helpful.",
))
with (
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
patch.object(agent, "_cleanup_task_resources"),
patch.object(agent.context_compressor, "update_model"),
patch.object(agent, "_compress_context", mock_compress),
):
result = agent.run_conversation("hello")
assert len(agent.client.chat.completions.create.call_args_list) == 2
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
assert result["completed"] is True
assert second_call["max_tokens"] <= 65_472
assert agent.context_compressor.context_length == 200_000
# The clamp, not provider failover, must have recovered: no fallback
# slot consumed and the model unchanged.
assert agent._fallback_index == 0
assert agent.model == "deepseekv4flash"
def test_output_cap_retry_with_large_api_only_content(self, agent):
"""When a large system prompt makes api_messages huge while persisted
messages stay tiny, the retry cap must still respect provider