fix(agent): route relay-wrapped output-cap 429s into the output-cap handler
Salvage follow-up for #72283: instead of a second pre-retry clamp block (which bypassed the #55546 clamp+compress path and broke its three regression tests), parse the output cap ONCE at classification time and: - exempt parseable wrapped output-cap 429s from the eager rate-limit provider fallback (a deterministic request-shape failure that failover cannot fix but the clamp fixes in one retry), and - widen is_context_length_error so they reach the SAME #55546 clamp+compress recovery as plain output-cap 400s. Adds both #72283 regression scenarios plus an ordering guard proving a NON-EMPTY fallback chain does not consume the wrapped 429 (fallback slot unspent, model unchanged). 119 fallback/rate-limit tests green.
This commit is contained in:
@@ -5304,6 +5304,21 @@ def run_conversation(
|
||||
FailoverReason.billing,
|
||||
FailoverReason.upstream_rate_limit,
|
||||
}
|
||||
# Relay-wrapped output-cap errors: some gateways wrap an
|
||||
# upstream "[400]: max_tokens (...) exceeds model's maximum
|
||||
# output tokens (...)" as HTTP 429, which classifies as
|
||||
# rate_limit. The failure is a deterministic request-shape
|
||||
# problem — falling back to another provider (or burning
|
||||
# generic retries) can't fix it, but the output-cap clamp
|
||||
# below can, in one retry (#72281). Parse once here; the
|
||||
# result gates both the eager-fallback exemption and the
|
||||
# widened is_context_length_error entry, and is reused as
|
||||
# available_out inside the handler.
|
||||
_wrapped_output_cap_budget = (
|
||||
parse_available_output_tokens_from_error(error_msg)
|
||||
if classified.reason == FailoverReason.rate_limit
|
||||
else None
|
||||
)
|
||||
_is_transport_failure = classified.reason in {
|
||||
FailoverReason.timeout,
|
||||
FailoverReason.overloaded,
|
||||
@@ -5321,7 +5336,7 @@ def run_conversation(
|
||||
if _is_zai_coding_overload:
|
||||
max_retries = max(max_retries, zai_coding_overload_retry_ceiling())
|
||||
_should_fallback = (
|
||||
is_rate_limited
|
||||
(is_rate_limited and _wrapped_output_cap_budget is None)
|
||||
or (_is_transport_failure and retry_count >= 2)
|
||||
)
|
||||
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
|
||||
@@ -5614,6 +5629,11 @@ def run_conversation(
|
||||
# server disconnect + large session pattern (#2153).
|
||||
is_context_length_error = (
|
||||
classified.reason == FailoverReason.context_overflow
|
||||
# Relay-wrapped output-cap 429s (parsed once above, where
|
||||
# the eager-fallback exemption is gated) route into the
|
||||
# output-cap clamp below instead of provider failover or
|
||||
# generic retries (#72281).
|
||||
or _wrapped_output_cap_budget is not None
|
||||
)
|
||||
|
||||
if is_context_length_error:
|
||||
|
||||
@@ -4535,6 +4535,118 @@ class TestRunConversation:
|
||||
assert agent.context_compressor.context_length == 200_000
|
||||
mock_compress.assert_called_once()
|
||||
|
||||
def test_output_cap_retry_before_generic_retry_exhaustion(self, agent):
|
||||
"""Provider max-output-cap 400s clamp via the output-cap handler, not
|
||||
the generic retry loop ("failed after 3 retries").
|
||||
"""
|
||||
self._setup_agent(agent)
|
||||
agent.api_mode = "chat_completions"
|
||||
agent.provider = "deepseek"
|
||||
agent.base_url = "https://api.deepseek.com/v1"
|
||||
agent.model = "deepseek-v4-flash"
|
||||
agent.max_tokens = 98_304
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.should_compress = MagicMock(return_value=False)
|
||||
|
||||
error_msg = (
|
||||
"[400]: max_tokens (98304) exceeds model's maximum output tokens "
|
||||
"(65536) for model deepseek-v4-flash "
|
||||
"(ref: 7735422e-9cb4-4075-a779-dfecb3204a0e)"
|
||||
)
|
||||
exc = Exception(error_msg)
|
||||
exc.status_code = 400
|
||||
exc.code = 400
|
||||
|
||||
ok_resp = _mock_response(content="done", finish_reason="stop")
|
||||
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
|
||||
|
||||
mock_compress = MagicMock(return_value=(
|
||||
[{"role": "user", "content": "hello"}],
|
||||
"You are helpful.",
|
||||
))
|
||||
with (
|
||||
patch.object(agent, "_persist_session"),
|
||||
patch.object(agent, "_save_trajectory"),
|
||||
patch.object(agent, "_cleanup_task_resources"),
|
||||
patch.object(agent.context_compressor, "update_model"),
|
||||
patch.object(agent, "_compress_context", mock_compress),
|
||||
):
|
||||
result = agent.run_conversation("hello")
|
||||
|
||||
assert len(agent.client.chat.completions.create.call_args_list) == 2
|
||||
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
|
||||
assert result["completed"] is True
|
||||
assert second_call["max_tokens"] <= 65_472
|
||||
assert agent.context_compressor.context_length == 200_000
|
||||
|
||||
def test_output_cap_retry_when_gateway_wraps_error_as_rate_limit(self, agent):
|
||||
"""Some relays wrap the upstream max-output 400 as HTTP 429. The
|
||||
parseable output cap must still route into the output-cap handler
|
||||
instead of burning generic rate-limit retries (#72281).
|
||||
"""
|
||||
self._run_wrapped_429_output_cap(agent, fallback_chain=[])
|
||||
|
||||
def test_wrapped_output_cap_429_not_consumed_by_eager_fallback(self, agent):
|
||||
"""With a NON-EMPTY fallback chain, the eager rate-limit fallback must
|
||||
NOT consume the wrapped output-cap 429 — the failure is a deterministic
|
||||
request-shape problem the clamp fixes in one retry; switching provider
|
||||
burns a fallback slot for nothing (#72281 ordering guard).
|
||||
"""
|
||||
self._run_wrapped_429_output_cap(
|
||||
agent,
|
||||
fallback_chain=[{"provider": "openrouter", "model": "anthropic/claude-sonnet-4"}],
|
||||
)
|
||||
|
||||
def _run_wrapped_429_output_cap(self, agent, *, fallback_chain):
|
||||
self._setup_agent(agent)
|
||||
agent.api_mode = "chat_completions"
|
||||
agent.provider = "custom"
|
||||
agent.base_url = "http://192.168.1.254:20128/v1"
|
||||
agent.model = "deepseekv4flash"
|
||||
agent.max_tokens = 98_304
|
||||
agent.compression_enabled = True
|
||||
agent._fallback_chain = fallback_chain
|
||||
agent._fallback_index = 0
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.should_compress = MagicMock(return_value=False)
|
||||
|
||||
error_msg = (
|
||||
"Error code: 429 - {'error': {'message': \"[400]: max_tokens "
|
||||
"(98304) exceeds model's maximum output tokens (65536) for model "
|
||||
"deepseek-v4-flash (ref: 37bde60f-44e7-44e2-b995-4af17fba6d6b)\", "
|
||||
"'type': 'rate_limit_error', 'code': 'rate_limit_exceeded'}}"
|
||||
)
|
||||
exc = Exception(error_msg)
|
||||
exc.status_code = 429
|
||||
exc.code = "rate_limit_exceeded"
|
||||
|
||||
ok_resp = _mock_response(content="done", finish_reason="stop")
|
||||
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
|
||||
|
||||
mock_compress = MagicMock(return_value=(
|
||||
[{"role": "user", "content": "hello"}],
|
||||
"You are helpful.",
|
||||
))
|
||||
with (
|
||||
patch.object(agent, "_persist_session"),
|
||||
patch.object(agent, "_save_trajectory"),
|
||||
patch.object(agent, "_cleanup_task_resources"),
|
||||
patch.object(agent.context_compressor, "update_model"),
|
||||
patch.object(agent, "_compress_context", mock_compress),
|
||||
):
|
||||
result = agent.run_conversation("hello")
|
||||
|
||||
assert len(agent.client.chat.completions.create.call_args_list) == 2
|
||||
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
|
||||
assert result["completed"] is True
|
||||
assert second_call["max_tokens"] <= 65_472
|
||||
assert agent.context_compressor.context_length == 200_000
|
||||
# The clamp, not provider failover, must have recovered: no fallback
|
||||
# slot consumed and the model unchanged.
|
||||
assert agent._fallback_index == 0
|
||||
assert agent.model == "deepseekv4flash"
|
||||
|
||||
def test_output_cap_retry_with_large_api_only_content(self, agent):
|
||||
"""When a large system prompt makes api_messages huge while persisted
|
||||
messages stay tiny, the retry cap must still respect provider
|
||||
|
||||
Reference in New Issue
Block a user