diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index e0250124ad..768d8e019e 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -5304,6 +5304,21 @@ def run_conversation( FailoverReason.billing, FailoverReason.upstream_rate_limit, } + # Relay-wrapped output-cap errors: some gateways wrap an + # upstream "[400]: max_tokens (...) exceeds model's maximum + # output tokens (...)" as HTTP 429, which classifies as + # rate_limit. The failure is a deterministic request-shape + # problem — falling back to another provider (or burning + # generic retries) can't fix it, but the output-cap clamp + # below can, in one retry (#72281). Parse once here; the + # result gates both the eager-fallback exemption and the + # widened is_context_length_error entry, and is reused as + # available_out inside the handler. + _wrapped_output_cap_budget = ( + parse_available_output_tokens_from_error(error_msg) + if classified.reason == FailoverReason.rate_limit + else None + ) _is_transport_failure = classified.reason in { FailoverReason.timeout, FailoverReason.overloaded, @@ -5321,7 +5336,7 @@ def run_conversation( if _is_zai_coding_overload: max_retries = max(max_retries, zai_coding_overload_retry_ceiling()) _should_fallback = ( - is_rate_limited + (is_rate_limited and _wrapped_output_cap_budget is None) or (_is_transport_failure and retry_count >= 2) ) if _should_fallback and agent._fallback_index < len(agent._fallback_chain): @@ -5614,6 +5629,11 @@ def run_conversation( # server disconnect + large session pattern (#2153). is_context_length_error = ( classified.reason == FailoverReason.context_overflow + # Relay-wrapped output-cap 429s (parsed once above, where + # the eager-fallback exemption is gated) route into the + # output-cap clamp below instead of provider failover or + # generic retries (#72281). + or _wrapped_output_cap_budget is not None ) if is_context_length_error: diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 68eb70bcc8..a72da553be 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -4535,6 +4535,118 @@ class TestRunConversation: assert agent.context_compressor.context_length == 200_000 mock_compress.assert_called_once() + def test_output_cap_retry_before_generic_retry_exhaustion(self, agent): + """Provider max-output-cap 400s clamp via the output-cap handler, not + the generic retry loop ("failed after 3 retries"). + """ + self._setup_agent(agent) + agent.api_mode = "chat_completions" + agent.provider = "deepseek" + agent.base_url = "https://api.deepseek.com/v1" + agent.model = "deepseek-v4-flash" + agent.max_tokens = 98_304 + agent.compression_enabled = True + agent.context_compressor.context_length = 200_000 + agent.context_compressor.should_compress = MagicMock(return_value=False) + + error_msg = ( + "[400]: max_tokens (98304) exceeds model's maximum output tokens " + "(65536) for model deepseek-v4-flash " + "(ref: 7735422e-9cb4-4075-a779-dfecb3204a0e)" + ) + exc = Exception(error_msg) + exc.status_code = 400 + exc.code = 400 + + ok_resp = _mock_response(content="done", finish_reason="stop") + agent.client.chat.completions.create.side_effect = [exc, ok_resp] + + mock_compress = MagicMock(return_value=( + [{"role": "user", "content": "hello"}], + "You are helpful.", + )) + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(agent.context_compressor, "update_model"), + patch.object(agent, "_compress_context", mock_compress), + ): + result = agent.run_conversation("hello") + + assert len(agent.client.chat.completions.create.call_args_list) == 2 + second_call = agent.client.chat.completions.create.call_args_list[1].kwargs + assert result["completed"] is True + assert second_call["max_tokens"] <= 65_472 + assert agent.context_compressor.context_length == 200_000 + + def test_output_cap_retry_when_gateway_wraps_error_as_rate_limit(self, agent): + """Some relays wrap the upstream max-output 400 as HTTP 429. The + parseable output cap must still route into the output-cap handler + instead of burning generic rate-limit retries (#72281). + """ + self._run_wrapped_429_output_cap(agent, fallback_chain=[]) + + def test_wrapped_output_cap_429_not_consumed_by_eager_fallback(self, agent): + """With a NON-EMPTY fallback chain, the eager rate-limit fallback must + NOT consume the wrapped output-cap 429 — the failure is a deterministic + request-shape problem the clamp fixes in one retry; switching provider + burns a fallback slot for nothing (#72281 ordering guard). + """ + self._run_wrapped_429_output_cap( + agent, + fallback_chain=[{"provider": "openrouter", "model": "anthropic/claude-sonnet-4"}], + ) + + def _run_wrapped_429_output_cap(self, agent, *, fallback_chain): + self._setup_agent(agent) + agent.api_mode = "chat_completions" + agent.provider = "custom" + agent.base_url = "http://192.168.1.254:20128/v1" + agent.model = "deepseekv4flash" + agent.max_tokens = 98_304 + agent.compression_enabled = True + agent._fallback_chain = fallback_chain + agent._fallback_index = 0 + agent.context_compressor.context_length = 200_000 + agent.context_compressor.should_compress = MagicMock(return_value=False) + + error_msg = ( + "Error code: 429 - {'error': {'message': \"[400]: max_tokens " + "(98304) exceeds model's maximum output tokens (65536) for model " + "deepseek-v4-flash (ref: 37bde60f-44e7-44e2-b995-4af17fba6d6b)\", " + "'type': 'rate_limit_error', 'code': 'rate_limit_exceeded'}}" + ) + exc = Exception(error_msg) + exc.status_code = 429 + exc.code = "rate_limit_exceeded" + + ok_resp = _mock_response(content="done", finish_reason="stop") + agent.client.chat.completions.create.side_effect = [exc, ok_resp] + + mock_compress = MagicMock(return_value=( + [{"role": "user", "content": "hello"}], + "You are helpful.", + )) + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(agent.context_compressor, "update_model"), + patch.object(agent, "_compress_context", mock_compress), + ): + result = agent.run_conversation("hello") + + assert len(agent.client.chat.completions.create.call_args_list) == 2 + second_call = agent.client.chat.completions.create.call_args_list[1].kwargs + assert result["completed"] is True + assert second_call["max_tokens"] <= 65_472 + assert agent.context_compressor.context_length == 200_000 + # The clamp, not provider failover, must have recovered: no fallback + # slot consumed and the model unchanged. + assert agent._fallback_index == 0 + assert agent.model == "deepseekv4flash" + def test_output_cap_retry_with_large_api_only_content(self, agent): """When a large system prompt makes api_messages huge while persisted messages stay tiny, the retry cap must still respect provider