From 99c980f466602d89b3b55150653136945e6f53c8 Mon Sep 17 00:00:00 2001 From: ekinnee Date: Thu, 20 Aug 2026 11:09:48 +0530 Subject: [PATCH] fix(model-metadata): parse 'exceeds model maximum output tokens' cap errors Recognizes the DeepSeek/OpenAI-compatible relay wording max_tokens (98304) exceeds model's maximum output tokens (65536) in both parse_available_output_tokens_from_error (returns the cap) and is_output_cap_error (keeps the 400 out of the compression death-loop). Salvaged from PR #72283; the conversation_loop early-clamp block was dropped in favor of routing through the existing output-cap handler (follow-up commit). --- agent/model_metadata.py | 20 ++++++++++++++++++++ tests/test_output_cap_parsing.py | 16 ++++++++++++++++ 2 files changed, 36 insertions(+) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index b2e96d9f20..0bbbaea061 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]: # The input itself fits — this is purely an output-cap error, so reduce # max_tokens and retry; do NOT compress. "range of max_tokens should be" in error_lower + ) or ( + # OpenAI-compatible relays may reject a request whose output cap exceeds + # the model's separate completion-token limit, e.g. + # "max_tokens (98304) exceeds model's maximum output tokens (65536)" + # This is independent of the input context window. + "exceeds model" in error_lower + and "maximum output tokens" in error_lower ) if not is_output_cap_error: return None + # Generic model-output-cap form: + # "max_tokens (98304) exceeds model's maximum output tokens (65536)" + _m_max_output = re.search( + r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?', + error_lower, + ) + if _m_max_output: + _cap = int(_m_max_output.group(1)) + if _cap >= 1: + return _cap + # DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]". # The upper bound is the available output cap. _m_range = re.search( @@ -1799,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool: or "should be" in error_lower # generic "max_tokens should be <= N" or "less than or equal" in error_lower or "must be" in error_lower + or ("exceeds model" in error_lower + and "maximum output tokens" in error_lower) ) if not output_cap_signal: return False diff --git a/tests/test_output_cap_parsing.py b/tests/test_output_cap_parsing.py index 1825569976..aa64d09a7e 100644 --- a/tests/test_output_cap_parsing.py +++ b/tests/test_output_cap_parsing.py @@ -68,6 +68,22 @@ class TestParseDashScopeOutputCap: assert parse_available_output_tokens_from_error(msg) == 32768 +class TestParseMaximumOutputTokensCap: + """Some OpenAI-compatible relays report the model's separate output cap.""" + + def test_parenthesized_max_output_cap(self): + msg = ( + "API call failed after 3 retries: [400]: max_tokens (98304) " + "exceeds model's maximum output tokens (65536)" + ) + assert parse_available_output_tokens_from_error(msg) == 65536 + + def test_parenthesized_max_output_cap_is_output_cap(self): + assert is_output_cap_error( + "max_tokens (98304) exceeds model's maximum output tokens (65536)" + ) is True + + class TestIsOutputCapError: """`is_output_cap_error` is the broader yes/no gate that keeps an output-cap 400 out of the compression death-loop even when we can't parse