fix(model_metadata): parse Google's 'supports up to N' context-limit phrasing
Google Gemini/Gemma overflow errors read 'Unable to submit request because the input token count is 32825 but model only supports up to 32768'. parse_context_limit_from_error had no pattern for the 'supports up to N' phrasing, so overflow recovery kept the wrong window and burned its retry attempts instead of recalibrating to the provider-reported limit. Add the anchored pattern (limit follows 'supports up to'; the larger input count before it is never captured) plus regression tests covering the exact message and the get_context_length_from_provider_error recalibration path. Reported by @Artemonim in #57275 (residual claim 5).
This commit is contained in:
@@ -1679,6 +1679,7 @@ def parse_context_limit_from_error(error_msg: str) -> Optional[int]:
|
||||
- "context_length_exceeded: 131072"
|
||||
- "Maximum context size 32768 exceeded"
|
||||
- "model's max context length is 65536"
|
||||
- "input token count is 32825 but model only supports up to 32768"
|
||||
"""
|
||||
error_lower = error_msg.lower()
|
||||
# Pattern: look for numbers near context-related keywords
|
||||
@@ -1690,6 +1691,12 @@ def parse_context_limit_from_error(error_msg: str) -> Optional[int]:
|
||||
r'(\d{4,})\s*(?:token)?\s*(?:context|limit)',
|
||||
r'>\s*(\d{4,})\s*(?:max|limit|token)', # "250000 tokens > 200000 maximum"
|
||||
r'(\d{4,})\s*(?:max(?:imum)?)\b', # "200000 maximum"
|
||||
# Google Gemini/Gemma: "Unable to submit request because the input
|
||||
# token count is 32825 but model only supports up to 32768." The
|
||||
# limit is the number AFTER "supports up to" — the input count that
|
||||
# precedes it must not be captured, so this pattern anchors on the
|
||||
# "supports up to" phrase itself.
|
||||
r'supports?\s+(?:only\s+)?up\s+to\s+(\d{4,})',
|
||||
]
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, error_lower)
|
||||
|
||||
@@ -1413,6 +1413,29 @@ class TestParseContextLimitFromError:
|
||||
fell through to None."""
|
||||
assert parse_context_limit_from_error(msg) == expected
|
||||
|
||||
@pytest.mark.parametrize("msg,expected", [
|
||||
# Google Gemini/Gemma overflow phrasing (#57275): the limit follows
|
||||
# "supports up to"; the larger input count before it must NOT win.
|
||||
("Unable to submit request because the input token count is 32825 "
|
||||
"but model only supports up to 32768. Reduce the input token count "
|
||||
"and try again.", 32768),
|
||||
("input token count is 140000 but model only supports up to 131072", 131072),
|
||||
("model supports up to 65536 tokens", 65536),
|
||||
])
|
||||
def test_google_supports_up_to_variants(self, msg, expected):
|
||||
"""Google's overflow error was previously unparseable — recovery kept
|
||||
the wrong window and burned its attempts (#57275, residual claim 5)."""
|
||||
assert parse_context_limit_from_error(msg) == expected
|
||||
|
||||
def test_google_supports_up_to_recalibrates_window(self):
|
||||
from agent.model_metadata import get_context_length_from_provider_error
|
||||
|
||||
msg = ("Unable to submit request because the input token count is "
|
||||
"32825 but model only supports up to 32768.")
|
||||
assert get_context_length_from_provider_error(msg, 131072) == 32768
|
||||
# Parsed limit not below current window → no recalibration.
|
||||
assert get_context_length_from_provider_error(msg, 32768) is None
|
||||
|
||||
def test_get_context_length_from_vllm_max_model_len_error(self):
|
||||
from agent.model_metadata import get_context_length_from_provider_error
|
||||
|
||||
|
||||
Reference in New Issue
Block a user