fix(model_metadata): parse Google's 'supports up to N' context-limit phrasing

Google Gemini/Gemma overflow errors read 'Unable to submit request because
the input token count is 32825 but model only supports up to 32768'.
parse_context_limit_from_error had no pattern for the 'supports up to N'
phrasing, so overflow recovery kept the wrong window and burned its retry
attempts instead of recalibrating to the provider-reported limit.

Add the anchored pattern (limit follows 'supports up to'; the larger input
count before it is never captured) plus regression tests covering the exact
message and the get_context_length_from_provider_error recalibration path.

Reported by @Artemonim in #57275 (residual claim 5).
This commit is contained in:
Teknium
2026-08-31 11:48:57 -07:00
parent 069d9d3529
commit 58f5b1e277
2 changed files with 30 additions and 0 deletions
+7
View File
@@ -1679,6 +1679,7 @@ def parse_context_limit_from_error(error_msg: str) -> Optional[int]:
- "context_length_exceeded: 131072"
- "Maximum context size 32768 exceeded"
- "model's max context length is 65536"
- "input token count is 32825 but model only supports up to 32768"
"""
error_lower = error_msg.lower()
# Pattern: look for numbers near context-related keywords
@@ -1690,6 +1691,12 @@ def parse_context_limit_from_error(error_msg: str) -> Optional[int]:
r'(\d{4,})\s*(?:token)?\s*(?:context|limit)',
r'>\s*(\d{4,})\s*(?:max|limit|token)', # "250000 tokens > 200000 maximum"
r'(\d{4,})\s*(?:max(?:imum)?)\b', # "200000 maximum"
# Google Gemini/Gemma: "Unable to submit request because the input
# token count is 32825 but model only supports up to 32768." The
# limit is the number AFTER "supports up to" — the input count that
# precedes it must not be captured, so this pattern anchors on the
# "supports up to" phrase itself.
r'supports?\s+(?:only\s+)?up\s+to\s+(\d{4,})',
]
for pattern in patterns:
match = re.search(pattern, error_lower)
+23
View File
@@ -1413,6 +1413,29 @@ class TestParseContextLimitFromError:
fell through to None."""
assert parse_context_limit_from_error(msg) == expected
@pytest.mark.parametrize("msg,expected", [
# Google Gemini/Gemma overflow phrasing (#57275): the limit follows
# "supports up to"; the larger input count before it must NOT win.
("Unable to submit request because the input token count is 32825 "
"but model only supports up to 32768. Reduce the input token count "
"and try again.", 32768),
("input token count is 140000 but model only supports up to 131072", 131072),
("model supports up to 65536 tokens", 65536),
])
def test_google_supports_up_to_variants(self, msg, expected):
"""Google's overflow error was previously unparseable — recovery kept
the wrong window and burned its attempts (#57275, residual claim 5)."""
assert parse_context_limit_from_error(msg) == expected
def test_google_supports_up_to_recalibrates_window(self):
from agent.model_metadata import get_context_length_from_provider_error
msg = ("Unable to submit request because the input token count is "
"32825 but model only supports up to 32768.")
assert get_context_length_from_provider_error(msg, 131072) == 32768
# Parsed limit not below current window → no recalibration.
assert get_context_length_from_provider_error(msg, 32768) is None
def test_get_context_length_from_vllm_max_model_len_error(self):
from agent.model_metadata import get_context_length_from_provider_error