diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 57168fc960..b2e96d9f20 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1726,11 +1726,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]: # Available output = window - input. When the input alone is at or over # the window this stays None, so the caller correctly falls through to # compression instead of futilely shrinking the output cap. + # + # Caveat: when max_tokens is the BINDING constraint, vLLM does not report + # the real prompt size at all. It back-computes a lower bound from the + # constraint itself -- "at least N input tokens" where + # N == window + 1 - requested_output -- so window - N is always exactly + # requested_output - 1. Subtracting the caller's safety margin then walks + # the cap down ~65 tokens per retry while the reported input walks up by + # the same amount, burning every compression attempt without ever fitting. + # Detect that degenerate case and halve the requested cap instead: it + # carries the same guarantee (strictly below what was rejected) and + # converges in one or two retries. _m_vllm_input = re.search( r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower ) if _m_ctx_tok and _m_vllm_input: _available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1)) + _m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower) + if 'at least' in error_lower and _m_requested_out: + _requested_out = int(_m_requested_out.group(1)) + if _available >= _requested_out - 1: + # The budget is derived from the constraint, not measured. + return max(1, _requested_out // 2) if _available >= 1: return _available diff --git a/tests/test_output_cap_parsing.py b/tests/test_output_cap_parsing.py index d477ef5da1..1825569976 100644 --- a/tests/test_output_cap_parsing.py +++ b/tests/test_output_cap_parsing.py @@ -121,10 +121,28 @@ class TestParseVllmTokenBasedOutputCap: "output tokens." ) - def test_vllm_token_based_format(self): - # available output = 131072 - 65537 = 65535 - assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 65535 + # Verbatim vLLM response where the input is MEASURED, not back-computed: + # window - input != requested - 1, so the reported figure is real. + _VLLM_MSG_REAL_INPUT = ( + "This model's maximum context length is 131072 tokens. However, you " + "requested 65536 output tokens and your prompt contains 100000 " + "input tokens, for a total of 165536 tokens. Please reduce the length " + "of the input prompt or the number of requested output tokens." + ) + def test_vllm_token_based_format(self): + # The reported input is a LOWER BOUND that vLLM back-computes from the + # constraint (65537 == 131072 + 1 - 65536), so window - input is just + # requested - 1 and carries no information about the real prompt. + # Halve the requested cap instead so the retry actually converges. + assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 32768 + + def test_vllm_measured_input_is_trusted(self): + # When the input is measured rather than derived, use it as-is. + # available output = 131072 - 100000 = 31072 + assert parse_available_output_tokens_from_error( + self._VLLM_MSG_REAL_INPUT + ) == 31072 def test_vllm_retry_fits_inside_window(self): # The retried cap plus the reported input must fit in the window. @@ -132,3 +150,28 @@ class TestParseVllmTokenBasedOutputCap: assert available is not None assert available + 65537 <= 131072 + def test_vllm_retry_converges(self): + """The retry sequence must reach a working cap in a few attempts. + + Regression test for the 65-tokens-per-retry crawl: with a 102400 + window and a real prompt of ~37000 tokens, retrying from a 65536 cap + used to produce 65471 -> 65406 -> 65341 and exhaust the compression + budget without ever fitting. + """ + window, real_input, cap = 102400, 37000, 65536 + for _ in range(5): + if real_input + cap <= window: + break + # vLLM's message when max_tokens is the binding constraint. + msg = ( + f"This model's maximum context length is {window} tokens. " + f"However, you requested {cap} output tokens and your prompt " + f"contains at least {window + 1 - cap} input tokens, for a " + f"total of at least {window + 1} tokens." + ) + available = parse_available_output_tokens_from_error(msg) + assert available is not None + assert available < cap, "each retry must lower the cap" + cap = available + assert real_input + cap <= window, f"did not converge: cap={cap}" +