From 7f2733b71c12009a6d67b74f829aa3dafd2db9bf Mon Sep 17 00:00:00 2001 From: Dhruv Modi Date: Wed, 19 Aug 2026 07:06:39 +0000 Subject: [PATCH] fix(model-metadata): converge output-cap retry on vLLM Fixes the retry loop that spins forever when a vLLM server rejects a request for having a max_tokens too big for what is left of the context window. The catch is that vLLM does not tell you how big your prompt actually is in that situation. It works the number backwards from the constraint it just failed, so you get: "requested 65536 output tokens and your prompt contains at least 36865 input tokens, for a total of at least 102401 tokens" That 36865 is just window + 1 - requested, and the total is always exactly window + 1. Subtracting it from the window hands back requested - 1 every single time, whatever the real prompt size is. parse_available_output_tokens_from_error believed it and returned requested - 1. conversation_loop then takes off its 64 token safety margin and retries, which walks the cap down 65 tokens at a time while the reported input walks up by the same 65: 65536 -> 65471 -> 65406 -> 65341 Three attempts is the default budget, so the session gives up with "Context length exceeded" having closed 195 tokens of a roughly 28000 token gap. Compression cannot save it either, because the input was never the problem, which is why the compressor keeps refusing with "summary would have GROWN". This is also what is behind the unexplained "input-token drift" in issue #61761. The input is not drifting. It is a derived number, and it moves because we moved max_tokens. So when that shape shows up (the "at least" wording, plus a budget that works out to exactly requested - 1), halve the requested cap instead. It is still guaranteed to sit under whatever was just rejected, and it converges on the first retry: 65536 -> 32768, which next to a real 36865 token prompt comes to 69633 against a 102400 window. Nothing else moves. A measured input is still trusted, and a genuine input overflow still returns None so the caller falls through to compression the way it always did. The existing test asserted the bogus 65535, so it is updated. Added tests for the measured input path, and for the retry actually converging. --- agent/model_metadata.py | 17 +++++++++++ tests/test_output_cap_parsing.py | 49 ++++++++++++++++++++++++++++++-- 2 files changed, 63 insertions(+), 3 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 57168fc960..b2e96d9f20 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1726,11 +1726,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]: # Available output = window - input. When the input alone is at or over # the window this stays None, so the caller correctly falls through to # compression instead of futilely shrinking the output cap. + # + # Caveat: when max_tokens is the BINDING constraint, vLLM does not report + # the real prompt size at all. It back-computes a lower bound from the + # constraint itself -- "at least N input tokens" where + # N == window + 1 - requested_output -- so window - N is always exactly + # requested_output - 1. Subtracting the caller's safety margin then walks + # the cap down ~65 tokens per retry while the reported input walks up by + # the same amount, burning every compression attempt without ever fitting. + # Detect that degenerate case and halve the requested cap instead: it + # carries the same guarantee (strictly below what was rejected) and + # converges in one or two retries. _m_vllm_input = re.search( r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower ) if _m_ctx_tok and _m_vllm_input: _available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1)) + _m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower) + if 'at least' in error_lower and _m_requested_out: + _requested_out = int(_m_requested_out.group(1)) + if _available >= _requested_out - 1: + # The budget is derived from the constraint, not measured. + return max(1, _requested_out // 2) if _available >= 1: return _available diff --git a/tests/test_output_cap_parsing.py b/tests/test_output_cap_parsing.py index d477ef5da1..1825569976 100644 --- a/tests/test_output_cap_parsing.py +++ b/tests/test_output_cap_parsing.py @@ -121,10 +121,28 @@ class TestParseVllmTokenBasedOutputCap: "output tokens." ) - def test_vllm_token_based_format(self): - # available output = 131072 - 65537 = 65535 - assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 65535 + # Verbatim vLLM response where the input is MEASURED, not back-computed: + # window - input != requested - 1, so the reported figure is real. + _VLLM_MSG_REAL_INPUT = ( + "This model's maximum context length is 131072 tokens. However, you " + "requested 65536 output tokens and your prompt contains 100000 " + "input tokens, for a total of 165536 tokens. Please reduce the length " + "of the input prompt or the number of requested output tokens." + ) + def test_vllm_token_based_format(self): + # The reported input is a LOWER BOUND that vLLM back-computes from the + # constraint (65537 == 131072 + 1 - 65536), so window - input is just + # requested - 1 and carries no information about the real prompt. + # Halve the requested cap instead so the retry actually converges. + assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 32768 + + def test_vllm_measured_input_is_trusted(self): + # When the input is measured rather than derived, use it as-is. + # available output = 131072 - 100000 = 31072 + assert parse_available_output_tokens_from_error( + self._VLLM_MSG_REAL_INPUT + ) == 31072 def test_vllm_retry_fits_inside_window(self): # The retried cap plus the reported input must fit in the window. @@ -132,3 +150,28 @@ class TestParseVllmTokenBasedOutputCap: assert available is not None assert available + 65537 <= 131072 + def test_vllm_retry_converges(self): + """The retry sequence must reach a working cap in a few attempts. + + Regression test for the 65-tokens-per-retry crawl: with a 102400 + window and a real prompt of ~37000 tokens, retrying from a 65536 cap + used to produce 65471 -> 65406 -> 65341 and exhaust the compression + budget without ever fitting. + """ + window, real_input, cap = 102400, 37000, 65536 + for _ in range(5): + if real_input + cap <= window: + break + # vLLM's message when max_tokens is the binding constraint. + msg = ( + f"This model's maximum context length is {window} tokens. " + f"However, you requested {cap} output tokens and your prompt " + f"contains at least {window + 1 - cap} input tokens, for a " + f"total of at least {window + 1} tokens." + ) + available = parse_available_output_tokens_from_error(msg) + assert available is not None + assert available < cap, "each retry must lower the cap" + cap = available + assert real_input + cap <= window, f"did not converge: cap={cap}" +