fix(model-metadata): converge output-cap retry on vLLM
Fixes the retry loop that spins forever when a vLLM server rejects a
request for having a max_tokens too big for what is left of the context
window.
The catch is that vLLM does not tell you how big your prompt actually is
in that situation. It works the number backwards from the constraint it
just failed, so you get:
"requested 65536 output tokens and your prompt contains at least
36865 input tokens, for a total of at least 102401 tokens"
That 36865 is just window + 1 - requested, and the total is always
exactly window + 1. Subtracting it from the window hands back
requested - 1 every single time, whatever the real prompt size is.
parse_available_output_tokens_from_error believed it and returned
requested - 1. conversation_loop then takes off its 64 token safety
margin and retries, which walks the cap down 65 tokens at a time while
the reported input walks up by the same 65:
65536 -> 65471 -> 65406 -> 65341
Three attempts is the default budget, so the session gives up with
"Context length exceeded" having closed 195 tokens of a roughly 28000
token gap. Compression cannot save it either, because the input was
never the problem, which is why the compressor keeps refusing with
"summary would have GROWN".
This is also what is behind the unexplained "input-token drift" in
issue #61761. The input is not drifting. It is a derived number, and it
moves because we moved max_tokens.
So when that shape shows up (the "at least" wording, plus a budget that
works out to exactly requested - 1), halve the requested cap instead. It
is still guaranteed to sit under whatever was just rejected, and it
converges on the first retry: 65536 -> 32768, which next to a real 36865
token prompt comes to 69633 against a 102400 window.
Nothing else moves. A measured input is still trusted, and a genuine
input overflow still returns None so the caller falls through to
compression the way it always did.
The existing test asserted the bogus 65535, so it is updated. Added
tests for the measured input path, and for the retry actually
converging.
This commit is contained in:
@@ -1726,11 +1726,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
|
||||
# Available output = window - input. When the input alone is at or over
|
||||
# the window this stays None, so the caller correctly falls through to
|
||||
# compression instead of futilely shrinking the output cap.
|
||||
#
|
||||
# Caveat: when max_tokens is the BINDING constraint, vLLM does not report
|
||||
# the real prompt size at all. It back-computes a lower bound from the
|
||||
# constraint itself -- "at least N input tokens" where
|
||||
# N == window + 1 - requested_output -- so window - N is always exactly
|
||||
# requested_output - 1. Subtracting the caller's safety margin then walks
|
||||
# the cap down ~65 tokens per retry while the reported input walks up by
|
||||
# the same amount, burning every compression attempt without ever fitting.
|
||||
# Detect that degenerate case and halve the requested cap instead: it
|
||||
# carries the same guarantee (strictly below what was rejected) and
|
||||
# converges in one or two retries.
|
||||
_m_vllm_input = re.search(
|
||||
r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower
|
||||
)
|
||||
if _m_ctx_tok and _m_vllm_input:
|
||||
_available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1))
|
||||
_m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower)
|
||||
if 'at least' in error_lower and _m_requested_out:
|
||||
_requested_out = int(_m_requested_out.group(1))
|
||||
if _available >= _requested_out - 1:
|
||||
# The budget is derived from the constraint, not measured.
|
||||
return max(1, _requested_out // 2)
|
||||
if _available >= 1:
|
||||
return _available
|
||||
|
||||
|
||||
@@ -121,10 +121,28 @@ class TestParseVllmTokenBasedOutputCap:
|
||||
"output tokens."
|
||||
)
|
||||
|
||||
def test_vllm_token_based_format(self):
|
||||
# available output = 131072 - 65537 = 65535
|
||||
assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 65535
|
||||
# Verbatim vLLM response where the input is MEASURED, not back-computed:
|
||||
# window - input != requested - 1, so the reported figure is real.
|
||||
_VLLM_MSG_REAL_INPUT = (
|
||||
"This model's maximum context length is 131072 tokens. However, you "
|
||||
"requested 65536 output tokens and your prompt contains 100000 "
|
||||
"input tokens, for a total of 165536 tokens. Please reduce the length "
|
||||
"of the input prompt or the number of requested output tokens."
|
||||
)
|
||||
|
||||
def test_vllm_token_based_format(self):
|
||||
# The reported input is a LOWER BOUND that vLLM back-computes from the
|
||||
# constraint (65537 == 131072 + 1 - 65536), so window - input is just
|
||||
# requested - 1 and carries no information about the real prompt.
|
||||
# Halve the requested cap instead so the retry actually converges.
|
||||
assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 32768
|
||||
|
||||
def test_vllm_measured_input_is_trusted(self):
|
||||
# When the input is measured rather than derived, use it as-is.
|
||||
# available output = 131072 - 100000 = 31072
|
||||
assert parse_available_output_tokens_from_error(
|
||||
self._VLLM_MSG_REAL_INPUT
|
||||
) == 31072
|
||||
|
||||
def test_vllm_retry_fits_inside_window(self):
|
||||
# The retried cap plus the reported input must fit in the window.
|
||||
@@ -132,3 +150,28 @@ class TestParseVllmTokenBasedOutputCap:
|
||||
assert available is not None
|
||||
assert available + 65537 <= 131072
|
||||
|
||||
def test_vllm_retry_converges(self):
|
||||
"""The retry sequence must reach a working cap in a few attempts.
|
||||
|
||||
Regression test for the 65-tokens-per-retry crawl: with a 102400
|
||||
window and a real prompt of ~37000 tokens, retrying from a 65536 cap
|
||||
used to produce 65471 -> 65406 -> 65341 and exhaust the compression
|
||||
budget without ever fitting.
|
||||
"""
|
||||
window, real_input, cap = 102400, 37000, 65536
|
||||
for _ in range(5):
|
||||
if real_input + cap <= window:
|
||||
break
|
||||
# vLLM's message when max_tokens is the binding constraint.
|
||||
msg = (
|
||||
f"This model's maximum context length is {window} tokens. "
|
||||
f"However, you requested {cap} output tokens and your prompt "
|
||||
f"contains at least {window + 1 - cap} input tokens, for a "
|
||||
f"total of at least {window + 1} tokens."
|
||||
)
|
||||
available = parse_available_output_tokens_from_error(msg)
|
||||
assert available is not None
|
||||
assert available < cap, "each retry must lower the cap"
|
||||
cap = available
|
||||
assert real_input + cap <= window, f"did not converge: cap={cap}"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user