Refresh context-length zero-guard on current upstream/main

Reapply the non-positive context-length guards onto the post-history-replacement
mainline without carrying any stale branch history. save_context_length() now
refuses to persist length <= 0 (keeping upstream's normalized _context_cache_key),
and get_model_context_length() drops non-positive cache hits at the head of the
invalidation chain (Codex/Kimi/MiniMax/Grok branches become elif) so a poisoned
entry re-resolves instead of short-circuiting to 0.

Refresh of PR #25812; original head d62ed5eb92f057d8c707ba937b44f168f2df0677.
This commit is contained in:
Omar Baradei
2026-07-09 17:46:55 -07:00
committed by Teknium
parent e49a7fe568
commit 2fb28ad3d4
+21 -1
View File
@@ -1452,6 +1452,15 @@ def save_context_length(model: str, base_url: str, length: int) -> None:
Cache key is ``model@base_url`` so the same model name served from
different providers can have different limits.
"""
# Never persist non-positive values — a 0 or negative context length
# is always a bug and would poison the cache, causing downstream
# `get_model_context_length()` to return 0 (since `0 is not None`).
if length <= 0:
logger.warning(
"Refusing to cache non-positive context length %s -> %s tokens",
f"{model}@{base_url}", length,
)
return
key = _context_cache_key(model, base_url)
cache = _load_context_cache()
if cache.get(key) == length:
@@ -2664,8 +2673,19 @@ def get_model_context_length(
if base_url and not _skip_persistent_context_cache(base_url, provider):
cached = get_cached_context_length(model, base_url)
if cached is not None:
# Reject non-positive cached values — a 0 or negative value
# is always a bug (corrupted cache, probe failure, or manual
# edit). Without this guard, `0 is not None` short-circuits
# the resolution chain and the compressor gets context_length=0,
# breaking every status-bar and /usage display downstream.
if cached <= 0:
logger.warning(
"Dropping non-positive cache entry %s@%s -> %s; re-resolving",
model, base_url, cached,
)
_invalidate_cached_context_length(model, base_url)
# Invalidate stale 32k cache entries for Kimi-family models.
if cached <= 32768 and _model_name_suggests_kimi(model):
elif cached <= 32768 and _model_name_suggests_kimi(model):
logger.info(
"Dropping stale Kimi cache entry %s@%s -> %s (OpenRouter underreport); "
"re-resolving via hardcoded defaults",