fix(agent): reject stale 32k metadata for MiniMax

This commit is contained in:
luoxiao6645
2026-05-12 23:51:03 +08:00
committed by Teknium
parent 69b27e3c74
commit b09e1daa84
2 changed files with 88 additions and 8 deletions
+31 -8
View File
@@ -2082,6 +2082,22 @@ def _stale_pre_catalog_cache_entry(model: str, cached: int) -> bool:
return cached <= threshold
def _model_name_suggests_minimax(model: str) -> bool:
"""Return True if the model name looks like a MiniMax-family model.
Catches ``MiniMax-M2.7``, ``minimax-m2.5``, ``MiniMaxAI/MiniMax-M2.5``,
and similar variants. Used as a guard against stale 32K metadata that
underreports the MiniMax M2 family, whose real context window is 204.8K.
"""
lower = model.lower()
return lower.startswith("minimax") or "minimaxai/" in lower
def _model_name_suggests_stale_32k_underreport(model: str) -> bool:
"""Return True for model families known to be wrongly underreported as 32K."""
return _model_name_suggests_kimi(model) or _model_name_suggests_minimax(model)
def _query_local_context_length(model: str, base_url: str, api_key: str = "") -> Optional[int]:
"""Query a local server for the model's context length (short-TTL cached).
@@ -2504,13 +2520,18 @@ def _resolve_nous_context_length(
metadata = fetch_model_metadata()
def _safe_ctx(or_id: str, entry: dict) -> Optional[int]:
"""Return context length, but reject known stale 32K underreports.
Apply the same guard used for the generic OpenRouter path (step 6 in
resolve_context_length) so the Nous portal path does not short-circuit it.
"""
ctx = entry.get("context_length")
if ctx is None:
return None
if ctx <= 32768 and _model_name_suggests_kimi(or_id):
if ctx <= 32768 and _model_name_suggests_stale_32k_underreport(or_id):
logger.info(
"Rejecting OpenRouter metadata context=%s for %r "
"(Kimi-family underreport, Nous path); falling through to hardcoded defaults",
"(known 32K underreport, Nous path); falling through to hardcoded defaults",
ctx, or_id,
)
return None
@@ -2696,10 +2717,11 @@ def get_model_context_length(
model, base_url, cached,
)
_invalidate_cached_context_length(model, base_url)
# Invalidate stale 32k cache entries for Kimi-family models.
elif cached <= 32768 and _model_name_suggests_kimi(model):
# Invalidate stale 32k cache entries for model families known to
# be underreported by stale third-party metadata (Kimi, MiniMax).
elif cached <= 32768 and _model_name_suggests_stale_32k_underreport(model):
logger.info(
"Dropping stale Kimi cache entry %s@%s -> %s (OpenRouter underreport); "
"Dropping stale cached context entry %s@%s -> %s (known 32K underreport); "
"re-resolving via hardcoded defaults",
model, base_url, f"{cached:,}",
)
@@ -3017,11 +3039,12 @@ def get_model_context_length(
metadata = fetch_model_metadata()
if model in metadata:
or_ctx = metadata[model].get("context_length", DEFAULT_FALLBACK_CONTEXT)
# Guard against stale OpenRouter metadata for Kimi-family models.
if or_ctx == 32768 and _model_name_suggests_kimi(model):
# Guard against stale OpenRouter metadata for model families
# known to be underreported as 32K.
if or_ctx == 32768 and _model_name_suggests_stale_32k_underreport(model):
logger.info(
"Rejecting OpenRouter metadata context=%s for %r "
"(Kimi-family underreport); falling through to hardcoded defaults",
"(known 32K underreport); falling through to hardcoded defaults",
or_ctx, model,
)
else:
+57
View File
@@ -740,6 +740,63 @@ class TestGetModelContextLength:
result = get_model_context_length("custom/model")
assert result == CONTEXT_PROBE_TIERS[0]
@patch("agent.model_metadata.fetch_model_metadata")
@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
def test_stale_minimax_cache_32k_is_invalidated(self, mock_models_dev, mock_fetch, tmp_path):
"""Stale 32K cache entries for MiniMax must not keep tripping the 64K floor."""
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
base_url = "https://api.minimax.io/anthropic"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("MiniMax-M2.7", base_url, 32768)
result = get_model_context_length(
"MiniMax-M2.7",
base_url=base_url,
provider="minimax",
)
assert result == 204800
assert get_cached_context_length("MiniMax-M2.7", base_url) is None
@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_openrouter_32k_underreport_for_minimax_falls_through_to_default(self, mock_fetch, mock_models_dev):
"""Unknown-provider fallback must reject stale OpenRouter 32K for MiniMax."""
mock_fetch.return_value = {
"MiniMax-M2.7": {"context_length": 32768}
}
result = get_model_context_length("MiniMax-M2.7")
assert result == 204800
@patch("agent.model_metadata.fetch_model_metadata")
@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
def test_non_minimax_32k_cache_is_still_respected(self, mock_models_dev, mock_fetch, tmp_path):
"""The stale-32K invalidation must stay narrow and not touch unrelated models."""
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
base_url = "http://local"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("qwen3.5:27b", base_url, 32768)
result = get_model_context_length(
"qwen3.5:27b",
base_url=base_url,
)
assert result == 32768
@patch("agent.model_metadata.fetch_model_metadata")
@patch("agent.model_metadata.fetch_endpoint_model_metadata")
def test_custom_endpoint_metadata_beats_fuzzy_default(self, mock_endpoint_fetch, mock_fetch):
mock_fetch.return_value = {}
mock_endpoint_fetch.return_value = {
"zai-org/GLM-5-TEE": {"context_length": 65536}
}
result = get_model_context_length(
"zai-org/GLM-5-TEE",
base_url="https://llm.chutes.ai/v1",
api_key="test-key",
)
assert result == 65536
@patch("agent.model_metadata.fetch_model_metadata")
@patch("agent.model_metadata.fetch_endpoint_model_metadata")