diff --git a/agent/model_metadata.py b/agent/model_metadata.py index dedb1ed9a0..6006c28b13 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -2082,6 +2082,22 @@ def _stale_pre_catalog_cache_entry(model: str, cached: int) -> bool: return cached <= threshold +def _model_name_suggests_minimax(model: str) -> bool: + """Return True if the model name looks like a MiniMax-family model. + + Catches ``MiniMax-M2.7``, ``minimax-m2.5``, ``MiniMaxAI/MiniMax-M2.5``, + and similar variants. Used as a guard against stale 32K metadata that + underreports the MiniMax M2 family, whose real context window is 204.8K. + """ + lower = model.lower() + return lower.startswith("minimax") or "minimaxai/" in lower + + +def _model_name_suggests_stale_32k_underreport(model: str) -> bool: + """Return True for model families known to be wrongly underreported as 32K.""" + return _model_name_suggests_kimi(model) or _model_name_suggests_minimax(model) + + def _query_local_context_length(model: str, base_url: str, api_key: str = "") -> Optional[int]: """Query a local server for the model's context length (short-TTL cached). @@ -2504,13 +2520,18 @@ def _resolve_nous_context_length( metadata = fetch_model_metadata() def _safe_ctx(or_id: str, entry: dict) -> Optional[int]: + """Return context length, but reject known stale 32K underreports. + + Apply the same guard used for the generic OpenRouter path (step 6 in + resolve_context_length) so the Nous portal path does not short-circuit it. + """ ctx = entry.get("context_length") if ctx is None: return None - if ctx <= 32768 and _model_name_suggests_kimi(or_id): + if ctx <= 32768 and _model_name_suggests_stale_32k_underreport(or_id): logger.info( "Rejecting OpenRouter metadata context=%s for %r " - "(Kimi-family underreport, Nous path); falling through to hardcoded defaults", + "(known 32K underreport, Nous path); falling through to hardcoded defaults", ctx, or_id, ) return None @@ -2696,10 +2717,11 @@ def get_model_context_length( model, base_url, cached, ) _invalidate_cached_context_length(model, base_url) - # Invalidate stale 32k cache entries for Kimi-family models. - elif cached <= 32768 and _model_name_suggests_kimi(model): + # Invalidate stale 32k cache entries for model families known to + # be underreported by stale third-party metadata (Kimi, MiniMax). + elif cached <= 32768 and _model_name_suggests_stale_32k_underreport(model): logger.info( - "Dropping stale Kimi cache entry %s@%s -> %s (OpenRouter underreport); " + "Dropping stale cached context entry %s@%s -> %s (known 32K underreport); " "re-resolving via hardcoded defaults", model, base_url, f"{cached:,}", ) @@ -3017,11 +3039,12 @@ def get_model_context_length( metadata = fetch_model_metadata() if model in metadata: or_ctx = metadata[model].get("context_length", DEFAULT_FALLBACK_CONTEXT) - # Guard against stale OpenRouter metadata for Kimi-family models. - if or_ctx == 32768 and _model_name_suggests_kimi(model): + # Guard against stale OpenRouter metadata for model families + # known to be underreported as 32K. + if or_ctx == 32768 and _model_name_suggests_stale_32k_underreport(model): logger.info( "Rejecting OpenRouter metadata context=%s for %r " - "(Kimi-family underreport); falling through to hardcoded defaults", + "(known 32K underreport); falling through to hardcoded defaults", or_ctx, model, ) else: diff --git a/tests/agent/test_model_metadata.py b/tests/agent/test_model_metadata.py index 726bd4e68f..94d281c696 100644 --- a/tests/agent/test_model_metadata.py +++ b/tests/agent/test_model_metadata.py @@ -740,6 +740,63 @@ class TestGetModelContextLength: result = get_model_context_length("custom/model") assert result == CONTEXT_PROBE_TIERS[0] + @patch("agent.model_metadata.fetch_model_metadata") + @patch("agent.models_dev.lookup_models_dev_context", return_value=None) + def test_stale_minimax_cache_32k_is_invalidated(self, mock_models_dev, mock_fetch, tmp_path): + """Stale 32K cache entries for MiniMax must not keep tripping the 64K floor.""" + mock_fetch.return_value = {} + cache_file = tmp_path / "cache.yaml" + base_url = "https://api.minimax.io/anthropic" + with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file): + save_context_length("MiniMax-M2.7", base_url, 32768) + result = get_model_context_length( + "MiniMax-M2.7", + base_url=base_url, + provider="minimax", + ) + assert result == 204800 + assert get_cached_context_length("MiniMax-M2.7", base_url) is None + + @patch("agent.models_dev.lookup_models_dev_context", return_value=None) + @patch("agent.model_metadata.fetch_model_metadata") + def test_openrouter_32k_underreport_for_minimax_falls_through_to_default(self, mock_fetch, mock_models_dev): + """Unknown-provider fallback must reject stale OpenRouter 32K for MiniMax.""" + mock_fetch.return_value = { + "MiniMax-M2.7": {"context_length": 32768} + } + result = get_model_context_length("MiniMax-M2.7") + assert result == 204800 + + @patch("agent.model_metadata.fetch_model_metadata") + @patch("agent.models_dev.lookup_models_dev_context", return_value=None) + def test_non_minimax_32k_cache_is_still_respected(self, mock_models_dev, mock_fetch, tmp_path): + """The stale-32K invalidation must stay narrow and not touch unrelated models.""" + mock_fetch.return_value = {} + cache_file = tmp_path / "cache.yaml" + base_url = "http://local" + with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file): + save_context_length("qwen3.5:27b", base_url, 32768) + result = get_model_context_length( + "qwen3.5:27b", + base_url=base_url, + ) + assert result == 32768 + + @patch("agent.model_metadata.fetch_model_metadata") + @patch("agent.model_metadata.fetch_endpoint_model_metadata") + def test_custom_endpoint_metadata_beats_fuzzy_default(self, mock_endpoint_fetch, mock_fetch): + mock_fetch.return_value = {} + mock_endpoint_fetch.return_value = { + "zai-org/GLM-5-TEE": {"context_length": 65536} + } + + result = get_model_context_length( + "zai-org/GLM-5-TEE", + base_url="https://llm.chutes.ai/v1", + api_key="test-key", + ) + + assert result == 65536 @patch("agent.model_metadata.fetch_model_metadata") @patch("agent.model_metadata.fetch_endpoint_model_metadata")