From 394f0f0902b6168df733a8c655eb59753b74e94e Mon Sep 17 00:00:00 2001 From: carlotestor Date: Thu, 16 Jul 2026 12:21:51 +0200 Subject: [PATCH] fix(context): prefer max_input_tokens over max_tokens for Anthropic proxies MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Local /v1/models probes treated Anthropic `max_tokens` (max output) as the context window when `max_model_len`/`context_length` were absent. Anthropic and Anthropic-compatible reverse proxies expose both: max_input_tokens = context window (e.g. 1M for claude-fable-5) max_tokens = max output (e.g. 128k) That under-reported windows (1M → 128k), persisted the wrong value into context_length_cache.yaml, and fired compression at ~96k (75% of 128k). Route model objects through a shared helper that prefers input-window keys via _extract_context_length, and only falls back to max_tokens when no input-window field is present. --- agent/model_metadata.py | 78 ++++++++---- tests/agent/test_model_metadata_local_ctx.py | 120 +++++++++++++++++++ 2 files changed, 178 insertions(+), 20 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 147f912070..73783ae66c 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1160,6 +1160,36 @@ def _extract_max_completion_tokens(payload: Dict[str, Any]) -> Optional[int]: return _extract_first_int(payload, _MAX_COMPLETION_KEYS) +def _context_length_from_model_payload(payload: Dict[str, Any]) -> Optional[int]: + """Extract a context *window* from a ``/v1/models`` model object. + + Prefers input-window keys (``max_model_len``, ``max_input_tokens``, + ``context_length``, …) via :func:`_extract_context_length`. Falls back to + ``max_tokens`` only when no input-window field is present. + + Anthropic (and Anthropic-compatible proxies such as local reverse + proxies) expose both ``max_input_tokens`` (context window, e.g. 1M) and + ``max_tokens`` (max *output* length, e.g. 128k). Using ``max_tokens`` as + the context window under-reports the real limit, persists a stale value + into ``context_length_cache.yaml``, and makes the compressor fire far too + early (e.g. at 75% of 128k instead of 75% of 1M). + """ + if not isinstance(payload, dict): + return None + ctx = _extract_context_length(payload) + if ctx is not None: + return ctx + # Last resort for OpenAI-compat servers that only report max_tokens as + # the window. Safe for Anthropic shapes because max_input_tokens is + # present and already handled above. + raw = payload.get("max_tokens") + if isinstance(raw, (int, float)): + ivalue = int(raw) + if ivalue > 0: + return ivalue + return None + + def _extract_pricing(payload: Dict[str, Any]) -> Dict[str, Any]: novita_input = payload.get("input_token_price_per_m") novita_output = payload.get("output_token_price_per_m") @@ -2305,14 +2335,17 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str return int(ctx) break - # LM Studio / vLLM / llama.cpp: try /v1/models/{model} + # LM Studio / vLLM / llama.cpp / Anthropic-compat proxies: + # try /v1/models/{model} resp = client.get(f"{server_url}/v1/models/{model}") if resp.status_code == 200: data = resp.json() - # vLLM returns max_model_len - ctx = data.get("max_model_len") or data.get("context_length") or data.get("max_tokens") - if ctx and isinstance(ctx, (int, float)): - return int(ctx) + if isinstance(data, dict): + # Prefer max_model_len / max_input_tokens / context_length + # over max_tokens (Anthropic max_tokens = max OUTPUT). + ctx = _context_length_from_model_payload(data) + if ctx is not None: + return ctx # Try /v1/models and find the model in the list. # Use _model_id_matches to handle "publisher/slug" vs bare "slug". @@ -2325,6 +2358,8 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str # so fall back to the sole model when nothing matches. matched = None for m in models_list: + if not isinstance(m, dict): + continue if _model_id_matches(m.get("id", ""), model): matched = m break @@ -2335,21 +2370,24 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str # vLLM/OpenAI keys are also checked. Runtime n_ctx is # preferred over n_ctx_train (the training maximum, which # can be larger than what the server actually allocates). - for source in (matched, matched.get("meta") or {}): - if not isinstance(source, dict): - continue - for key in ( - "n_ctx", - "context_length", - "context_window", - "max_model_len", - "max_context_length", - "max_tokens", - "n_ctx_train", - ): - val = source.get(key) - if isinstance(val, (int, float)) and val: - return int(val) + sources = [ + s + for s in (matched, matched.get("meta") or {}) + if isinstance(s, dict) + ] + for source in sources: + val = source.get("n_ctx") + if isinstance(val, (int, float)) and val: + return int(val) + # Canonical context-WINDOW keys (via _CONTEXT_LENGTH_KEYS) + # with max_tokens demoted to an explicit last resort — see + # _context_length_from_model_payload for why max_tokens + # must never win over a real window key (it is the max + # OUTPUT cap on Anthropic/OpenAI-compatible passthroughs). + for source in sources: + ctx = _context_length_from_model_payload(source) + if ctx is not None: + return ctx except Exception as exc: if _is_connect_timeout(exc): _note_endpoint_blackholed(server_url) diff --git a/tests/agent/test_model_metadata_local_ctx.py b/tests/agent/test_model_metadata_local_ctx.py index c428b5f857..c7b3b7e939 100644 --- a/tests/agent/test_model_metadata_local_ctx.py +++ b/tests/agent/test_model_metadata_local_ctx.py @@ -272,6 +272,126 @@ class TestQueryLocalContextLengthModelsList: assert result == 256000 +class TestContextLengthFromModelPayload: + """Anthropic / Anthropic-proxy model objects expose max_input_tokens + (context window) and max_tokens (max OUTPUT). The local probe must not + treat max_tokens as the context window.""" + + def test_prefers_max_input_tokens_over_max_tokens(self): + from agent.model_metadata import _context_length_from_model_payload + + # Real Anthropic /v1/models shape for claude-fable-5 + payload = { + "type": "model", + "id": "claude-fable-5", + "max_input_tokens": 1_000_000, + "max_tokens": 128_000, # output cap, NOT context + } + assert _context_length_from_model_payload(payload) == 1_000_000 + + def test_prefers_max_model_len_over_max_tokens(self): + from agent.model_metadata import _context_length_from_model_payload + + payload = {"id": "local-model", "max_model_len": 131072, "max_tokens": 4096} + assert _context_length_from_model_payload(payload) == 131072 + + def test_falls_back_to_max_tokens_when_no_input_window_field(self): + from agent.model_metadata import _context_length_from_model_payload + + # Some OpenAI-compat servers only expose max_tokens for the window. + payload = {"id": "odd-server", "max_tokens": 65536} + assert _context_length_from_model_payload(payload) == 65536 + + def test_returns_none_for_empty_payload(self): + from agent.model_metadata import _context_length_from_model_payload + + assert _context_length_from_model_payload({}) is None + assert _context_length_from_model_payload(None) is None # type: ignore[arg-type] + + +class TestQueryLocalContextLengthAnthropicProxy: + """Local Anthropic-compatible reverse proxies (e.g. 127.0.0.1:47821) + return Anthropic-shaped /v1/models entries. The probe must read + max_input_tokens, not max_tokens.""" + + def _make_resp(self, status_code, body): + resp = MagicMock() + resp.status_code = status_code + resp.json.return_value = body + return resp + + def test_models_list_prefers_max_input_tokens(self): + from agent.model_metadata import _query_local_context_length + + detail_resp = self._make_resp(404, {}) + list_resp = self._make_resp(200, { + "data": [ + { + "type": "model", + "id": "claude-fable-5", + "display_name": "Claude Fable 5", + "max_input_tokens": 1_000_000, + "max_tokens": 128_000, + }, + { + "type": "model", + "id": "claude-haiku-4-5-20251001", + "max_input_tokens": 200_000, + "max_tokens": 64_000, + }, + ] + }) + + call_count = [0] + + def side_effect(url, **kwargs): + call_count[0] += 1 + if call_count[0] == 1: + return detail_resp # /v1/models/claude-fable-5 + return list_resp # /v1/models + + client_mock = MagicMock() + client_mock.__enter__ = lambda s: client_mock + client_mock.__exit__ = MagicMock(return_value=False) + client_mock.post.return_value = self._make_resp(404, {}) + client_mock.get.side_effect = side_effect + + with patch("agent.model_metadata.detect_local_server_type", return_value=None), \ + patch("httpx.Client", return_value=client_mock): + result = _query_local_context_length( + "claude-fable-5", "http://127.0.0.1:47821" + ) + + assert result == 1_000_000, ( + f"Expected max_input_tokens (1M), got {result}. " + "If Hermes uses Anthropic max_tokens (128k), compression fires ~8x early." + ) + + def test_model_detail_prefers_max_input_tokens(self): + from agent.model_metadata import _query_local_context_length + + detail_resp = self._make_resp(200, { + "type": "model", + "id": "claude-fable-5", + "max_input_tokens": 1_000_000, + "max_tokens": 128_000, + }) + + client_mock = MagicMock() + client_mock.__enter__ = lambda s: client_mock + client_mock.__exit__ = MagicMock(return_value=False) + client_mock.post.return_value = self._make_resp(404, {}) + client_mock.get.return_value = detail_resp + + with patch("agent.model_metadata.detect_local_server_type", return_value=None), \ + patch("httpx.Client", return_value=client_mock): + result = _query_local_context_length( + "claude-fable-5", "http://127.0.0.1:47821/v1" + ) + + assert result == 1_000_000 + + class TestQueryLocalContextLengthLmStudio: """_query_local_context_length with LM Studio native /api/v1/models response."""