fix(context): prefer max_input_tokens over max_tokens for Anthropic proxies
Local /v1/models probes treated Anthropic `max_tokens` (max output) as the context window when `max_model_len`/`context_length` were absent. Anthropic and Anthropic-compatible reverse proxies expose both: max_input_tokens = context window (e.g. 1M for claude-fable-5) max_tokens = max output (e.g. 128k) That under-reported windows (1M → 128k), persisted the wrong value into context_length_cache.yaml, and fired compression at ~96k (75% of 128k). Route model objects through a shared helper that prefers input-window keys via _extract_context_length, and only falls back to max_tokens when no input-window field is present.
This commit is contained in:
+58
-20
@@ -1160,6 +1160,36 @@ def _extract_max_completion_tokens(payload: Dict[str, Any]) -> Optional[int]:
|
||||
return _extract_first_int(payload, _MAX_COMPLETION_KEYS)
|
||||
|
||||
|
||||
def _context_length_from_model_payload(payload: Dict[str, Any]) -> Optional[int]:
|
||||
"""Extract a context *window* from a ``/v1/models`` model object.
|
||||
|
||||
Prefers input-window keys (``max_model_len``, ``max_input_tokens``,
|
||||
``context_length``, …) via :func:`_extract_context_length`. Falls back to
|
||||
``max_tokens`` only when no input-window field is present.
|
||||
|
||||
Anthropic (and Anthropic-compatible proxies such as local reverse
|
||||
proxies) expose both ``max_input_tokens`` (context window, e.g. 1M) and
|
||||
``max_tokens`` (max *output* length, e.g. 128k). Using ``max_tokens`` as
|
||||
the context window under-reports the real limit, persists a stale value
|
||||
into ``context_length_cache.yaml``, and makes the compressor fire far too
|
||||
early (e.g. at 75% of 128k instead of 75% of 1M).
|
||||
"""
|
||||
if not isinstance(payload, dict):
|
||||
return None
|
||||
ctx = _extract_context_length(payload)
|
||||
if ctx is not None:
|
||||
return ctx
|
||||
# Last resort for OpenAI-compat servers that only report max_tokens as
|
||||
# the window. Safe for Anthropic shapes because max_input_tokens is
|
||||
# present and already handled above.
|
||||
raw = payload.get("max_tokens")
|
||||
if isinstance(raw, (int, float)):
|
||||
ivalue = int(raw)
|
||||
if ivalue > 0:
|
||||
return ivalue
|
||||
return None
|
||||
|
||||
|
||||
def _extract_pricing(payload: Dict[str, Any]) -> Dict[str, Any]:
|
||||
novita_input = payload.get("input_token_price_per_m")
|
||||
novita_output = payload.get("output_token_price_per_m")
|
||||
@@ -2305,14 +2335,17 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
|
||||
return int(ctx)
|
||||
break
|
||||
|
||||
# LM Studio / vLLM / llama.cpp: try /v1/models/{model}
|
||||
# LM Studio / vLLM / llama.cpp / Anthropic-compat proxies:
|
||||
# try /v1/models/{model}
|
||||
resp = client.get(f"{server_url}/v1/models/{model}")
|
||||
if resp.status_code == 200:
|
||||
data = resp.json()
|
||||
# vLLM returns max_model_len
|
||||
ctx = data.get("max_model_len") or data.get("context_length") or data.get("max_tokens")
|
||||
if ctx and isinstance(ctx, (int, float)):
|
||||
return int(ctx)
|
||||
if isinstance(data, dict):
|
||||
# Prefer max_model_len / max_input_tokens / context_length
|
||||
# over max_tokens (Anthropic max_tokens = max OUTPUT).
|
||||
ctx = _context_length_from_model_payload(data)
|
||||
if ctx is not None:
|
||||
return ctx
|
||||
|
||||
# Try /v1/models and find the model in the list.
|
||||
# Use _model_id_matches to handle "publisher/slug" vs bare "slug".
|
||||
@@ -2325,6 +2358,8 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
|
||||
# so fall back to the sole model when nothing matches.
|
||||
matched = None
|
||||
for m in models_list:
|
||||
if not isinstance(m, dict):
|
||||
continue
|
||||
if _model_id_matches(m.get("id", ""), model):
|
||||
matched = m
|
||||
break
|
||||
@@ -2335,21 +2370,24 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
|
||||
# vLLM/OpenAI keys are also checked. Runtime n_ctx is
|
||||
# preferred over n_ctx_train (the training maximum, which
|
||||
# can be larger than what the server actually allocates).
|
||||
for source in (matched, matched.get("meta") or {}):
|
||||
if not isinstance(source, dict):
|
||||
continue
|
||||
for key in (
|
||||
"n_ctx",
|
||||
"context_length",
|
||||
"context_window",
|
||||
"max_model_len",
|
||||
"max_context_length",
|
||||
"max_tokens",
|
||||
"n_ctx_train",
|
||||
):
|
||||
val = source.get(key)
|
||||
if isinstance(val, (int, float)) and val:
|
||||
return int(val)
|
||||
sources = [
|
||||
s
|
||||
for s in (matched, matched.get("meta") or {})
|
||||
if isinstance(s, dict)
|
||||
]
|
||||
for source in sources:
|
||||
val = source.get("n_ctx")
|
||||
if isinstance(val, (int, float)) and val:
|
||||
return int(val)
|
||||
# Canonical context-WINDOW keys (via _CONTEXT_LENGTH_KEYS)
|
||||
# with max_tokens demoted to an explicit last resort — see
|
||||
# _context_length_from_model_payload for why max_tokens
|
||||
# must never win over a real window key (it is the max
|
||||
# OUTPUT cap on Anthropic/OpenAI-compatible passthroughs).
|
||||
for source in sources:
|
||||
ctx = _context_length_from_model_payload(source)
|
||||
if ctx is not None:
|
||||
return ctx
|
||||
except Exception as exc:
|
||||
if _is_connect_timeout(exc):
|
||||
_note_endpoint_blackholed(server_url)
|
||||
|
||||
@@ -272,6 +272,126 @@ class TestQueryLocalContextLengthModelsList:
|
||||
assert result == 256000
|
||||
|
||||
|
||||
class TestContextLengthFromModelPayload:
|
||||
"""Anthropic / Anthropic-proxy model objects expose max_input_tokens
|
||||
(context window) and max_tokens (max OUTPUT). The local probe must not
|
||||
treat max_tokens as the context window."""
|
||||
|
||||
def test_prefers_max_input_tokens_over_max_tokens(self):
|
||||
from agent.model_metadata import _context_length_from_model_payload
|
||||
|
||||
# Real Anthropic /v1/models shape for claude-fable-5
|
||||
payload = {
|
||||
"type": "model",
|
||||
"id": "claude-fable-5",
|
||||
"max_input_tokens": 1_000_000,
|
||||
"max_tokens": 128_000, # output cap, NOT context
|
||||
}
|
||||
assert _context_length_from_model_payload(payload) == 1_000_000
|
||||
|
||||
def test_prefers_max_model_len_over_max_tokens(self):
|
||||
from agent.model_metadata import _context_length_from_model_payload
|
||||
|
||||
payload = {"id": "local-model", "max_model_len": 131072, "max_tokens": 4096}
|
||||
assert _context_length_from_model_payload(payload) == 131072
|
||||
|
||||
def test_falls_back_to_max_tokens_when_no_input_window_field(self):
|
||||
from agent.model_metadata import _context_length_from_model_payload
|
||||
|
||||
# Some OpenAI-compat servers only expose max_tokens for the window.
|
||||
payload = {"id": "odd-server", "max_tokens": 65536}
|
||||
assert _context_length_from_model_payload(payload) == 65536
|
||||
|
||||
def test_returns_none_for_empty_payload(self):
|
||||
from agent.model_metadata import _context_length_from_model_payload
|
||||
|
||||
assert _context_length_from_model_payload({}) is None
|
||||
assert _context_length_from_model_payload(None) is None # type: ignore[arg-type]
|
||||
|
||||
|
||||
class TestQueryLocalContextLengthAnthropicProxy:
|
||||
"""Local Anthropic-compatible reverse proxies (e.g. 127.0.0.1:47821)
|
||||
return Anthropic-shaped /v1/models entries. The probe must read
|
||||
max_input_tokens, not max_tokens."""
|
||||
|
||||
def _make_resp(self, status_code, body):
|
||||
resp = MagicMock()
|
||||
resp.status_code = status_code
|
||||
resp.json.return_value = body
|
||||
return resp
|
||||
|
||||
def test_models_list_prefers_max_input_tokens(self):
|
||||
from agent.model_metadata import _query_local_context_length
|
||||
|
||||
detail_resp = self._make_resp(404, {})
|
||||
list_resp = self._make_resp(200, {
|
||||
"data": [
|
||||
{
|
||||
"type": "model",
|
||||
"id": "claude-fable-5",
|
||||
"display_name": "Claude Fable 5",
|
||||
"max_input_tokens": 1_000_000,
|
||||
"max_tokens": 128_000,
|
||||
},
|
||||
{
|
||||
"type": "model",
|
||||
"id": "claude-haiku-4-5-20251001",
|
||||
"max_input_tokens": 200_000,
|
||||
"max_tokens": 64_000,
|
||||
},
|
||||
]
|
||||
})
|
||||
|
||||
call_count = [0]
|
||||
|
||||
def side_effect(url, **kwargs):
|
||||
call_count[0] += 1
|
||||
if call_count[0] == 1:
|
||||
return detail_resp # /v1/models/claude-fable-5
|
||||
return list_resp # /v1/models
|
||||
|
||||
client_mock = MagicMock()
|
||||
client_mock.__enter__ = lambda s: client_mock
|
||||
client_mock.__exit__ = MagicMock(return_value=False)
|
||||
client_mock.post.return_value = self._make_resp(404, {})
|
||||
client_mock.get.side_effect = side_effect
|
||||
|
||||
with patch("agent.model_metadata.detect_local_server_type", return_value=None), \
|
||||
patch("httpx.Client", return_value=client_mock):
|
||||
result = _query_local_context_length(
|
||||
"claude-fable-5", "http://127.0.0.1:47821"
|
||||
)
|
||||
|
||||
assert result == 1_000_000, (
|
||||
f"Expected max_input_tokens (1M), got {result}. "
|
||||
"If Hermes uses Anthropic max_tokens (128k), compression fires ~8x early."
|
||||
)
|
||||
|
||||
def test_model_detail_prefers_max_input_tokens(self):
|
||||
from agent.model_metadata import _query_local_context_length
|
||||
|
||||
detail_resp = self._make_resp(200, {
|
||||
"type": "model",
|
||||
"id": "claude-fable-5",
|
||||
"max_input_tokens": 1_000_000,
|
||||
"max_tokens": 128_000,
|
||||
})
|
||||
|
||||
client_mock = MagicMock()
|
||||
client_mock.__enter__ = lambda s: client_mock
|
||||
client_mock.__exit__ = MagicMock(return_value=False)
|
||||
client_mock.post.return_value = self._make_resp(404, {})
|
||||
client_mock.get.return_value = detail_resp
|
||||
|
||||
with patch("agent.model_metadata.detect_local_server_type", return_value=None), \
|
||||
patch("httpx.Client", return_value=client_mock):
|
||||
result = _query_local_context_length(
|
||||
"claude-fable-5", "http://127.0.0.1:47821/v1"
|
||||
)
|
||||
|
||||
assert result == 1_000_000
|
||||
|
||||
|
||||
class TestQueryLocalContextLengthLmStudio:
|
||||
"""_query_local_context_length with LM Studio native /api/v1/models response."""
|
||||
|
||||
|
||||
Reference in New Issue
Block a user