From 21a93f0a67088ebaf74ce15078286d71f1c6a770 Mon Sep 17 00:00:00 2001 From: HexLab98 Date: Sun, 23 Aug 2026 12:21:17 +0700 Subject: [PATCH] fix(vision): size native embeds for history reuse vision_analyze baked up to 4 MB / 7900px screenshots into immutable history, so every later turn re-sent ~400K chars. Cap embeds at 256 KB and 1568px (the long edge models actually read) so screenshot QA no longer blows the context. --- tests/tools/test_vision_native_fast_path.py | 22 ++++++---- tests/tools/test_vision_region.py | 2 +- tools/vision_tools.py | 45 +++++++++------------ 3 files changed, 35 insertions(+), 34 deletions(-) diff --git a/tests/tools/test_vision_native_fast_path.py b/tests/tools/test_vision_native_fast_path.py index dd6fac5bc3..715b3ee781 100644 --- a/tests/tools/test_vision_native_fast_path.py +++ b/tests/tools/test_vision_native_fast_path.py @@ -108,12 +108,10 @@ class TestVisionAnalyzeNative: """Regression for the wedged-session incident (May 2026). A vision tool-result image is baked into conversation history and - re-sent on every subsequent turn. Anthropic rejects any single - base64 image over 5 MB with a 400, and immutable history means the - bad bytes can't be cleared by retrying — the session is permanently - wedged. The native fast path must proactively resize down to the - embed cap (well under 5 MB) BEFORE embedding, not just at the 20 MB - hard ceiling. Skips if Pillow isn't available (resize is a no-op). + re-sent on every subsequent turn. The native fast path must + proactively resize down to the history-reuse embed cap BEFORE + embedding, not just at the 20 MB hard ceiling. Skips if Pillow + isn't available (resize is a no-op). """ pytest = __import__("pytest") try: @@ -138,10 +136,18 @@ class TestVisionAnalyzeNative: if p.get("type") == "image_url" ) assert len(url) <= _EMBED_TARGET_BYTES, ( - f"embedded image {len(url) / 1024 / 1024:.1f} MB exceeds embed cap " - f"{_EMBED_TARGET_BYTES / 1024 / 1024:.0f} MB — would wedge sessions on Anthropic" + f"embedded image {len(url) / 1024:.0f} KB exceeds embed cap " + f"{_EMBED_TARGET_BYTES / 1024:.0f} KB — would bloat every later turn" ) + def test_embed_caps_are_sized_for_history_reuse(self): + """Native embeds ride every later turn, so caps must stay well below + the Anthropic 5 MB / 8000px reject limits (#92699).""" + from tools.vision_tools import _EMBED_MAX_DIMENSION, _EMBED_TARGET_BYTES + + assert _EMBED_TARGET_BYTES <= 512 * 1024 + assert _EMBED_MAX_DIMENSION <= 2048 + # ─── _handle_vision_analyze fast-path gating ───────────────────────────────── diff --git a/tests/tools/test_vision_region.py b/tests/tools/test_vision_region.py index 3560458f5f..dcc7c5f255 100644 --- a/tests/tools/test_vision_region.py +++ b/tests/tools/test_vision_region.py @@ -142,7 +142,7 @@ class TestNativePathRegion: downscaled with the rest.""" from tools.vision_tools import _EMBED_MAX_DIMENSION, _vision_analyze_native - # Taller than the 7900px embed cap — full shot would be downscaled. + # Taller than the embed long-edge cap — full shot would be downscaled. big = tmp_path / "big.png" Image.new("RGB", (200, _EMBED_MAX_DIMENSION + 500), (0, 100, 0)).save( big, format="PNG" diff --git a/tools/vision_tools.py b/tools/vision_tools.py index e008983f7f..37898302ed 100644 --- a/tools/vision_tools.py +++ b/tools/vision_tools.py @@ -625,25 +625,21 @@ def _image_to_base64_data_url(image_path: Path, mime_type: Optional[str] = None) # provider accepts the image and we reject outright. _MAX_BASE64_BYTES = 20 * 1024 * 1024 -# Proactive embed cap (4 MB). This is the size we resize an image DOWN to -# before embedding it into conversation history, regardless of the 20 MB hard -# ceiling. Anthropic's per-image base64 limit is 5 MB; once an oversized image -# is baked into history (e.g. a vision tool-result), it is re-sent on every -# subsequent turn and permanently wedges the session with a 400 that retries -# can't clear (the bad bytes are immutable history). Capping at embed time — -# with headroom under 5 MB — is the only durable fix. Matches the post-failure -# shrink target in agent.conversation_compression so behaviour is consistent -# whether we resize proactively or reactively. -_EMBED_TARGET_BYTES = 4 * 1024 * 1024 +# Proactive embed cap for conversation-history reuse. Native vision_analyze +# bakes the data-URL into the tool result, which is re-sent on every later +# turn. The 20 MB hard ceiling / Anthropic 5 MB reject-cap still apply as +# safety nets; those are one-shot viewing limits, not history-reuse sizes. +# A 4 MB / 7900px embed was observed at ~400K chars and ~100–260K billed +# tokens per image (#92699), so we size for model reading instead: 256 KB +# is tens of KB after JPEG encode of a 1568px screenshot, well under every +# provider's per-image limit, and cheap enough to ride the session. +_EMBED_TARGET_BYTES = 256 * 1024 -# Proactive embed dimension cap (px, longest side). Anthropic enforces an -# 8000px per-side ceiling INDEPENDENTLY of the 5 MB byte cap — a tall full-page -# screenshot can be well under 5 MB yet far over 8000px (e.g. 1200×12000 at -# 0.06 MB), so the byte-only embed check above lets it slip into immutable -# history un-resized and the session bricks on a non-retryable 400. We cap at -# 7900 (headroom under 8000) so the proactive resize shrinks tall small-byte -# images before they are embedded. -_EMBED_MAX_DIMENSION = 7900 +# Proactive embed dimension cap (px, longest side). Anthropic still rejects +# above 8000px independently of the byte cap, but its tokenizer downsamples +# to a 1568px long edge — pixels past that cost wire bytes and never extra +# model fidelity. Cap at 1568 so history embeds match what the model sees. +_EMBED_MAX_DIMENSION = 1568 # Target size when auto-resizing on API failure (5 MB). After a provider # rejects an image, we downscale to this target and retry once. @@ -1240,13 +1236,12 @@ async def _vision_analyze_native( ) # Proactive embed cap: this image gets baked into conversation - # history and re-sent on every subsequent turn. Anthropic rejects - # any single base64 image over 5 MB OR over 8000px per side with a - # 400, and because history is immutable, an oversized embed - # permanently wedges the session — retries can't clear bytes (or - # pixels) that are already in the request. Resize DOWN to the embed - # target (4 MB / 7900px, headroom under both ceilings) whenever the - # payload exceeds either limit, not just at the 20 MB hard ceiling. + # history and re-sent on every subsequent turn. Resize DOWN to the + # history-reuse target whenever the payload exceeds either the byte + # or long-edge cap, not just at the 20 MB hard ceiling. Anthropic + # still rejects >5 MB / >8000px with a non-retryable 400, but those + # are one-shot viewing limits — history embeds are sized smaller so + # repeated vision_analyze turns don't blow the context (#92699). _over_bytes = len(image_data_url) > _EMBED_TARGET_BYTES _over_dims = await _run_encode_on_cpu_executor( _image_exceeds_dimension, temp_image_path, _EMBED_MAX_DIMENSION,