fix(vision): size native embeds for history reuse

vision_analyze baked up to 4 MB / 7900px screenshots into immutable
history, so every later turn re-sent ~400K chars. Cap embeds at 256 KB
and 1568px (the long edge models actually read) so screenshot QA no
longer blows the context.
This commit is contained in:
HexLab98
2026-08-23 12:21:17 +07:00
committed by kshitij
parent 3c44cd0c67
commit 21a93f0a67
3 changed files with 35 additions and 34 deletions
+14 -8
View File
@@ -108,12 +108,10 @@ class TestVisionAnalyzeNative:
"""Regression for the wedged-session incident (May 2026).
A vision tool-result image is baked into conversation history and
re-sent on every subsequent turn. Anthropic rejects any single
base64 image over 5 MB with a 400, and immutable history means the
bad bytes can't be cleared by retrying — the session is permanently
wedged. The native fast path must proactively resize down to the
embed cap (well under 5 MB) BEFORE embedding, not just at the 20 MB
hard ceiling. Skips if Pillow isn't available (resize is a no-op).
re-sent on every subsequent turn. The native fast path must
proactively resize down to the history-reuse embed cap BEFORE
embedding, not just at the 20 MB hard ceiling. Skips if Pillow
isn't available (resize is a no-op).
"""
pytest = __import__("pytest")
try:
@@ -138,10 +136,18 @@ class TestVisionAnalyzeNative:
if p.get("type") == "image_url"
)
assert len(url) <= _EMBED_TARGET_BYTES, (
f"embedded image {len(url) / 1024 / 1024:.1f} MB exceeds embed cap "
f"{_EMBED_TARGET_BYTES / 1024 / 1024:.0f} MB — would wedge sessions on Anthropic"
f"embedded image {len(url) / 1024:.0f} KB exceeds embed cap "
f"{_EMBED_TARGET_BYTES / 1024:.0f} KB — would bloat every later turn"
)
def test_embed_caps_are_sized_for_history_reuse(self):
"""Native embeds ride every later turn, so caps must stay well below
the Anthropic 5 MB / 8000px reject limits (#92699)."""
from tools.vision_tools import _EMBED_MAX_DIMENSION, _EMBED_TARGET_BYTES
assert _EMBED_TARGET_BYTES <= 512 * 1024
assert _EMBED_MAX_DIMENSION <= 2048
# ─── _handle_vision_analyze fast-path gating ─────────────────────────────────
+1 -1
View File
@@ -142,7 +142,7 @@ class TestNativePathRegion:
downscaled with the rest."""
from tools.vision_tools import _EMBED_MAX_DIMENSION, _vision_analyze_native
# Taller than the 7900px embed cap — full shot would be downscaled.
# Taller than the embed long-edge cap — full shot would be downscaled.
big = tmp_path / "big.png"
Image.new("RGB", (200, _EMBED_MAX_DIMENSION + 500), (0, 100, 0)).save(
big, format="PNG"
+20 -25
View File
@@ -625,25 +625,21 @@ def _image_to_base64_data_url(image_path: Path, mime_type: Optional[str] = None)
# provider accepts the image and we reject outright.
_MAX_BASE64_BYTES = 20 * 1024 * 1024
# Proactive embed cap (4 MB). This is the size we resize an image DOWN to
# before embedding it into conversation history, regardless of the 20 MB hard
# ceiling. Anthropic's per-image base64 limit is 5 MB; once an oversized image
# is baked into history (e.g. a vision tool-result), it is re-sent on every
# subsequent turn and permanently wedges the session with a 400 that retries
# can't clear (the bad bytes are immutable history). Capping at embed time —
# with headroom under 5 MB — is the only durable fix. Matches the post-failure
# shrink target in agent.conversation_compression so behaviour is consistent
# whether we resize proactively or reactively.
_EMBED_TARGET_BYTES = 4 * 1024 * 1024
# Proactive embed cap for conversation-history reuse. Native vision_analyze
# bakes the data-URL into the tool result, which is re-sent on every later
# turn. The 20 MB hard ceiling / Anthropic 5 MB reject-cap still apply as
# safety nets; those are one-shot viewing limits, not history-reuse sizes.
# A 4 MB / 7900px embed was observed at ~400K chars and ~100–260K billed
# tokens per image (#92699), so we size for model reading instead: 256 KB
# is tens of KB after JPEG encode of a 1568px screenshot, well under every
# provider's per-image limit, and cheap enough to ride the session.
_EMBED_TARGET_BYTES = 256 * 1024
# Proactive embed dimension cap (px, longest side). Anthropic enforces an
# 8000px per-side ceiling INDEPENDENTLY of the 5 MB byte cap — a tall full-page
# screenshot can be well under 5 MB yet far over 8000px (e.g. 1200×12000 at
# 0.06 MB), so the byte-only embed check above lets it slip into immutable
# history un-resized and the session bricks on a non-retryable 400. We cap at
# 7900 (headroom under 8000) so the proactive resize shrinks tall small-byte
# images before they are embedded.
_EMBED_MAX_DIMENSION = 7900
# Proactive embed dimension cap (px, longest side). Anthropic still rejects
# above 8000px independently of the byte cap, but its tokenizer downsamples
# to a 1568px long edge — pixels past that cost wire bytes and never extra
# model fidelity. Cap at 1568 so history embeds match what the model sees.
_EMBED_MAX_DIMENSION = 1568
# Target size when auto-resizing on API failure (5 MB). After a provider
# rejects an image, we downscale to this target and retry once.
@@ -1240,13 +1236,12 @@ async def _vision_analyze_native(
)
# Proactive embed cap: this image gets baked into conversation
# history and re-sent on every subsequent turn. Anthropic rejects
# any single base64 image over 5 MB OR over 8000px per side with a
# 400, and because history is immutable, an oversized embed
# permanently wedges the session — retries can't clear bytes (or
# pixels) that are already in the request. Resize DOWN to the embed
# target (4 MB / 7900px, headroom under both ceilings) whenever the
# payload exceeds either limit, not just at the 20 MB hard ceiling.
# history and re-sent on every subsequent turn. Resize DOWN to the
# history-reuse target whenever the payload exceeds either the byte
# or long-edge cap, not just at the 20 MB hard ceiling. Anthropic
# still rejects >5 MB / >8000px with a non-retryable 400, but those
# are one-shot viewing limits — history embeds are sized smaller so
# repeated vision_analyze turns don't blow the context (#92699).
_over_bytes = len(image_data_url) > _EMBED_TARGET_BYTES
_over_dims = await _run_encode_on_cpu_executor(
_image_exceeds_dimension, temp_image_path, _EMBED_MAX_DIMENSION,