From dcc2f3de1d3a11b21289d50343bc404cddc55635 Mon Sep 17 00:00:00 2001 From: adikpb <67222969+adikpb@users.noreply.github.com> Date: Fri, 31 Jul 2026 09:50:47 +0400 Subject: [PATCH] fix(vision): stop capping aux vision output with hardcoded max_tokens MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The vision tools' call_kwargs hardcode max_tokens caps (2000 for vision_analyze/browser_vision, 4000 for video analysis), truncating descriptions of complex images at the cap. The centralized aux client already omits max_tokens by default (#34845) so providers use their model max output; these three call sites were the leftovers that bypassed that policy. Remove the hardcoded caps entirely — the aux client handles the mandatory-max_tokens Anthropic wire via _resolve_anthropic_messages_max_tokens (model output ceiling) and Gemini native omits maxOutputTokens (65K ceiling), so no wire needs an explicit cap. --- tools/browser_tool.py | 1 - tools/vision_tools.py | 2 -- 2 files changed, 3 deletions(-) diff --git a/tools/browser_tool.py b/tools/browser_tool.py index b831efbec5..eaaff9d969 100644 --- a/tools/browser_tool.py +++ b/tools/browser_tool.py @@ -4692,7 +4692,6 @@ def browser_vision(question: str, annotate: bool = False, task_id: Optional[str] ], } ], - "max_tokens": 2000, "temperature": vision_temperature, "timeout": vision_timeout, } diff --git a/tools/vision_tools.py b/tools/vision_tools.py index e84daaff7f..e008983f7f 100644 --- a/tools/vision_tools.py +++ b/tools/vision_tools.py @@ -1501,7 +1501,6 @@ async def vision_analyze_tool( "task": "vision", "messages": messages, "temperature": vision_temperature, - "max_tokens": 2000, "timeout": vision_timeout, } if model: @@ -2073,7 +2072,6 @@ async def video_analyze_tool( "task": "vision", "messages": messages, "temperature": vision_temperature, - "max_tokens": 4000, "timeout": vision_timeout, } if model: