diff --git a/tools/annotate_preview_tool.py b/tools/annotate_preview_tool.py index 1b709d0134..a066b343a4 100644 --- a/tools/annotate_preview_tool.py +++ b/tools/annotate_preview_tool.py @@ -1,28 +1,13 @@ #!/usr/bin/env python3 -"""Leave a mark on the page in the Hermes desktop GUI's in-app browser. +"""Persistent element annotations in the Hermes desktop GUI's in-app browser. -``drive_preview`` already draws every move it makes — the field it can reach, -a box round its target, the cursor going there — but those are transients: -each one stands for a single action and retires itself. That is right for -narrating a click and no use at all for holding a finding on screen. - -This is the deliberate one. An annotation outlines an element — or, with -``hold``, the entire visible field at once — and stays until the agent takes it -down, so it can show the user what it found, flag the fields -it is about to fill, or keep its place while it works elsewhere on the page. -Named for TouchDesigner's Annotate — the labelled box you drop around part of a -network to call it out. - -Annotations are bound to elements, not coordinates: they ride scrolls and -reflows, and they go when their element does, so a navigation clears them -without the agent having to. - -Rides the same ``preview.act`` bridge as ``drive_preview`` rather than opening -a second channel — the renderer already resolves ``@e`` refs and owns the -overlay, so this is one more verb on a wire that exists. - -Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for -desktop-sourced sessions. +``drive_preview`` draws transient marks (one per action, self-retiring). An +annotation outlines an element — or, with ``hold``, the whole visible field — +and stays until the agent removes it. Annotations bind to elements, not +coordinates: they ride scrolls/reflows and vanish with their element, so a +navigation clears them. Rides the same ``preview.act`` bridge as +``drive_preview`` (the renderer resolves ``@e`` refs and owns the overlay). +Lives in the ``desktop_ui`` toolset, enabled only for desktop-sourced sessions. """ import json diff --git a/tools/image_generation_catalog.py b/tools/image_generation_catalog.py index 8cd93af163..d8e3c3660e 100644 --- a/tools/image_generation_catalog.py +++ b/tools/image_generation_catalog.py @@ -11,7 +11,7 @@ faces when default-on, so upscaling is strictly per-call opt-in. Pricing strings are as-of-commit and allowed to drift. """ -from typing import Any, Dict +from typing import Any, Dict, Optional _PRESET_SIZES = { "landscape": "landscape_16_9", @@ -19,533 +19,330 @@ _PRESET_SIZES = { "portrait": "portrait_16_9", } _ASPECT_SIZES = {"landscape": "16:9", "square": "1:1", "portrait": "9:16"} +_DEFAULT_SIZES = {"image_size_preset": _PRESET_SIZES, "aspect_ratio": _ASPECT_SIZES} + + +def _model( + display: str, + speed: str, + strengths: str, + price: str, + *, + style: str = "image_size_preset", + sizes: Optional[Dict[str, Any]] = None, + defaults: Dict[str, Any], + supports: set, + edit_endpoint: Optional[str] = None, + edit_supports: Optional[set] = None, + max_reference_images: Optional[int] = None, +) -> Dict[str, Any]: + """Build one catalog entry; edit keys are present only for edit-capable models.""" + entry: Dict[str, Any] = { + "display": display, + "speed": speed, + "strengths": strengths, + "price": price, + "size_style": style, + "sizes": sizes if sizes is not None else _DEFAULT_SIZES[style], + "defaults": defaults, + "supports": supports, + "upscale": False, + } + if edit_endpoint: + entry["edit_endpoint"] = edit_endpoint + entry["edit_supports"] = edit_supports + entry["max_reference_images"] = max_reference_images + return entry + FAL_MODELS: Dict[str, Dict[str, Any]] = { - "fal-ai/flux-2/klein/9b": { - "display": "FLUX 2 Klein 9B", - "speed": "<1s", - "strengths": "Fast, crisp text", - "price": "$0.006/MP", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "num_inference_steps": 4, - "output_format": "png", - "enable_safety_checker": False, + "fal-ai/flux-2/klein/9b": _model( + "FLUX 2 Klein 9B", "<1s", "Fast, crisp text", "$0.006/MP", + defaults={ + "num_inference_steps": 4, "output_format": "png", "enable_safety_checker": False, }, - "supports": { - "prompt", "image_size", "num_inference_steps", "seed", - "output_format", "enable_safety_checker", + supports={ + "prompt", "image_size", "num_inference_steps", "seed", "output_format", "enable_safety_checker", }, - "upscale": False, - # Image-to-image / editing: FLUX.2 [klein] 9B edit endpoint takes - # `image_urls` (list). Natural-language edits, multi-ref. - "edit_endpoint": "fal-ai/flux-2/klein/9b/edit", - "edit_supports": { - "prompt", "image_urls", "num_inference_steps", "seed", - "output_format", "enable_safety_checker", + edit_endpoint="fal-ai/flux-2/klein/9b/edit", + edit_supports={ + "prompt", "image_urls", "num_inference_steps", "seed", "output_format", "enable_safety_checker", }, - "max_reference_images": 9, - }, - "fal-ai/flux-2-pro": { - "display": "FLUX 2 Pro", - "speed": "~6s", - "strengths": "Studio photorealism", - "price": "$0.03/MP", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "num_inference_steps": 50, - "guidance_scale": 4.5, - "num_images": 1, - "output_format": "png", - "enable_safety_checker": False, - "safety_tolerance": "5", + max_reference_images=9, + ), + "fal-ai/flux-2-pro": _model( + "FLUX 2 Pro", "~6s", "Studio photorealism", "$0.03/MP", + defaults={ + "num_inference_steps": 50, "guidance_scale": 4.5, "num_images": 1, + "output_format": "png", "enable_safety_checker": False, "safety_tolerance": "5", "sync_mode": True, }, - "supports": { - "prompt", "image_size", "num_inference_steps", "guidance_scale", - "num_images", "output_format", "enable_safety_checker", - "safety_tolerance", "sync_mode", "seed", + supports={ + "prompt", "image_size", "num_inference_steps", "guidance_scale", "num_images", "output_format", + "enable_safety_checker", "safety_tolerance", "sync_mode", "seed", }, - "upscale": False, - # Edit endpoint accepts up to 9 reference images. - "edit_endpoint": "fal-ai/flux-2-pro/edit", - "edit_supports": { - "prompt", "image_urls", "num_inference_steps", "guidance_scale", - "num_images", "output_format", "enable_safety_checker", - "safety_tolerance", "sync_mode", "seed", + edit_endpoint="fal-ai/flux-2-pro/edit", + edit_supports={ + "prompt", "image_urls", "num_inference_steps", "guidance_scale", "num_images", "output_format", + "enable_safety_checker", "safety_tolerance", "sync_mode", "seed", }, - "max_reference_images": 9, - }, - "fal-ai/z-image/turbo": { - "display": "Z-Image Turbo", - "speed": "~2s", - "strengths": "Bilingual EN/CN, 6B", - "price": "$0.005/MP", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "num_inference_steps": 8, - "num_images": 1, - "output_format": "png", - "enable_safety_checker": False, - "enable_prompt_expansion": False, # avoid the extra per-request charge + max_reference_images=9, + ), + "fal-ai/z-image/turbo": _model( + "Z-Image Turbo", "~2s", "Bilingual EN/CN, 6B", "$0.005/MP", + defaults={ # prompt expansion off: avoids the extra per-request charge + "num_inference_steps": 8, "num_images": 1, "output_format": "png", + "enable_safety_checker": False, "enable_prompt_expansion": False, }, - "supports": { - "prompt", "image_size", "num_inference_steps", "num_images", - "seed", "output_format", "enable_safety_checker", - "enable_prompt_expansion", + supports={ + "prompt", "image_size", "num_inference_steps", "num_images", "seed", "output_format", + "enable_safety_checker", "enable_prompt_expansion", }, - "upscale": False, - }, - "fal-ai/nano-banana-pro": { - "display": "Nano Banana Pro (Gemini 3 Pro Image)", - "speed": "~8s", - "strengths": "Gemini 3 Pro, reasoning depth, text rendering", - "price": "$0.15/image (1K)", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "num_images": 1, - "output_format": "png", - "safety_tolerance": "5", - # "1K" is the cheapest tier; 4K doubles the per-image cost. - # Users on Nous Subscription should stay at 1K for predictable billing. + ), + "fal-ai/nano-banana-pro": _model( + "Nano Banana Pro (Gemini 3 Pro Image)", "~8s", "Gemini 3 Pro, reasoning depth, text rendering", "$0.15/image (1K)", + style="aspect_ratio", + # "1K" is the cheapest tier; 4K doubles the per-image cost (Nous Subscription billing). + defaults={ + "num_images": 1, "output_format": "png", "safety_tolerance": "5", "resolution": "1K", }, - "supports": { - "prompt", "aspect_ratio", "num_images", "output_format", - "safety_tolerance", "seed", "sync_mode", "resolution", - "enable_web_search", "limit_generations", - }, - "upscale": False, - # Nano Banana Pro edit (Gemini 3 Pro Image): natural-language edits - # with up to 2 reference images via `image_urls`. - "edit_endpoint": "fal-ai/nano-banana-pro/edit", - "edit_supports": { - "prompt", "image_urls", "aspect_ratio", "num_images", - "output_format", "safety_tolerance", "seed", "sync_mode", + supports={ + "prompt", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", "sync_mode", "resolution", "enable_web_search", "limit_generations", }, - "max_reference_images": 2, - }, - "fal-ai/nano-banana-2": { - "display": "Nano Banana 2 (Gemini 3.1 Flash Image)", - "speed": "~3s", - "strengths": "Fast reasoning, multilingual text, infographics", - "price": "Lower-cost Flash tier", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "num_images": 1, - "output_format": "png", - "safety_tolerance": "4", - "resolution": "1K", - "limit_generations": True, + edit_endpoint="fal-ai/nano-banana-pro/edit", + edit_supports={ + "prompt", "image_urls", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", + "sync_mode", "resolution", "enable_web_search", "limit_generations", }, - "supports": { - "prompt", "aspect_ratio", "num_images", "output_format", - "safety_tolerance", "seed", "sync_mode", "system_prompt", - "resolution", "enable_web_search", "limit_generations", + max_reference_images=2, + ), + "fal-ai/nano-banana-2": _model( + "Nano Banana 2 (Gemini 3.1 Flash Image)", "~3s", "Fast reasoning, multilingual text, infographics", "Lower-cost Flash tier", + style="aspect_ratio", + defaults={ + "num_images": 1, "output_format": "png", "safety_tolerance": "4", + "resolution": "1K", "limit_generations": True, + }, + supports={ + "prompt", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", "sync_mode", + "system_prompt", "resolution", "enable_web_search", "limit_generations", "thinking_level", + }, + edit_endpoint="fal-ai/nano-banana-2/edit", + edit_supports={ + "prompt", "image_urls", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", + "sync_mode", "system_prompt", "resolution", "enable_web_search", "limit_generations", "thinking_level", }, - "upscale": False, - "edit_endpoint": "fal-ai/nano-banana-2/edit", - "edit_supports": { - "prompt", "image_urls", "aspect_ratio", "num_images", - "output_format", "safety_tolerance", "seed", "sync_mode", - "system_prompt", "resolution", "enable_web_search", - "limit_generations", "thinking_level", + max_reference_images=14, + ), + "fal-ai/gpt-image-1.5": _model( + "GPT Image 1.5", "~15s", "Prompt adherence", "$0.034/image", + style="gpt_literal", sizes={ + "landscape": "1536x1024", "square": "1024x1024", "portrait": "1024x1536", }, - "max_reference_images": 14, - }, - "fal-ai/gpt-image-1.5": { - "display": "GPT Image 1.5", - "speed": "~15s", - "strengths": "Prompt adherence", - "price": "$0.034/image", - "size_style": "gpt_literal", - "sizes": { - "landscape": "1536x1024", - "square": "1024x1024", - "portrait": "1024x1536", + # quality pinned to medium (also for gpt-image-2) so portal billing stays + # predictable: low is too rough, high is 3-6x the per-image cost. + defaults={"quality": "medium", "num_images": 1, "output_format": "png"}, + supports={ + "prompt", "image_size", "quality", "num_images", "output_format", "background", "sync_mode", }, - "defaults": { - # Quality is pinned to medium to keep portal billing predictable - # across all users (low is too rough, high is 4-6x more expensive). - "quality": "medium", - "num_images": 1, - "output_format": "png", + edit_endpoint="fal-ai/gpt-image-1.5/edit", + edit_supports={ + "prompt", "image_urls", "image_size", "quality", "num_images", "output_format", "sync_mode", }, - "supports": { - "prompt", "image_size", "quality", "num_images", "output_format", - "background", "sync_mode", + max_reference_images=16, + ), + # GPT Image 2 uses FAL's preset enum (unlike 1.5's literal dims) mapped to the + # 4:3 variants: the 16:9 presets (1024x576) fall below its 655,360 min-pixel + # requirement. openai_api_key (BYOK) is deliberately not in `supports` — all + # users go through the shared FAL billing path. Its edit endpoint lives under + # the OpenAI namespace (NOT fal-ai/) and auto-infers size, so no image_size. + "fal-ai/gpt-image-2": _model( + "GPT Image 2", "~20s", "SOTA text rendering + CJK, world-aware photorealism", "$0.04–0.06/image", + style="image_size_preset", sizes={ + "landscape": "landscape_4_3", "square": "square_hd", "portrait": "portrait_4_3", }, - "upscale": False, - # Edit endpoint: high-fidelity edits preserving composition/lighting. - "edit_endpoint": "fal-ai/gpt-image-1.5/edit", - "edit_supports": { - "prompt", "image_urls", "image_size", "quality", "num_images", - "output_format", "sync_mode", + defaults={"quality": "medium", "num_images": 1, "output_format": "png"}, + supports={ + "prompt", "image_size", "quality", "num_images", "output_format", "sync_mode", }, - "max_reference_images": 16, - }, - "fal-ai/gpt-image-2": { - "display": "GPT Image 2", - "speed": "~20s", - "strengths": "SOTA text rendering + CJK, world-aware photorealism", - "price": "$0.04–0.06/image", - # GPT Image 2 uses FAL's standard preset enum (unlike 1.5's literal - # dimensions). We map to the 4:3 variants — the 16:9 presets - # (1024x576) fall below GPT-Image-2's 655,360 min-pixel requirement - # and would be rejected. 4:3 keeps us above the minimum on all - # three aspect ratios. - "size_style": "image_size_preset", - "sizes": { - "landscape": "landscape_4_3", # 1024x768 - "square": "square_hd", # 1024x1024 - "portrait": "portrait_4_3", # 768x1024 + edit_endpoint="openai/gpt-image-2/edit", + edit_supports={ + "prompt", "image_urls", "quality", "num_images", "output_format", "sync_mode", "mask_image_url", }, - "defaults": { - # Same quality pinning as gpt-image-1.5: medium keeps Nous - # Portal billing predictable. "high" is 3-4x the per-image - # cost at the same size; "low" is too rough for production use. - "quality": "medium", - "num_images": 1, - "output_format": "png", + max_reference_images=16, + ), + "fal-ai/ideogram/v3": _model( + "Ideogram V3", "~5s", "Best typography", "$0.03-0.09/image", + defaults={"rendering_speed": "BALANCED", "expand_prompt": True, "style": "AUTO"}, + supports={ + "prompt", "image_size", "rendering_speed", "expand_prompt", "style", "seed", }, - "supports": { - "prompt", "image_size", "quality", "num_images", "output_format", - "sync_mode", - # openai_api_key (BYOK) intentionally omitted — all users go - # through the shared FAL billing path. + edit_endpoint="fal-ai/ideogram/v3/edit", + edit_supports={ + "prompt", "image_urls", "rendering_speed", "expand_prompt", "style", "seed", }, - "upscale": False, - # GPT Image 2 edit endpoint lives under the OpenAI namespace on FAL - # (NOT fal-ai/). Takes `image_urls` (list) + optional mask. We don't - # send `image_size` on edit so the model auto-infers from input. - "edit_endpoint": "openai/gpt-image-2/edit", - "edit_supports": { - "prompt", "image_urls", "quality", "num_images", "output_format", - "sync_mode", "mask_image_url", + max_reference_images=1, + ), + "fal-ai/recraft/v4/pro/text-to-image": _model( + "Recraft V4 Pro", "~8s", "Design, brand systems, production-ready", "$0.25/image", + defaults={"enable_safety_checker": False}, # V4 Pro dropped V3's required `style` enum + supports={ + "prompt", "image_size", "enable_safety_checker", "colors", "background_color", }, - "max_reference_images": 16, - }, - "fal-ai/ideogram/v3": { - "display": "Ideogram V3", - "speed": "~5s", - "strengths": "Best typography", - "price": "$0.03-0.09/image", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "rendering_speed": "BALANCED", - "expand_prompt": True, - "style": "AUTO", + ), + "fal-ai/qwen-image": _model( + "Qwen Image", "~12s", "LLM-based, complex text", "$0.02/MP", + defaults={ + "num_inference_steps": 30, "guidance_scale": 2.5, "num_images": 1, + "output_format": "png", "acceleration": "regular", }, - "supports": { - "prompt", "image_size", "rendering_speed", "expand_prompt", - "style", "seed", + supports={ + "prompt", "image_size", "num_inference_steps", "guidance_scale", "num_images", "output_format", + "acceleration", "seed", "sync_mode", }, - "upscale": False, - # Ideogram V3 edit endpoint takes `image_urls` (list). - "edit_endpoint": "fal-ai/ideogram/v3/edit", - "edit_supports": { - "prompt", "image_urls", "rendering_speed", "expand_prompt", - "style", "seed", + edit_endpoint="fal-ai/qwen-image-2/pro/edit", + edit_supports={ + "prompt", "image_urls", "num_inference_steps", "guidance_scale", "num_images", "output_format", + "acceleration", "seed", "sync_mode", }, - "max_reference_images": 1, - }, - "fal-ai/recraft/v4/pro/text-to-image": { - "display": "Recraft V4 Pro", - "speed": "~8s", - "strengths": "Design, brand systems, production-ready", - "price": "$0.25/image", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - # V4 Pro dropped V3's required `style` enum — defaults handle taste now. - "enable_safety_checker": False, + max_reference_images=3, + ), + # Krea 2 on FAL — same family as ``plugins/image_gen/krea`` but billed through + # FAL / the FAL managed gateway. Native ``krea-2-*`` ids route to the plugin. + "fal-ai/krea/v2/medium/text-to-image": _model( + "Krea 2 Medium", "~15-25s", "Illustration, anime, painting, expressive/artistic styles", "$0.030 (text) / $0.035 (style refs)", + style="aspect_ratio", + defaults={"creativity": "medium"}, + supports={ + "prompt", "aspect_ratio", "creativity", "seed", "image_style_references", }, - "supports": { - "prompt", "image_size", "enable_safety_checker", - "colors", "background_color", + ), + "fal-ai/krea/v2/large/text-to-image": _model( + "Krea 2 Large", "~25-60s", "Photorealism, raw textured looks (motion blur, grain, film)", "$0.060 (text) / $0.065 (style refs)", + style="aspect_ratio", + defaults={"creativity": "medium"}, + supports={ + "prompt", "aspect_ratio", "creativity", "seed", "image_style_references", }, - "upscale": False, - }, - "fal-ai/qwen-image": { - "display": "Qwen Image", - "speed": "~12s", - "strengths": "LLM-based, complex text", - "price": "$0.02/MP", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "num_inference_steps": 30, - "guidance_scale": 2.5, - "num_images": 1, - "output_format": "png", - "acceleration": "regular", - }, - "supports": { - "prompt", "image_size", "num_inference_steps", "guidance_scale", - "num_images", "output_format", "acceleration", "seed", "sync_mode", - }, - "upscale": False, - # Qwen edit uses the Qwen Image 2.0 Pro editing endpoint, which takes - # `image_urls` (list) + natural-language edit instructions. - "edit_endpoint": "fal-ai/qwen-image-2/pro/edit", - "edit_supports": { - "prompt", "image_urls", "num_inference_steps", "guidance_scale", - "num_images", "output_format", "acceleration", "seed", "sync_mode", - }, - "max_reference_images": 3, - }, - # Krea 2 on FAL — same model family as ``plugins/image_gen/krea``, but billed - # through FAL / the FAL managed gateway. Native ``krea-2-*`` ids route to the - # dedicated Krea plugin instead. - "fal-ai/krea/v2/medium/text-to-image": { - "display": "Krea 2 Medium", - "speed": "~15-25s", - "strengths": "Illustration, anime, painting, expressive/artistic styles", - "price": "$0.030 (text) / $0.035 (style refs)", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "creativity": "medium", - }, - "supports": { - "prompt", "aspect_ratio", "creativity", "seed", - "image_style_references", - }, - "upscale": False, - }, - "fal-ai/krea/v2/large/text-to-image": { - "display": "Krea 2 Large", - "speed": "~25-60s", - "strengths": "Photorealism, raw textured looks (motion blur, grain, film)", - "price": "$0.060 (text) / $0.065 (style refs)", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "creativity": "medium", - }, - "supports": { - "prompt", "aspect_ratio", "creativity", "seed", - "image_style_references", - }, - "upscale": False, - }, - # ─── Aug 2026 catalog expansion ──────────────────────────────────────── - # Endpoint ids, `supports` whitelists and enum defaults below are taken - # from each model's FAL OpenAPI schema, so a key we send is a key the - # vendor declares. Paired `/edit` apps hang off their text-to-image entry - # rather than appearing as separate picker rows. - "bytedance/seedream/v5/pro/text-to-image": { - "display": "Seedream 5.0 Pro", - "speed": "~10s", - "strengths": "ByteDance flagship, dense layouts, native text in 14 languages", - "price": "$0.0675/image (≤1536²)", - "size_style": "image_size_preset", - # Pro requires total pixels between 1024x1024 and 2048x2048 — - # explicit ImageSize dicts keep every aspect inside that window. - "sizes": { + ), + # Entries below take endpoint ids, `supports` whitelists and enum defaults from + # each model's FAL OpenAPI schema; paired `/edit` apps hang off their + # text-to-image entry rather than appearing as separate picker rows. + # Seedream Pro requires total pixels between 1024² and 2048² — explicit + # ImageSize dicts keep every aspect inside that window. + "bytedance/seedream/v5/pro/text-to-image": _model( + "Seedream 5.0 Pro", "~10s", "ByteDance flagship, dense layouts, native text in 14 languages", "$0.0675/image (≤1536²)", + style="image_size_preset", sizes={ "landscape": {"width": 2048, "height": 1152}, "square": {"width": 1536, "height": 1536}, "portrait": {"width": 1152, "height": 2048}, }, - "defaults": { - "num_images": 1, - "output_format": "png", + defaults={ + "num_images": 1, "output_format": "png", "enable_safety_checker": False, + }, + supports={ + "prompt", "image_size", "num_images", "output_format", "sync_mode", "enable_safety_checker", + }, + edit_endpoint="bytedance/seedream/v5/pro/edit", + edit_supports={ + "prompt", "image_urls", "image_size", "num_images", "output_format", "sync_mode", + "enable_safety_checker", + }, + max_reference_images=10, + ), + # Lite wants 2560x1440..4096x4096 total pixels: use the documented presets (FAL + # auto-scales under the floor) rather than hand-rolled dicts that drift. + "bytedance/seedream/v5/lite/text-to-image": _model( + "Seedream 5.0 Lite", "~5s", "Fast/cheap Seedream tier, high-res output", "$0.035/image", + defaults={"num_images": 1, "enable_safety_checker": False}, + supports={ + "prompt", "image_size", "num_images", "max_images", "sync_mode", "enable_safety_checker", + }, + ), + "ideogram/v4/instant": _model( + "Ideogram V4 (Instant)", "<1s", "Latest Ideogram typography, posters/logos, instant", "$0.0075/MP", + defaults={ + "expansion_model": "Medium", "output_format": "png", "enable_safety_checker": False, }, - "supports": { - "prompt", "image_size", "num_images", "output_format", - "sync_mode", "enable_safety_checker", + supports={ + "prompt", "image_size", "expansion_model", "num_images", "seed", "sync_mode", + "enable_safety_checker", "output_format", }, - "upscale": False, - # Region-precise editing with up to 10 reference images. - "edit_endpoint": "bytedance/seedream/v5/pro/edit", - "edit_supports": { - "prompt", "image_urls", "image_size", "num_images", - "output_format", "sync_mode", "enable_safety_checker", + ), + "ideogram/v4/fast": _model( + "Ideogram V4 (Fast)", "~1s", "Ideogram V4 quality tiers via rendering_speed", "$0.005-0.018/MP", + defaults={"expansion_model": "Medium", "rendering_speed": "BALANCED"}, + supports={ + "prompt", "image_size", "expansion_model", "rendering_speed", "num_images", "seed", "sync_mode", }, - "max_reference_images": 10, - }, - "bytedance/seedream/v5/lite/text-to-image": { - "display": "Seedream 5.0 Lite", - "speed": "~5s", - "strengths": "Fast/cheap Seedream tier, high-res output", - "price": "$0.035/image", - "size_style": "image_size_preset", - # Lite wants total pixels between 2560x1440 and 4096x4096. Use the - # documented presets (FAL auto-scales if a preset is under the floor) - # instead of hand-rolled ImageSize dicts that drift from the schema. - "sizes": _PRESET_SIZES, - "defaults": { - "num_images": 1, + ), + "alibaba/qwen-image-3/text-to-image": _model( + "Qwen Image 3", "~8s", "Complex CN/EN text rendering, prompt-guided resolution", "$0.04 (1K) / $0.075 (2K) per image", + defaults={ + "num_images": 1, "output_format": "png", "enable_prompt_expansion": False, "enable_safety_checker": False, }, - "supports": { - "prompt", "image_size", "num_images", "max_images", - "sync_mode", "enable_safety_checker", - }, - "upscale": False, - }, - "ideogram/v4/instant": { - "display": "Ideogram V4 (Instant)", - "speed": "<1s", - "strengths": "Latest Ideogram typography, posters/logos, instant", - "price": "$0.0075/MP", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "expansion_model": "Medium", - "output_format": "png", - "enable_safety_checker": False, - }, - "supports": { - "prompt", "image_size", "expansion_model", "num_images", - "seed", "sync_mode", "enable_safety_checker", "output_format", - }, - "upscale": False, - }, - "ideogram/v4/fast": { - "display": "Ideogram V4 (Fast)", - "speed": "~1s", - "strengths": "Ideogram V4 quality tiers via rendering_speed", - "price": "$0.005-0.018/MP", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "expansion_model": "Medium", - "rendering_speed": "BALANCED", - }, - "supports": { - "prompt", "image_size", "expansion_model", "rendering_speed", - "num_images", "seed", "sync_mode", - }, - "upscale": False, - }, - "alibaba/qwen-image-3/text-to-image": { - "display": "Qwen Image 3", - "speed": "~8s", - "strengths": "Complex CN/EN text rendering, prompt-guided resolution", - "price": "$0.04 (1K) / $0.075 (2K) per image", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "num_images": 1, - "output_format": "png", - "enable_prompt_expansion": False, # avoid the LLM rewrite surprise - "enable_safety_checker": False, - }, - "supports": { - "prompt", "negative_prompt", "image_size", "num_images", - "seed", "sync_mode", "output_format", + supports={ + "prompt", "negative_prompt", "image_size", "num_images", "seed", "sync_mode", "output_format", "enable_prompt_expansion", "enable_safety_checker", }, - "upscale": False, - # Qwen Image 3 edit: 1-3 reference images, identity-preserving edits. - "edit_endpoint": "alibaba/qwen-image-3/edit", - "edit_supports": { - "prompt", "image_urls", "negative_prompt", "num_images", - "seed", "sync_mode", "output_format", + edit_endpoint="alibaba/qwen-image-3/edit", + edit_supports={ + "prompt", "image_urls", "negative_prompt", "num_images", "seed", "sync_mode", "output_format", "enable_prompt_expansion", "enable_safety_checker", }, - "max_reference_images": 3, - }, - "microsoft/mai-image-2.5-pro": { - "display": "MAI Image 2.5 Pro", - "speed": "~10s", - "strengths": "Microsoft flagship, hero imagery, precise typography", - "price": "~$0.17/image", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "num_images": 1, - "output_format": "png", - }, - "supports": { - "prompt", "aspect_ratio", "num_images", "output_format", - "sync_mode", - }, - "upscale": False, - }, - "google/nano-banana-2-lite": { - "display": "Nano Banana 2 Lite", - "speed": "<2s", - "strengths": "Gemini image family, sub-2s, 14 aspect ratios incl. extreme", - "price": "~$0.04/image (1K fixed)", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "num_images": 1, - "output_format": "png", - "safety_tolerance": "5", - }, - "supports": { - "prompt", "aspect_ratio", "num_images", "seed", - "output_format", "safety_tolerance", "sync_mode", + max_reference_images=3, + ), + "microsoft/mai-image-2.5-pro": _model( + "MAI Image 2.5 Pro", "~10s", "Microsoft flagship, hero imagery, precise typography", "~$0.17/image", + style="aspect_ratio", + defaults={"num_images": 1, "output_format": "png"}, + supports={"prompt", "aspect_ratio", "num_images", "output_format", "sync_mode"}, + ), + "google/nano-banana-2-lite": _model( + "Nano Banana 2 Lite", "<2s", "Gemini image family, sub-2s, 14 aspect ratios incl. extreme", "~$0.04/image (1K fixed)", + style="aspect_ratio", + defaults={"num_images": 1, "output_format": "png", "safety_tolerance": "5"}, + supports={ + "prompt", "aspect_ratio", "num_images", "seed", "output_format", "safety_tolerance", "sync_mode", "system_prompt", "limit_generations", "thinking_level", }, - "upscale": False, - # Fast multi-turn local edits with reference images via `image_urls`. - "edit_endpoint": "google/nano-banana-2-lite/edit", - "edit_supports": { - "prompt", "image_urls", "aspect_ratio", "num_images", - "seed", "output_format", "safety_tolerance", "sync_mode", - "system_prompt", + edit_endpoint="google/nano-banana-2-lite/edit", + edit_supports={ + "prompt", "image_urls", "aspect_ratio", "num_images", "seed", "output_format", "safety_tolerance", + "sync_mode", "system_prompt", }, - "max_reference_images": 4, - }, - "fal-ai/recraft/v4.1/text-to-image": { - "display": "Recraft V4.1", - "speed": "~8s", - "strengths": "Design-first raster, brand systems, editorial", - "price": "$0.035/image", - "size_style": "image_size_preset", - "sizes": _PRESET_SIZES, - "defaults": { - "enable_safety_checker": False, + max_reference_images=4, + ), + "fal-ai/recraft/v4.1/text-to-image": _model( + "Recraft V4.1", "~8s", "Design-first raster, brand systems, editorial", "$0.035/image", + defaults={"enable_safety_checker": False}, + supports={ + "prompt", "image_size", "enable_safety_checker", "colors", "background_color", }, - "supports": { - "prompt", "image_size", "enable_safety_checker", - "colors", "background_color", + ), + "xai/grok-imagine-image/v2.0/text-to-image": _model( + "Grok Imagine Image 2.0", "~5s", "xAI. Design-grade typography/layout, instruction following", "$0.06/image (1K medium)", + style="aspect_ratio", + # 1k + medium is the cheapest sensible tier; 2k is roughly +33%/image. 1k native + # is sub-2MP — pass upscale=true per call when needed. Edits omit aspect_ratio + # (defaults to "auto", following the first input image). + defaults={ + "num_images": 1, "output_format": "png", "resolution": "1k", "quality": "medium", }, - "upscale": False, - }, - "xai/grok-imagine-image/v2.0/text-to-image": { - "display": "Grok Imagine Image 2.0", - "speed": "~5s", - "strengths": "xAI. Design-grade typography/layout, instruction following", - "price": "$0.06/image (1K medium)", - "size_style": "aspect_ratio", - "sizes": _ASPECT_SIZES, - "defaults": { - "num_images": 1, - "output_format": "png", - # 1k + medium is the cheapest sensible tier ($0.06/image); - # 2k roughly +33% per image. - "resolution": "1k", - "quality": "medium", + supports={ + "prompt", "aspect_ratio", "num_images", "output_format", "resolution", "quality", "sync_mode", }, - "supports": { - "prompt", "aspect_ratio", "num_images", "output_format", - "resolution", "quality", "sync_mode", + edit_endpoint="xai/grok-imagine-image/v2.0/edit", + edit_supports={ + "prompt", "image_urls", "num_images", "output_format", "resolution", "quality", "sync_mode", }, - "upscale": False, - # Edit endpoint takes `image_urls` (max 3) + the same knobs; - # aspect_ratio defaults to "auto" (follows the first input image), - # so we don't send it on edits. - "edit_endpoint": "xai/grok-imagine-image/v2.0/edit", - "edit_supports": { - "prompt", "image_urls", "num_images", "output_format", - "resolution", "quality", "sync_mode", - }, - "max_reference_images": 3, - }, + max_reference_images=3, + ), } diff --git a/tools/image_generation_tool.py b/tools/image_generation_tool.py index b98f3a4361..11fde4e4ac 100644 --- a/tools/image_generation_tool.py +++ b/tools/image_generation_tool.py @@ -1,19 +1,10 @@ #!/usr/bin/env python3 -""" -Image Generation Tools Module +"""Image generation via FAL.ai (model picked in ``hermes tools``, persisted to ``image_gen.model``). -Provides image generation via FAL.ai. Multiple FAL models are supported and -selectable via ``hermes tools`` → Image Generation; the active model is -persisted to ``image_gen.model`` in ``config.yaml``. - -Architecture: -- ``FAL_MODELS`` (``tools.image_generation_catalog``) holds per-model metadata - (size-style family, defaults, ``supports`` whitelist, upscaler flag). -- ``_build_fal_payload()`` / ``_build_fal_edit_payload()`` translate the - unified inputs into the model-specific payload, filtered to the whitelist so - models never receive rejected keys. -- Upscaling (Clarity Upscaler) is strictly per-call opt-in: chained by default - it degraded text rendering, CJK and faces, so ``upscale`` is False everywhere. +``FAL_MODELS`` (``tools.image_generation_catalog``) holds per-model metadata; +``_build_fal_payload()`` / ``_build_fal_edit_payload()`` translate unified inputs into +the model payload filtered to its ``supports`` whitelist so models never receive rejected +keys. Clarity upscaling is strictly per-call opt-in: default-on degraded text/CJK/faces. """ import json @@ -24,10 +15,9 @@ import threading import uuid from typing import Any, Dict, Optional -# fal_client is imported lazily (see _load_fal_client): an eager import cost -# ~64 ms on every CLI cold start because discover_builtin_tools() imports this -# module unconditionally. Tests that monkeypatch this attribute keep working: -# _load_fal_client() short-circuits when it is already truthy. +# Imported lazily by _load_fal_client(): the eager import cost ~64 ms on every CLI cold +# start (discover_builtin_tools() imports this module unconditionally). Tests that +# monkeypatch this attribute keep working because the loader short-circuits when truthy. fal_client: Any = None @@ -84,35 +74,25 @@ _managed_fal_client_lock = threading.Lock() # Managed FAL gateway (Nous Subscription) # --------------------------------------------------------------------------- def _resolve_managed_fal_gateway(): - """Resolve the FAL route from the stored `hermes tools` selection. + """Managed gateway config for the stored `hermes tools` selection, or ``None`` for direct FAL. - - ``"nous"`` (or legacy ``use_gateway: true``) → managed gateway ONLY; not - entitled/unreachable is a selection-naming error, never a silent FAL_KEY fallback. - - any other stored provider → direct FAL ONLY; missing FAL_KEY is an error - naming FAL_KEY and the selection, never a silent managed reroute. - - never configured → legacy autodetect: direct when FAL_KEY is set, else - the managed gateway when resolvable, else None. - - Returns the managed gateway config, or ``None`` for the direct route. + ``"nous"`` (or legacy ``use_gateway: true``) → managed ONLY: not entitled/unreachable is a + selection-naming error, never a silent FAL_KEY fallback. Any other stored provider → direct + ONLY: missing FAL_KEY is an error naming FAL_KEY and the selection, never a silent managed + reroute. Never configured → legacy autodetect: direct if FAL_KEY, else managed if resolvable. """ selected = read_selection("image_gen") if selected == NOUS_MANAGED_PROVIDER: gateway = resolve_managed_tool_gateway("fal-queue") if gateway is None: raise ValueError(selection_error( - "image_gen", - NOUS_MANAGED_PROVIDER, - "the Nous Tool Gateway is not available (not entitled or " - "unreachable)", + "image_gen", NOUS_MANAGED_PROVIDER, + "the Nous Tool Gateway is not available (not entitled or unreachable)", )) return gateway if selected is not None: if not fal_key_is_configured(): - raise ValueError(selection_error( - "image_gen", - selected, - "FAL_KEY is not set", - )) + raise ValueError(selection_error("image_gen", selected, "FAL_KEY is not set")) return None # Never-configured category: legacy credential autodetect (do NOT persist). if fal_key_is_configured(): @@ -124,16 +104,12 @@ def _get_managed_fal_client(managed_gateway): """Reuse the managed FAL client so its internal httpx.Client is not leaked per call.""" global _managed_fal_client, _managed_fal_client_config - client_config = ( - managed_gateway.gateway_origin.rstrip("/"), - managed_gateway.nous_user_token, - ) + client_config = (managed_gateway.gateway_origin.rstrip("/"), managed_gateway.nous_user_token) with _managed_fal_client_lock: if _managed_fal_client is not None and _managed_fal_client_config == client_config: return _managed_fal_client - # Resolve fal_client on this module so monkeypatching - # ``image_generation_tool.fal_client`` still takes effect. + # Resolved on this module so monkeypatching ``image_generation_tool.fal_client`` still applies. _load_fal_client() _managed_fal_client = _ManagedFalSyncClient( fal_client, @@ -149,12 +125,11 @@ class ImageGenerationInterrupted(Exception): def _wait_fal_result(handler, *, poll_seconds: float = 0.5): - """Interrupt-aware replacement for a blind ``handler.get()``. + """Interrupt-aware ``handler.get()``: the SDK blocks 30-60s, hiding user interrupts. - ``handler.get()`` blocks inside the FAL SDK for 30-60s, during which a user - interrupt was invisible. Run it on a daemon worker and poll the per-thread - interrupt bit between join slices; on interrupt, abandon the worker (the - remote job keeps running) and raise ``ImageGenerationInterrupted``. + Runs the get on a daemon worker and polls the per-thread interrupt bit between join + slices; on interrupt the worker is abandoned (remote job keeps running) and + ``ImageGenerationInterrupted`` is raised. """ from tools.interrupt import is_interrupted @@ -172,8 +147,7 @@ def _wait_fal_result(handler, *, poll_seconds: float = 0.5): while worker.is_alive(): if is_interrupted(): raise ImageGenerationInterrupted( - "Image generation interrupted by user — abandoned the " - "in-flight FAL job." + "Image generation interrupted by user — abandoned the in-flight FAL job." ) worker.join(timeout=poll_seconds) if error_box: @@ -191,25 +165,16 @@ def _submit_fal_request(model: str, arguments: Dict[str, Any]): managed_client = _get_managed_fal_client(managed_gateway) try: - return managed_client.submit( - model, - arguments=arguments, - headers=request_headers, - ) + return managed_client.submit(model, arguments=arguments, headers=request_headers) except Exception as exc: - # A 4xx from the managed gateway usually means the portal doesn't proxy - # this model (allowlist miss, billing gate) — give actionable remediation - # instead of a raw httpx error. + # A managed-gateway 4xx usually means the portal doesn't proxy this model + # (allowlist miss, billing gate): give remediation instead of a raw httpx error. status = _extract_http_status(exc) if status is not None and 400 <= status < 500: gateway_message = "" if status in {401, 402, 403}: - gateway_message = ( - "\n\n" - + nous_tool_gateway_unavailable_message( - "managed FAL image generation", - force_fresh=True, - ) + gateway_message = "\n\n" + nous_tool_gateway_unavailable_message( + "managed FAL image generation", force_fresh=True, ) raise ValueError( f"Nous Subscription gateway rejected model '{model}' " @@ -231,54 +196,47 @@ def _read_image_gen_key(key: str) -> Optional[str]: from hermes_cli.config import load_config cfg = load_config() section = cfg.get("image_gen") if isinstance(cfg, dict) else None - if isinstance(section, dict): - value = section.get(key) - if isinstance(value, str) and value.strip(): - return value.strip() + value = section.get(key) if isinstance(section, dict) else None + if isinstance(value, str) and value.strip(): + return value.strip() except Exception as exc: logger.debug("Could not read image_gen.%s: %s", key, exc) return None def _read_configured_image_model(): - """Return the value of ``image_gen.model`` from config.yaml, or None.""" + """``image_gen.model`` from config.yaml, or None.""" return _read_image_gen_key("model") def _read_configured_image_provider(): - """Return ``image_gen.provider`` from config.yaml, or None. + """``image_gen.provider`` from config.yaml, or None. - The plugin registry is consulted only when this is explicitly set — an - unset value keeps users on the in-tree FAL fallback even when other - providers happen to be registered (e.g. OPENAI_API_KEY present for other - features). ``"fal"`` explicitly routes through ``plugins/image_gen/fal/``, - which delegates back into this module via call-time indirection. + The plugin registry is consulted only when this is explicitly set — unset keeps + users on the in-tree FAL fallback even when other providers are registered (e.g. + OPENAI_API_KEY present for other features). ``"fal"`` routes through + ``plugins/image_gen/fal/``, which delegates back here via call-time indirection. """ return _read_image_gen_key("provider") def _resolve_fal_model() -> tuple: - """Resolve the active FAL model from config.yaml (primary) or default. - - Returns (model_id, metadata_dict). Falls back to DEFAULT_MODEL if the - configured model is unknown (logged as a warning). - """ + """Return ``(model_id, meta)`` for the configured FAL model, falling back to DEFAULT_MODEL (warned) when unknown.""" # FAL_IMAGE_MODEL is an undocumented escape hatch (backward-compat for tests/scripts). model_id = _read_image_gen_key("model") or os.getenv("FAL_IMAGE_MODEL", "").strip() - - if not model_id: - return DEFAULT_MODEL, FAL_MODELS[DEFAULT_MODEL] - - if model_id not in FAL_MODELS: + if model_id and model_id not in FAL_MODELS: logger.warning( "Unknown FAL model '%s' in config; falling back to %s", model_id, DEFAULT_MODEL, ) - return DEFAULT_MODEL, FAL_MODELS[DEFAULT_MODEL] - + model_id = None + model_id = model_id or DEFAULT_MODEL return model_id, FAL_MODELS[model_id] +_SIZE_KEY_BY_STYLE = {"image_size_preset": "image_size", "gpt_literal": "image_size", "aspect_ratio": "aspect_ratio"} + + def _build_payload( model_id: str, prompt: str, @@ -287,21 +245,17 @@ def _build_payload( overrides: Optional[Dict[str, Any]], image_urls: Optional[list] = None, ) -> Dict[str, Any]: - """Shared text-to-image / edit payload builder (``image_urls`` selects edit mode). + """Text-to-image / edit payload (``image_urls`` selects edit mode): defaults + native size + spec + overrides, filtered to the model whitelist. - Translates aspect_ratio into the model's native size spec, merges model - defaults, applies caller overrides, then filters to the model's whitelist. - Edit endpoints mostly auto-infer output size from the input image, so the - size key is only sent when ``edit_supports`` advertises it. ``prompt`` (and - ``image_urls`` on edits) are required by every FAL endpoint and are kept - even if a whitelist omits them, so a catalog gap can't send a broken request. + Edit endpoints mostly auto-infer size from the input, so the size key is sent only when + ``edit_supports`` advertises it. ``prompt`` (and ``image_urls`` on edits) are required by + every FAL endpoint and survive a whitelist gap so a catalog mistake can't send a broken request. """ meta = FAL_MODELS[model_id] edit = image_urls is not None supports = (meta.get("edit_supports") or set()) if edit else meta["supports"] - size_style = meta["size_style"] sizes = meta["sizes"] - aspect = (aspect_ratio or DEFAULT_ASPECT_RATIO).lower().strip() if aspect not in sizes: aspect = DEFAULT_ASPECT_RATIO @@ -313,48 +267,24 @@ def _build_payload( payload["image_urls"] = list(image_urls) required.add("image_urls") - if size_style in {"image_size_preset", "gpt_literal"}: - size_key = "image_size" - elif size_style == "aspect_ratio": - size_key = "aspect_ratio" - elif edit: - size_key = None - else: - raise ValueError(f"Unknown size_style: {size_style!r}") + size_key = _SIZE_KEY_BY_STYLE.get(meta["size_style"]) + if size_key is None and not edit: + raise ValueError(f"Unknown size_style: {meta['size_style']!r}") if size_key is not None and (not edit or size_key in supports): payload[size_key] = sizes[aspect] - - if seed is not None and isinstance(seed, int): + if isinstance(seed, int): payload["seed"] = seed - - if overrides: - for k, v in overrides.items(): - if v is not None: - payload[k] = v - + payload.update({k: v for k, v in (overrides or {}).items() if v is not None}) return {k: v for k, v in payload.items() if k in supports or k in required} -def _build_fal_payload( - model_id: str, - prompt: str, - aspect_ratio: str = DEFAULT_ASPECT_RATIO, - seed: Optional[int] = None, - overrides: Optional[Dict[str, Any]] = None, -) -> Dict[str, Any]: - """Build a FAL text-to-image payload for `model_id` from unified inputs.""" +def _build_fal_payload(model_id, prompt, aspect_ratio=DEFAULT_ASPECT_RATIO, seed=None, overrides=None): + """FAL text-to-image payload for ``model_id`` from unified inputs.""" return _build_payload(model_id, prompt, aspect_ratio, seed, overrides) -def _build_fal_edit_payload( - model_id: str, - prompt: str, - image_urls: list, - aspect_ratio: str = DEFAULT_ASPECT_RATIO, - seed: Optional[int] = None, - overrides: Optional[Dict[str, Any]] = None, -) -> Dict[str, Any]: - """Build a FAL *edit* (image-to-image) payload: ``image_urls`` + prompt, filtered to ``edit_supports``.""" +def _build_fal_edit_payload(model_id, prompt, image_urls, aspect_ratio=DEFAULT_ASPECT_RATIO, seed=None, overrides=None): + """FAL *edit* (image-to-image) payload: ``image_urls`` + prompt, filtered to ``edit_supports``.""" return _build_payload(model_id, prompt, aspect_ratio, seed, overrides, image_urls=image_urls) @@ -365,8 +295,7 @@ def _upscale_image(image_url: str, original_prompt: str) -> Optional[Dict[str, A """Upscale via FAL's Clarity Upscaler; None on failure (caller keeps the original).""" try: logger.info("Upscaling image with Clarity Upscaler...") - - upscaler_arguments = { + handler = _submit_fal_request(UPSCALER_MODEL, arguments={ "image_url": image_url, "prompt": f"{UPSCALER_DEFAULT_PROMPT}, {original_prompt}", "upscale_factor": UPSCALER_FACTOR, @@ -376,24 +305,17 @@ def _upscale_image(image_url: str, original_prompt: str) -> Optional[Dict[str, A "guidance_scale": UPSCALER_GUIDANCE_SCALE, "num_inference_steps": UPSCALER_NUM_INFERENCE_STEPS, "enable_safety_checker": UPSCALER_SAFETY_CHECKER, - } - - handler = _submit_fal_request(UPSCALER_MODEL, arguments=upscaler_arguments) + }) result = _wait_fal_result(handler) - if result and "image" in result: - upscaled_image = result["image"] + up = result["image"] logger.info( "Image upscaled successfully to %sx%s", - upscaled_image.get("width", "unknown"), - upscaled_image.get("height", "unknown"), + up.get("width", "unknown"), up.get("height", "unknown"), ) return { - "url": upscaled_image["url"], - "width": upscaled_image.get("width", 0), - "height": upscaled_image.get("height", 0), - "upscaled": True, - "upscale_factor": UPSCALER_FACTOR, + "url": up["url"], "width": up.get("width", 0), "height": up.get("height", 0), + "upscaled": True, "upscale_factor": UPSCALER_FACTOR, } logger.error("Upscaler returned invalid response") return None @@ -410,14 +332,9 @@ def _upscale_image(image_url: str, original_prompt: str) -> Optional[Dict[str, A # Artifact path hinting for non-local terminal backends # --------------------------------------------------------------------------- def _looks_like_absolute_file_path(value: str) -> bool: - if not value or not isinstance(value, str): + if not value or not isinstance(value, str) or value.lower().startswith(("http://", "https://", "data:")): return False - lower = value.lower() - if lower.startswith(("http://", "https://", "data:")): - return False - if os.path.isabs(value): - return True - return len(value) >= 3 and value[1] == ":" and value[2] in {"/", "\\"} + return os.path.isabs(value) or (len(value) >= 3 and value[1] == ":" and value[2] in {"/", "\\"}) def _active_terminal_env(task_id: str | None): @@ -446,30 +363,21 @@ def _agent_cache_base_for_env(env: Any) -> str | None: remote_home = getattr(env, "_remote_home", None) if remote_home: return f"{str(remote_home).rstrip('/')}/.hermes" - - env_name = env.__class__.__name__ - if env_name in {"DockerEnvironment", "SingularityEnvironment", "ModalEnvironment"}: + if env.__class__.__name__ in {"DockerEnvironment", "SingularityEnvironment", "ModalEnvironment"}: return "/root/.hermes" # No environment yet: only backends with deterministic cache roots can be # translated without side effects. SSH can use a shell-visible tilde path; # its first environment sync uploads the cache file before the first command. backend = (os.getenv("TERMINAL_ENV") or "local").strip().lower() - if backend in {"docker", "singularity", "modal"}: - return "/root/.hermes" - if backend == "ssh": - return "~/.hermes" - return None + return {"docker": "/root/.hermes", "singularity": "/root/.hermes", + "modal": "/root/.hermes", "ssh": "~/.hermes"}.get(backend) def _agent_visible_cache_path(host_path: str, env: Any) -> str | None: - if not _looks_like_absolute_file_path(host_path): - return None - - cache_base = _agent_cache_base_for_env(env) + cache_base = _agent_cache_base_for_env(env) if _looks_like_absolute_file_path(host_path) else None if not cache_base: return None - try: from tools.credential_files import map_cache_path_to_container @@ -490,20 +398,14 @@ def _force_artifact_sync(env: Any) -> None: def _postprocess_image_generate_result(raw: str, task_id: str | None = None) -> str: - """Annotate successful local image results with backend-visible paths. - - ``image`` stays the host/gateway-deliverable path; when the active terminal - backend has a different filesystem, ``agent_visible_image`` is the path the - agent can use with terminal/file tools. - """ + """Annotate successful local results: ``image`` stays the host/gateway-deliverable path; + ``agent_visible_image`` is the same file as seen by a non-local terminal backend.""" try: payload = json.loads(raw) if isinstance(raw, str) else raw except Exception: return raw - if not isinstance(payload, dict) or not payload.get("success"): return raw - image = payload.get("image") if not isinstance(image, str) or not _looks_like_absolute_file_path(image): return raw @@ -512,10 +414,8 @@ def _postprocess_image_generate_result(raw: str, task_id: str | None = None) -> agent_path = _agent_visible_cache_path(image, env) if not agent_path or agent_path == image: return raw - if env is not None: _force_artifact_sync(env) - payload.setdefault("host_image", image) payload.setdefault("agent_visible_image", agent_path) return json.dumps(payload, ensure_ascii=False) @@ -524,6 +424,84 @@ def _postprocess_image_generate_result(raw: str, task_id: str | None = None) -> # --------------------------------------------------------------------------- # Tool entry point # --------------------------------------------------------------------------- +def _collect_source_images(image_url, reference_image_urls) -> list: + """Primary + reference source images as one ordered list of stripped, non-empty strings.""" + candidates = [image_url] + if isinstance(reference_image_urls, (list, tuple)): + candidates.extend(reference_image_urls) + return [c.strip() for c in candidates if isinstance(c, str) and c.strip()] + + +def _format_images(images: list, should_upscale: bool, prompt: str) -> list: + """Normalize FAL result images, optionally chaining the upscaler (falls back to the original on failure).""" + formatted = [] + for img in images: + if not (isinstance(img, dict) and "url" in img): + continue + if should_upscale: + upscaled = _upscale_image(img["url"], prompt.strip()) + if upscaled: + formatted.append(upscaled) + continue + logger.warning("Using original image as fallback (upscale failed)") + formatted.append({ + "url": img["url"], "width": img.get("width", 0), "height": img.get("height", 0), + "upscaled": False, + }) + return formatted + + +def _finish_image_call(debug_call_data: Dict[str, Any], generation_time: float, response: Dict[str, Any]) -> str: + """Record generation time, log the debug entry and return the JSON result.""" + debug_call_data["generation_time"] = generation_time + _debug.log_call("image_generate_tool", debug_call_data) + _debug.save() + return json.dumps(response, indent=2, ensure_ascii=False) + + +def _prepare_fal_request(model_id, meta, prompt, aspect_ratio, seed, overrides, source_images): + """Validate inputs and return ``(endpoint, arguments)``; raises ValueError with the user-facing message.""" + if not isinstance(prompt, str) or not prompt.strip(): + raise ValueError("Prompt is required and must be a non-empty string") + + # A stored-but-broken selection raises the selection-naming error from + # _resolve_managed_fal_gateway(); only never-configured reports "no backend at all". + if not (fal_key_is_configured() or _resolve_managed_fal_gateway()): + raise ValueError(_build_no_backend_setup_message()) + + edit_endpoint = meta.get("edit_endpoint") + display = meta.get("display", model_id) + # Fail clearly rather than silently dropping sources and producing an unrelated picture. + if source_images and not edit_endpoint: + raise ValueError( + f"Model '{display}' ({model_id}) is not " + f"capable of image-to-image / editing. Provide a text-only " + f"prompt (omit image_url), or switch to an edit-capable model " + f"via `hermes tools` → Image Generation." + ) + + aspect_lc = (aspect_ratio or DEFAULT_ASPECT_RATIO).lower().strip() + if aspect_lc not in VALID_ASPECT_RATIOS: + logger.warning("Invalid aspect_ratio '%s', defaulting to '%s'", aspect_ratio, DEFAULT_ASPECT_RATIO) + aspect_lc = DEFAULT_ASPECT_RATIO + + if source_images: + # Clamp reference count to the model's declared cap. + max_refs = int(meta.get("max_reference_images") or 1) + clamped_sources = source_images[:max_refs] if max_refs > 0 else source_images + arguments = _build_fal_edit_payload( + model_id, prompt, clamped_sources, aspect_lc, seed=seed, overrides=overrides, + ) + logger.info( + "Editing image with %s (%s) — %d source image(s), prompt: %s", + display, edit_endpoint, len(clamped_sources), prompt[:80], + ) + return edit_endpoint, arguments + arguments = _build_fal_payload(model_id, prompt, aspect_lc, seed=seed, overrides=overrides) + logger.info("Generating image with %s (%s) — prompt: %s", display, model_id, prompt[:80]) + return model_id, arguments + + def image_generate_tool( prompt: str, aspect_ratio: str = DEFAULT_ASPECT_RATIO, @@ -536,155 +514,57 @@ def image_generate_tool( reference_image_urls: Optional[list] = None, upscale: Optional[bool] = None, ) -> str: - """Generate an image from a text prompt, or edit a source image, via FAL. + """Generate (or, with source images + an ``edit_endpoint`` model, edit) an image via FAL. - Routing: ``image_url`` / ``reference_image_urls`` plus a model with an - ``edit_endpoint`` → image-to-image; otherwise text-to-image. The extra - kwargs are overrides for direct Python callers, filtered per-model via the - ``supports`` / ``edit_supports`` whitelist (unsupported ones are dropped - silently so legacy callers survive model switches). - - Returns a JSON string with ``{"success": bool, "image": url | None, - "modality": "text" | "image", "error": str, "error_type": str}``. + Extra kwargs are overrides for direct Python callers, filtered per-model via the + ``supports`` / ``edit_supports`` whitelist (dropped silently so legacy callers survive + model switches). Returns JSON ``{"success", "image", "modality", "error", "error_type"}``. """ model_id, meta = _resolve_fal_model() - - # Collect any source images (primary + references) into one ordered list. - source_images: list = [] - if isinstance(image_url, str) and image_url.strip(): - source_images.append(image_url.strip()) - if isinstance(reference_image_urls, (list, tuple)): - for ref in reference_image_urls: - if isinstance(ref, str) and ref.strip(): - source_images.append(ref.strip()) - - edit_endpoint = meta.get("edit_endpoint") - use_edit = bool(source_images) and bool(edit_endpoint) + source_images = _collect_source_images(image_url, reference_image_urls) + use_edit = bool(source_images) and bool(meta.get("edit_endpoint")) modality = "image" if use_edit else "text" + params = { + "prompt": prompt, "aspect_ratio": aspect_ratio, + "num_inference_steps": num_inference_steps, "guidance_scale": guidance_scale, + "num_images": num_images, "output_format": output_format, "seed": seed, + } debug_call_data = { "model": model_id, - "parameters": { - "prompt": prompt, - "aspect_ratio": aspect_ratio, - "num_inference_steps": num_inference_steps, - "guidance_scale": guidance_scale, - "num_images": num_images, - "output_format": output_format, - "seed": seed, - "modality": modality, - "source_images": len(source_images), - }, - "error": None, - "success": False, - "images_generated": 0, - "generation_time": 0, + "parameters": {**params, "modality": modality, "source_images": len(source_images)}, + "error": None, "success": False, "images_generated": 0, "generation_time": 0, } - start_time = datetime.datetime.now() try: - if not prompt or not isinstance(prompt, str) or len(prompt.strip()) == 0: - raise ValueError("Prompt is required and must be a non-empty string") - - # A stored-but-broken selection raises the selection-naming error from - # _resolve_managed_fal_gateway(); only the never-configured path can - # report "no backend at all". - if not (fal_key_is_configured() or _resolve_managed_fal_gateway()): - raise ValueError(_build_no_backend_setup_message()) - - # Source images on a model without an edit endpoint: fail clearly rather - # than silently dropping them and producing an unrelated picture. - if source_images and not edit_endpoint: - raise ValueError( - f"Model '{meta.get('display', model_id)}' ({model_id}) is not " - f"capable of image-to-image / editing. Provide a text-only " - f"prompt (omit image_url), or switch to an edit-capable model " - f"via `hermes tools` → Image Generation." - ) - - aspect_lc = (aspect_ratio or DEFAULT_ASPECT_RATIO).lower().strip() - if aspect_lc not in VALID_ASPECT_RATIOS: - logger.warning( - "Invalid aspect_ratio '%s', defaulting to '%s'", - aspect_ratio, DEFAULT_ASPECT_RATIO, - ) - aspect_lc = DEFAULT_ASPECT_RATIO - overrides: Dict[str, Any] = { - k: v for k, v in ( - ("num_inference_steps", num_inference_steps), - ("guidance_scale", guidance_scale), - ("num_images", num_images), - ("output_format", output_format), - ) if v is not None + k: params[k] for k in ("num_inference_steps", "guidance_scale", "num_images", "output_format") + if params[k] is not None } - - if use_edit: - # Clamp reference count to the model's declared cap. - max_refs = int(meta.get("max_reference_images") or 1) - clamped_sources = source_images[:max_refs] if max_refs > 0 else source_images - arguments = _build_fal_edit_payload( - model_id, prompt, clamped_sources, aspect_lc, - seed=seed, overrides=overrides, - ) - endpoint = edit_endpoint - logger.info( - "Editing image with %s (%s) — %d source image(s), prompt: %s", - meta.get("display", model_id), endpoint, len(clamped_sources), - prompt[:80], - ) - else: - arguments = _build_fal_payload( - model_id, prompt, aspect_lc, seed=seed, overrides=overrides, - ) - endpoint = model_id - logger.info( - "Generating image with %s (%s) — prompt: %s", - meta.get("display", model_id), model_id, prompt[:80], - ) - + endpoint, arguments = _prepare_fal_request( + model_id, meta, prompt, aspect_ratio, seed, overrides, source_images, + ) handler = _submit_fal_request(endpoint, arguments=arguments) result = _wait_fal_result(handler) - generation_time = (datetime.datetime.now() - start_time).total_seconds() if not result or "images" not in result: raise ValueError("Invalid response from FAL.ai API — no images returned") - images = result.get("images", []) if not images: raise ValueError("No images were generated") - # An explicit ``upscale`` wins over the catalog default, including for - # edits (an explicit request is intentional). The catalog default never - # upscales edits: Clarity is a text-to-image quality pass and must not - # silently alter edit compositions. + # An explicit ``upscale`` wins over the catalog default, including for edits + # (an explicit request is intentional). The catalog default never upscales + # edits: Clarity is a text-to-image quality pass and must not silently alter + # edit compositions. if upscale is not None: should_upscale = bool(upscale) else: should_upscale = bool(meta.get("upscale", False)) and not use_edit - formatted_images = [] - for img in images: - if not (isinstance(img, dict) and "url" in img): - continue - original_image = { - "url": img["url"], - "width": img.get("width", 0), - "height": img.get("height", 0), - } - - if should_upscale: - upscaled_image = _upscale_image(img["url"], prompt.strip()) - if upscaled_image: - formatted_images.append(upscaled_image) - continue - logger.warning("Using original image as fallback (upscale failed)") - - original_image["upscaled"] = False - formatted_images.append(original_image) - + formatted_images = _format_images(images, should_upscale, prompt) if not formatted_images: raise ValueError("No valid image URLs returned from API") @@ -694,47 +574,33 @@ def image_generate_tool( len(formatted_images), generation_time, upscaled_count, endpoint, modality, ) - - response_data = { - "success": True, - "image": formatted_images[0]["url"] if formatted_images else None, - "modality": modality, - "upscaled": bool(formatted_images and formatted_images[0].get("upscaled")), - } - debug_call_data["success"] = True debug_call_data["images_generated"] = len(formatted_images) - debug_call_data["generation_time"] = generation_time - _debug.log_call("image_generate_tool", debug_call_data) - _debug.save() - - return json.dumps(response_data, indent=2, ensure_ascii=False) + return _finish_image_call(debug_call_data, generation_time, { + "success": True, + "image": formatted_images[0]["url"], + "modality": modality, + "upscaled": bool(formatted_images[0].get("upscaled")), + }) except Exception as e: - generation_time = (datetime.datetime.now() - start_time).total_seconds() error_msg = f"Error generating image: {str(e)}" logger.error("%s", error_msg, exc_info=True) - - response_data = { + debug_call_data["error"] = error_msg + generation_time = (datetime.datetime.now() - start_time).total_seconds() + return _finish_image_call(debug_call_data, generation_time, { "success": False, "image": None, "error": str(e), "error_type": type(e).__name__, - } - - debug_call_data["error"] = error_msg - debug_call_data["generation_time"] = generation_time - _debug.log_call("image_generate_tool", debug_call_data) - _debug.save() - - return json.dumps(response_data, indent=2, ensure_ascii=False) + }) def check_fal_api_key() -> bool: - """True if the FAL backend selected via `hermes tools` (or, never configured, any FAL backend) is available. + """True if the selected FAL backend (never configured: any FAL backend) is available. - A stored-but-broken selection reports False here (registry gating); the - selection-naming error surfaces at call time from ``_resolve_managed_fal_gateway``. + A stored-but-broken selection reports False here (registry gating); the naming error + surfaces at call time from ``_resolve_managed_fal_gateway``. """ selected = read_selection("image_gen") if selected == NOUS_MANAGED_PROVIDER: @@ -745,28 +611,23 @@ def check_fal_api_key() -> bool: def _build_no_backend_setup_message() -> str: - """Actionable error when no FAL backend is reachable: FAL_KEY signup, - managed-gateway status (if Nous tools enabled), and the plugin alternative.""" - lines = ["Image generation is unavailable in this environment.", ""] - lines.append("Missing requirements:") - if managed_nous_tools_enabled(): - lines.append( - " - FAL_KEY is not set and the managed FAL gateway is unreachable" - ) + """Actionable no-backend error: FAL_KEY signup, managed-gateway status, plugin alternative.""" + managed = managed_nous_tools_enabled() + lines = ["Image generation is unavailable in this environment.", "", "Missing requirements:"] + if managed: + lines.append(" - FAL_KEY is not set and the managed FAL gateway is unreachable") else: lines.append(" - FAL_KEY environment variable is not set") - gateway_message = nous_tool_gateway_unavailable_message( - "managed FAL image generation", - ) + gateway_message = nous_tool_gateway_unavailable_message("managed FAL image generation") if gateway_message: lines.append(f" - {gateway_message}") - lines.append("") - lines.append("To enable image generation, do one of:") - lines.append( + lines += [ + "", + "To enable image generation, do one of:", " 1. Get a free API key at https://fal.ai and set " - "FAL_KEY= (then restart the session)" - ) - if managed_nous_tools_enabled(): + "FAL_KEY= (then restart the session)", + ] + if managed: lines.append( " 2. Sign in to a Nous account that has the managed FAL " "gateway enabled (`hermes setup`)" @@ -780,7 +641,7 @@ def _build_no_backend_setup_message() -> str: def _get_plugin_provider(name: str): - """Discover plugins (import is local so importing this module never triggers discovery) and return the named provider.""" + """Discover plugins (local import: importing this module must not trigger discovery) and return the named provider.""" from agent.image_gen_registry import get_provider from hermes_cli.plugins import _ensure_plugins_discovered @@ -792,8 +653,7 @@ def check_image_generation_requirements() -> bool: """True if FAL or the explicitly configured image backend is available.""" try: if check_fal_api_key(): - # The lazy import doubles as the SDK presence check: ImportError - # when ``fal-client`` isn't installed falls through to plugin probing. + # Lazy import doubles as the SDK presence check: ImportError falls through to plugins. _load_fal_client() return True except ImportError: @@ -803,8 +663,7 @@ def check_image_generation_requirements() -> bool: if not configured or configured in ("fal", NOUS_MANAGED_PROVIDER): return False - # Probe only the explicitly selected plugin. Merely possessing a cloud - # provider key must not opt a user into a paid image-generation backend. + # Probe only the selected plugin: a cloud key alone must not opt a user into a paid backend. try: provider = _get_plugin_provider(configured) return bool(provider and provider.is_available()) @@ -819,13 +678,10 @@ from tools.registry import registry, tool_error IMAGE_GENERATE_SCHEMA = { "name": "image_generate", - # Placeholder — description AND params are rebuilt dynamically at - # get_tool_definitions() time from the active backend's declared - # capabilities (FAL catalog metadata, or plugin provider.capabilities()). - # Edit-only args (image_url, reference_image_urls) and upscale are - # advertised ONLY when the active model actually supports them; the - # handler accepts them regardless (replay compat + teaching errors). - # See _build_dynamic_image_schema(). + # Placeholder: description AND params are rebuilt at get_tool_definitions() time by + # _build_dynamic_image_schema() from the active backend's capabilities. Edit-only args + # and upscale are advertised ONLY when supported; the handler accepts them regardless + # (replay compat + teaching errors). "description": ( "Generate images from text prompts. The active model's edit/reference " "capabilities are rendered at serving time." @@ -847,8 +703,7 @@ IMAGE_GENERATE_SCHEMA = { "description": "The aspect ratio of the generated image. 'landscape' is 16:9 wide, 'portrait' is 16:9 tall, 'square' is 1:1.", "default": DEFAULT_ASPECT_RATIO, }, - # image_url / reference_image_urls / upscale are added per-capability - # by _build_dynamic_image_schema. Do not re-add them statically. + # image_url / reference_image_urls / upscale are added per-capability; never statically. }, "required": ["prompt"], }, @@ -860,12 +715,7 @@ IMAGE_GENERATE_SCHEMA = { # --------------------------------------------------------------------------- def _provider_error(error: str, error_type: str) -> str: """JSON error envelope shared by every provider-dispatch failure path.""" - return json.dumps({ - "success": False, - "image": None, - "error": error, - "error_type": error_type, - }) + return json.dumps({"success": False, "image": None, "error": error, "error_type": error_type}) def _add_provider_kwargs( @@ -880,13 +730,12 @@ def _add_provider_kwargs( kwargs["model"] = model if isinstance(image_url, str) and image_url.strip(): kwargs["image_url"] = image_url.strip() - norm_refs = None if reference_image_urls is not None: from agent.image_gen_provider import normalize_reference_images norm_refs = normalize_reference_images(reference_image_urls) - if norm_refs: - kwargs["reference_image_urls"] = norm_refs + if norm_refs: + kwargs["reference_image_urls"] = norm_refs if upscale is not None: kwargs["upscale"] = bool(upscale) return kwargs @@ -899,23 +748,15 @@ def _dispatch_to_plugin_provider( reference_image_urls: Optional[list] = None, upscale: Optional[bool] = None, ): - """Route the call to a plugin-registered provider when one is selected. + """JSON result from the selected plugin provider, or ``None`` to fall through to in-tree FAL. - Returns a JSON string on dispatch, or ``None`` to fall through to the - in-tree FAL pipeline. Fires when ``image_gen.provider`` is set to anything - other than unset / ``"fal"`` / ``"nous"`` — those run the legacy pipeline - (``"nous"`` routes it through the managed fal-queue gateway). - - ``image_url`` / ``reference_image_urls`` are forwarded so the backend can - route to its edit endpoint; ``upscale`` requests a post-generation - high-res pass (providers without it ignore it via ``**kwargs``). + Fires when ``image_gen.provider`` is anything but unset / ``"fal"`` / ``"nous"`` (those run + the legacy pipeline; ``"nous"`` via the managed fal-queue gateway). Edit args are + forwarded for the backend's edit endpoint; providers without ``upscale`` ignore it via ``**kwargs``. """ configured = _read_configured_image_provider() if not configured or configured in ("fal", NOUS_MANAGED_PROVIDER): return None - - configured_model = _read_configured_image_model() - try: from hermes_cli.plugins import _ensure_plugins_discovered @@ -925,10 +766,9 @@ def _dispatch_to_plugin_provider( return None if provider is None: + # Long-lived sessions may have discovered plugins before a bundled backend + # was patched in or config changed: retry once with a forced refresh. try: - # Long-lived sessions may have discovered plugins before a bundled - # backend was patched in or config changed: retry once with a forced - # refresh before surfacing a missing-provider error. from agent.image_gen_registry import get_provider _ensure_plugins_discovered(force=True) @@ -947,14 +787,11 @@ def _dispatch_to_plugin_provider( pname = getattr(provider, "name", "?") kwargs: Dict[str, Any] = {"prompt": prompt, "aspect_ratio": aspect_ratio} try: - _add_provider_kwargs( - kwargs, image_url, reference_image_urls, upscale, model=configured_model, - ) + _add_provider_kwargs(kwargs, image_url, reference_image_urls, upscale, model=_read_configured_image_model()) result = provider.generate(**kwargs) except TypeError as exc: - # A provider whose generate() predates image_url support (third-party - # plugin not yet updated): text-to-image keeps working, but surface a - # clear note when the user actually asked for an edit. + # generate() predating image_url support (third-party plugin not yet updated): + # text-to-image keeps working; surface a clear note when an edit was requested. if "image_url" in kwargs or "reference_image_urls" in kwargs: logger.warning( "image_gen provider '%s' rejected image-to-image kwargs " @@ -979,20 +816,15 @@ def _dispatch_to_plugin_provider( return json.dumps(result) -# Native ``krea-2-*`` plugin model ids are served by the dedicated Krea managed -# gateway; ``fal-ai/krea/v2/*`` catalog ids stay on the FAL path. Routing only -# fires in managed mode — direct/BYO users keep their unchanged pipeline. +# Native ``krea-2-*`` ids are served by the Krea managed gateway (managed mode only — +# direct/BYO users keep their pipeline); ``fal-ai/krea/v2/*`` catalog ids stay on FAL. _KREA_NATIVE_MODELS = {"krea-2-medium", "krea-2-large", "krea-2-medium-turbo"} def _normalize_krea_model(model_id: Optional[str]) -> Optional[str]: """Return the native Krea plugin model id when ``model_id`` is ``krea-2-*``.""" - if not isinstance(model_id, str): - return None - candidate = model_id.strip() - if candidate in _KREA_NATIVE_MODELS: - return candidate - return None + candidate = model_id.strip() if isinstance(model_id, str) else None + return candidate if candidate in _KREA_NATIVE_MODELS else None def _maybe_route_managed_krea( @@ -1002,13 +834,11 @@ def _maybe_route_managed_krea( reference_image_urls: Optional[list] = None, upscale: Optional[bool] = None, ) -> Optional[str]: - """Route a native ``krea-2-*`` model to the managed Krea gateway, in managed mode. + """JSON result from the managed Krea gateway, or ``None`` to fall through. - Returns a JSON result string when handled, or ``None`` to fall through to - the normal plugin/FAL pipeline. Fires only when the configured model is a - native ``krea-2-*`` id AND no explicit ``image_gen.provider`` other than the - managed ``"nous"`` selection is stored (a picker choice dispatches normally) - AND the managed Krea gateway is resolvable. + Fires only when the configured model is a native ``krea-2-*`` id AND no + ``image_gen.provider`` other than ``"nous"`` is stored (a picker choice dispatches + normally) AND the managed Krea gateway is resolvable. """ configured_provider = _read_configured_image_provider() if configured_provider is not None and configured_provider != NOUS_MANAGED_PROVIDER: @@ -1026,7 +856,6 @@ def _maybe_route_managed_krea( except Exception as exc: # noqa: BLE001 logger.debug("Managed Krea routing probe failed: %s", exc) return None - try: provider = _get_plugin_provider("krea") except Exception as exc: # noqa: BLE001 @@ -1050,15 +879,12 @@ def _maybe_route_managed_krea( def _confine_source_images( image_url, reference_image_urls, task_id, *, permitted: tuple = ("image",) ): - """Route path-like source images through the sandbox-aware resolver. - - Under a non-local terminal backend (ssh/docker/…), model-supplied local - paths resolve via ``tools.image_source`` (in-sandbox exec-read, media-cache - host reads, credential guard) into ``data:`` URLs before any provider sees - them, so generation obeys the same confinement boundary as vision/video - analysis and sandbox-only files work as edit sources. URLs and data: URLs - pass through; the local backend is a no-op (providers keep host reads). + """Resolve path-like sources to ``data:`` URLs under a non-local terminal backend. + Goes through ``tools.image_source`` (in-sandbox exec-read, media-cache host reads, + credential guard) before any provider sees them, so generation obeys the same + confinement boundary as vision/video analysis and sandbox-only files work as edit + sources. URLs/data: pass through; the local backend is a no-op (providers keep host reads). Returns ``(image_url, reference_image_urls, error_json_or_None)``. """ backend = (os.getenv("TERMINAL_ENV") or "local").strip().lower() @@ -1068,16 +894,14 @@ def _confine_source_images( from model_tools import _run_async from tools.image_source import ImageResolutionError, resolve_local_source_to_data_url + def resolve(ref): + return _run_async(resolve_local_source_to_data_url(ref, task_id, permitted=permitted)) + try: if isinstance(image_url, str) and image_url.strip(): - image_url = _run_async(resolve_local_source_to_data_url( - image_url, task_id, permitted=permitted)) + image_url = resolve(image_url) if isinstance(reference_image_urls, (list, tuple)): - reference_image_urls = [ - _run_async(resolve_local_source_to_data_url(ref, task_id, permitted=permitted)) - if isinstance(ref, str) else ref - for ref in list(reference_image_urls) - ] + reference_image_urls = [resolve(r) if isinstance(r, str) else r for r in reference_image_urls] except ImageResolutionError as exc: return image_url, reference_image_urls, _provider_error( f"Could not read source image: {exc}", type(exc).__name__, @@ -1090,75 +914,51 @@ def _handle_image_generate(args, **kw): if not prompt: return tool_error("prompt is required for image generation") aspect_ratio = args.get("aspect_ratio", DEFAULT_ASPECT_RATIO) - image_url = args.get("image_url") - reference_image_urls = args.get("reference_image_urls") upscale = args.get("upscale") if not isinstance(upscale, bool): upscale = None task_id = kw.get("task_id") - # Confinement chokepoint: path-like sources become data: URLs BEFORE any - # dispatch, so plugin, managed Krea and in-tree FAL all get sandbox-confined bytes. + # Confinement chokepoint BEFORE any dispatch: plugin, managed Krea and in-tree FAL + # all receive sandbox-confined bytes. image_url, reference_image_urls, confine_error = _confine_source_images( - image_url, reference_image_urls, task_id) + args.get("image_url"), args.get("reference_image_urls"), task_id) if confine_error is not None: return confine_error # Order matters: explicit plugin provider (incl. provider == "krea"), then # model-driven managed Krea interception (only when no provider is set, so # the BYO/direct FAL path stays untouched), then the in-tree FAL pipeline. - raw = _dispatch_to_plugin_provider( - prompt, aspect_ratio, - image_url=image_url, - reference_image_urls=reference_image_urls, - upscale=upscale, - ) - if raw is None: - raw = _maybe_route_managed_krea( - prompt, aspect_ratio, - image_url=image_url, - reference_image_urls=reference_image_urls, - upscale=upscale, - ) - if raw is None: - raw = image_generate_tool( - prompt=prompt, - aspect_ratio=aspect_ratio, - image_url=image_url, - reference_image_urls=reference_image_urls, - upscale=upscale, - ) + sources = dict(image_url=image_url, reference_image_urls=reference_image_urls, upscale=upscale) + raw = None + for route in (_dispatch_to_plugin_provider, _maybe_route_managed_krea, image_generate_tool): + raw = route(prompt, aspect_ratio, **sources) + if raw is not None: + break return _postprocess_image_generate_result(raw, task_id=task_id) # --------------------------------------------------------------------------- # Dynamic schema — reflect the active backend's image-to-image capability # --------------------------------------------------------------------------- -# Whether the active model can edit depends on the configured backend + model; -# telling the model up front saves a wasted turn. Memoized by config.yaml mtime -# in model_tools.get_tool_definitions(), so it rebuilds on provider/model switch. +# Telling the model up front whether it can edit saves a wasted turn. Memoized by +# config.yaml mtime in model_tools.get_tool_definitions(), so it rebuilds on switch. def _active_image_capabilities() -> Dict[str, Any]: """Best-effort capabilities of the active backend/model; never raises. - Resolution mirrors runtime dispatch: a set ``image_gen.provider`` asks that - plugin, otherwise the in-tree FAL catalog. Fail-closed on every axis: an - undeclared capability is advertised as absent (an under-declaring provider - is that provider's bug, not a safety problem). + Mirrors runtime dispatch: a set ``image_gen.provider`` asks that plugin, otherwise the + FAL catalog. Fail-closed: an undeclared capability is advertised as absent (an + under-declaring provider is that provider's bug, not a safety problem). """ - info: Dict[str, Any] = { - "modalities": ["text"], - "max_reference_images": 0, - "supports_upscale": False, - } + info: Dict[str, Any] = {"modalities": ["text"], "max_reference_images": 0, "supports_upscale": False} configured_provider = _read_configured_image_provider() if configured_provider and configured_provider != "fal": try: provider = _get_plugin_provider(configured_provider) if provider is not None: - caps = {} try: caps = provider.capabilities() or {} except Exception: # noqa: BLE001 @@ -1178,16 +978,13 @@ def _active_image_capabilities() -> Dict[str, Any]: # In-tree FAL path (provider unset or == "fal"). try: model_id, meta = _resolve_fal_model() + can_edit = bool(meta.get("edit_endpoint")) info["provider"] = "FAL.ai" info["model"] = meta.get("display", model_id) - if meta.get("edit_endpoint"): - info["modalities"] = ["text", "image"] - info["max_reference_images"] = int(meta.get("max_reference_images") or 1) - else: - info["modalities"] = ["text"] - info["max_reference_images"] = 0 - # FAL: Clarity is a separate endpoint chained on explicit request for ANY - # catalog model (the per-model ``upscale`` key is only the default flag). + info["modalities"] = ["text", "image"] if can_edit else ["text"] + info["max_reference_images"] = int(meta.get("max_reference_images") or 1) if can_edit else 0 + # Clarity is a separate endpoint available on request for ANY catalog model + # (the per-model ``upscale`` key is only the default flag). info["supports_upscale"] = True except Exception: # noqa: BLE001 pass @@ -1217,11 +1014,8 @@ _UPSCALE_PARAM = { def _build_dynamic_image_schema() -> Dict[str, Any]: - """Render description AND params from the active model's capabilities. - - Args a model cannot honor are NOT advertised — the handler still accepts - them (replay compat) and answers with a capability error. - """ + """Render description AND params from the active model's capabilities; args it cannot + honor are NOT advertised (the handler still accepts them for replay compat).""" base_desc = ( "Generate high-quality images from text prompts{edit_clause}. " "Returns the result in the `image` field — a URL or an absolute " @@ -1232,22 +1026,17 @@ def _build_dynamic_image_schema() -> Dict[str, Any]: try: info = _active_image_capabilities() except Exception: # noqa: BLE001 - info = {"modalities": ["text"], "max_reference_images": 0, - "supports_upscale": False} + info = {"modalities": ["text"], "max_reference_images": 0, "supports_upscale": False} - modalities = set(info.get("modalities") or ["text"]) max_refs = int(info.get("max_reference_images") or 0) - can_edit = "image" in modalities - + can_edit = "image" in set(info.get("modalities") or ["text"]) + static_props = IMAGE_GENERATE_SCHEMA["parameters"]["properties"] properties: Dict[str, Any] = { - "prompt": IMAGE_GENERATE_SCHEMA["parameters"]["properties"]["prompt"], - "aspect_ratio": IMAGE_GENERATE_SCHEMA["parameters"]["properties"]["aspect_ratio"], + "prompt": static_props["prompt"], "aspect_ratio": static_props["aspect_ratio"], } if can_edit: - edit_clause = ( - ", or edit / transform an existing image by passing image_url" - ) + edit_clause = ", or edit / transform an existing image by passing image_url" properties["image_url"] = _IMAGE_URL_PARAM if max_refs > 1: properties["reference_image_urls"] = { @@ -1261,18 +1050,13 @@ def _build_dynamic_image_schema() -> Dict[str, Any]: ), } else: - edit_clause = ( - " (text-to-image only — the active model cannot edit existing " - "images)" - ) + edit_clause = " (text-to-image only — the active model cannot edit existing images)" if info.get("supports_upscale"): properties["upscale"] = _UPSCALE_PARAM - description = base_desc.format(edit_clause=edit_clause) - return { - "description": description, + "description": base_desc.format(edit_clause=edit_clause), "parameters": { "type": "object", "properties": properties, diff --git a/tools/image_source.py b/tools/image_source.py index 8e60c0812b..3158d32b1e 100644 --- a/tools/image_source.py +++ b/tools/image_source.py @@ -2,33 +2,24 @@ All source handling (data:/http(s)/file/local/container) funnels through :func:`resolve_image_source` so size and magic-byte checks are enforced exactly -once. Returns raw bytes (not a path): the downstream step is base64 -> data URL -(RFC 2397) and provider base64 content blocks. +once. Returns raw bytes (not a path): the downstream step is base64 -> data URL. -Images are the default and the historical purpose. Callers whose argument -takes video opt in via ``permitted=("video",)`` — the same confinement and -credential-guard pipeline applies, and only the type check at the end differs -(extension-table typing plus an mp4 magic sniff, rather than image magic -bytes). Every existing call site keeps the image-only default unchanged. +Images are the default. Callers whose argument takes video opt in via +``permitted=("video",)``: same confinement and credential-guard pipeline, only +the final type check differs (extension table + mp4 magic sniff). Security (terminal-backend confinement, GHSA-gpxw-6wxv-w3qq): under a non-local -terminal backend the file tools are confined to the sandbox (SECURITY.md 2.2), -but vision read images host-side. This resolver enforces the same boundary: +backend the file tools are confined to the sandbox, so vision must be too: - * local backend -> read any host path (chosen posture, unchanged) + * local backend -> read any host path (chosen posture) * non-local backend: - path in a media cache -> host-read (the gateway/download caches live on - the host and are bind-mounted into the sandbox) - path anywhere else -> read the bytes *inside the sandbox* via exec-read - (the agent can already ``cat`` any container file; - this stays within the sandbox boundary and never - reaches the host's ``/etc/passwd`` / ``~/.ssh``). + path in a media cache -> host-read (gateway/download caches live on the + host and are bind-mounted into the sandbox) + path anywhere else -> read *inside the sandbox* via exec-read (the + agent can already ``cat`` any container file) So a prompt-injected ``vision_analyze('/etc/passwd')`` under Docker reads the -*container's* file (what every other tool sees), not the host's — no escape — -while container-only images (tmpfs ``/workspace``, root-owned) are still -deliverable. This is the unified delivery + confinement model: the same -mechanism that fixes "vision can't see container files" also closes the escape. +container's file, never the host's, while container-only images stay deliverable. """ from __future__ import annotations @@ -40,12 +31,9 @@ from dataclasses import dataclass from pathlib import Path from typing import Optional -# Raw-bytes INGEST budget — what the resolver will load before handing off. -# This is deliberately the 50MB download cap (tools/vision_tools._VISION_MAX_DOWNLOAD_BYTES), -# NOT the 20MB provider payload cap. The 20MB cap (_MAX_BASE64_BYTES) is a -# *post-resize* limit enforced at the call sites: an oversized raw image must -# still reach the resizer so it can be downscaled under the payload cap. Capping -# raw bytes at 20MB here would reject every 20-50MB photo before resize can run. +# Raw-bytes INGEST budget: deliberately the 50MB download cap, NOT the 20MB +# provider payload cap — that one is enforced post-resize at the call sites, and +# a 20-50MB photo must still reach the resizer. _MAX_INGEST_BYTES = 50 * 1024 * 1024 @@ -87,8 +75,7 @@ class ResolvedImage: origin: str # one of: data | http | file | local | container -# Explicit URL scheme, e.g. "ftp://", "s3://". Bare Windows drive paths -# ("C:\x.png") don't match because they lack the "//". +# Explicit URL scheme ("ftp://", "s3://"). Bare Windows drive paths lack the "//". _SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*://") @@ -118,23 +105,15 @@ async def resolve_image_source( ) # Everything else is a filesystem path — including bare relative names - # like "pic.png" (accepted on main; a path-shape gate here regressed them). + # like "pic.png" (a path-shape gate here regressed them once). candidate = s[len("file://"):] if s.lower().startswith("file://") else s p = Path(os.path.expanduser(candidate)) - # Confinement decision (see module docstring). Under a non-local backend - # a path is host-readable ONLY if it lands in a media cache (after - # translating a container-visible cache path back to its host mount); - # every other path is read inside the sandbox via exec-read, so a host - # path outside the caches never yields the host's bytes. host_target = _permitted_host_read_target(p, ctx) if host_target is not None and host_target.is_file(): - # Shared credential-read guard (agent.file_safety, #57698): refuse - # secret-bearing files (.env, auth.json, ...) with an intentional, - # specific error instead of relying on the magic-byte sniff to - # reject them incidentally. Same chokepoint the image-gen/video-gen - # provider plugins enforce on model-supplied local paths. Import is - # best-effort (guard unavailability must not break image loading); - # a real block always propagates. + # Shared credential-read guard: refuse secret-bearing files (.env, + # auth.json) with a specific error rather than relying on the magic + # sniff to reject them incidentally. Guard import is best-effort; a + # real block always propagates. try: from agent.file_safety import raise_if_read_blocked except Exception: # noqa: BLE001 — guard unavailable: proceed @@ -147,12 +126,8 @@ async def resolve_image_source( data = await asyncio.to_thread(host_target.read_bytes) return _finalize(data, "", "file", s, permitted) if _is_local_terminal_backend(): - # Local backend: any path was host-readable, so a miss simply means - # the file doesn't exist — no sandbox to fall back to. + # Any path was host-readable, so a miss means the file doesn't exist. raise SourceNotFound(f"media file not found: '{p}'", src=s, origin="file") - # Not a permitted host read (or the host file is absent) -> read the - # bytes inside the sandbox. Under a sandbox this reads the container's - # filesystem, never the host's. return await _resolve_container_fallback(p, ctx, s, permitted) @@ -172,15 +147,9 @@ def _resolve_data_url(s: str) -> tuple[bytes, str]: def _http_block_reason(url: str) -> Optional[str]: - """Return a human-readable block reason, or None when the URL is allowed. - - Pre-flight short-circuit: policy-blocked URLs are refused BEFORE any - network I/O. ``_download_image`` re-checks policy internally (per attempt - and against the final redirect target) — that second evaluation is - intentional, not redundant: this one guarantees no bytes move for a - blocked URL; the inner one covers redirects and non-resolver callers. - Preserves the specific website-policy message so the agent sees *why*. - """ + """Block reason, or None when allowed. Refuses policy-blocked URLs BEFORE any + network I/O; ``_download_image`` re-checks per attempt and against the final + redirect target — the second evaluation is intentional, not redundant.""" from tools.url_safety import is_safe_url from tools.website_policy import check_website_access @@ -210,27 +179,19 @@ async def _download_to_bytes(url: str) -> bytes: def _is_local_terminal_backend() -> bool: - """True when the terminal backend runs directly on the host. - - Mirrors ``tools.browser_tool._is_local_backend`` and terminal_tool's own - dispatch, which key off ``TERMINAL_ENV``. - """ + """True when the terminal backend runs directly on the host (keys off ``TERMINAL_ENV``).""" return os.getenv("TERMINAL_ENV", "local").strip().lower() in ("local", "") def _media_cache_roots() -> list: - """Agent-managed media cache directories under HERMES_HOME (host side). - - The only host paths vision may read under a non-local backend: gateway- - downloaded inbound media and the tools' own URL-download temp dirs. Covers - the consolidated ``cache/`` layout and the legacy flat directories. - """ + """Host-side media caches: the only host paths vision may read under a + non-local backend (gateway inbound media + the tools' own download temp dirs).""" from hermes_constants import get_hermes_home home = get_hermes_home() return [ home / "cache", # cache/images, cache/vision, cache/video(s), cache/audio - home / "images", # desktop/clipboard/PDF uploads (tui_gateway) — #69575 + home / "images", # desktop/clipboard/PDF uploads (tui_gateway) home / "image_cache", home / "audio_cache", home / "video_cache", @@ -240,14 +201,12 @@ def _media_cache_roots() -> list: def _permitted_host_read_target(p: Path, ctx: ResolveContext) -> Optional[Path]: - """Return the host path to read, or ``None`` if a host read is not permitted. + """Host path to read, or ``None`` if a host read is not permitted. - - Local backend: any path is permitted (chosen posture). Returns ``p``. - - Non-local backend: permitted only if the path resolves inside a media - cache root. A container-visible cache path (e.g. ``/root/.hermes/cache/ - images/x.png``) is first translated back to its host mount; anything that - is not under a cache returns ``None`` so the caller routes it to the - in-sandbox exec-read instead of reading the host filesystem. + Local backend: any path. Non-local: only paths resolving inside a media + cache root (a container-visible cache path is first translated back to its + host mount); anything else returns ``None`` so the caller exec-reads inside + the sandbox instead of touching the host filesystem. """ if _is_local_terminal_backend(): try: @@ -283,13 +242,12 @@ def _get_active_env(task_id: Optional[str]): def _ensure_container_env(task_id: Optional[str]) -> None: - """Lazily bring up the sandbox (SSH/Docker/…) before an in-sandbox read. + """Lazily bring up the sandbox before an in-sandbox read. - Unlike the terminal tool, vision never triggered environment creation, so a - session whose first action is ``vision_analyze`` on a container-only path - under a non-local backend found no active env and failed — until a terminal - command happened to create one (issue #62825). Best-effort: any failure just - leaves the env absent and the caller hits the existing fail-closed error. + Vision never triggered environment creation, so a session whose first action + was ``vision_analyze`` on a container-only path found no active env until a + terminal command created one. Best-effort: failure leaves the env absent and + the caller hits the fail-closed error. """ if not task_id: return @@ -304,35 +262,17 @@ def _ensure_container_env(task_id: Optional[str]) -> None: async def _resolve_container_fallback( p: Path, ctx: ResolveContext, src: str, permitted: tuple = ("image",) ) -> ResolvedImage: - """Read the image bytes inside the sandbox (fail-closed when none exists). + """Read the bytes inside the sandbox; fail-closed when no env exists (a + non-cache host path under a sandbox must never leak via a host fallback). - Reached when a host read is not permitted or the host file is absent. The - agent can already ``cat`` any container file (file_operations.py reads - root-owned mode-600 files this way), so this stays within the same sandbox - boundary and never touches the host filesystem. ``--`` stops a leading-dash - path from being parsed as a ``base64`` option; ``base64 -w0`` is GNU-only, - so pipe through ``tr -d`` for BusyBox. - - Fail-closed: if there is no active sandbox env we refuse rather than falling - back to a host read, so a non-cache host path under a sandbox never leaks. - - Cold-start retry: under Docker the very first exec against a freshly - started container can fail (empty pipe / partial setup) while an identical - second call succeeds. We retry once with a short delay before giving up, - so callers don't see "could not read inside the sandbox" on a file that is - verifiably readable on the immediate retry. See #76566. - - Diagnostic: when every attempt fails, the container's own output (stderr - + stdout) is folded into the raised error so the user can distinguish - "no such file" from "permission denied" from "container never came up" - instead of staring at one opaque message. + Cold-start retry: under Docker the first exec against a fresh container can + fail (empty pipe) while an identical second call succeeds, so retry once + after a short delay. On final failure the container's own output is folded + into the error so "no such file" / "permission denied" / "never came up" + are distinguishable. """ - import asyncio import shlex - # Bring the sandbox up on demand: without this, the first vision_analyze of - # a session (before any terminal command) has no active env to read from - # under a non-local backend (issue #62825). _ensure_container_env(ctx.task_id) env = _get_active_env(ctx.task_id) @@ -342,14 +282,11 @@ async def _resolve_container_fallback( f"session is available to read it", src=src, origin="container") - # Bound the read INSIDE the sandbox: head -c caps at ingest-limit+1 bytes - # so a huge file (or /dev/zero) can't stream unbounded base64 into host - # memory — the +1 byte lets us distinguish "exactly at the cap" from - # "over the cap" after decode. The input redirect (< path) avoids argv - # entirely, so leading-dash paths can't be parsed as options; base64 - # -w0 is GNU-only, so pipe through tr -d for BusyBox. - # env.execute is a blocking backend exec; keep it off the event loop so a - # multi-MB base64 read doesn't stall every other coroutine. + # Bound the read INSIDE the sandbox: head -c caps at ingest-limit+1 so + # /dev/zero can't stream unbounded base64 into host memory (the +1 + # distinguishes "at the cap" from "over"). The input redirect avoids argv, so + # leading-dash paths can't parse as options; base64 -w0 is GNU-only, hence + # tr -d for BusyBox. env.execute blocks — keep it off the event loop. qp = shlex.quote(str(p)) cmd = f"head -c {_MAX_INGEST_BYTES + 1} < {qp} | base64 | tr -d '\\n'" @@ -359,14 +296,9 @@ async def _resolve_container_fallback( if last_res.get("returncode", 1) == 0: break if attempt == 0: - # Cold-start: give the container a moment to settle its pipes - # before retrying. 150ms covers Docker exec warm-up in practice - # without making a real failure feel sluggish. - await asyncio.sleep(0.15) + await asyncio.sleep(0.15) # covers Docker exec warm-up in practice if last_res.get("returncode", 1) != 0: diag = (last_res.get("output") or "").strip().splitlines() - # Keep the diagnostic small and noise-free: first non-empty line, - # trimmed to a sane length so it slots into the agent's error UI. first = next((ln.strip() for ln in diag if ln.strip()), "") suffix = f" ({first[:200]})" if first else "" raise SourceNotFound( @@ -384,18 +316,11 @@ async def _resolve_container_fallback( def _finalize( data: bytes, declared_mime: str, origin: str, src: str, permitted: tuple = ("image",) ) -> ResolvedImage: - """Intrinsic-correctness chokepoint: ingest byte cap + type check. + """Chokepoint: 50MB ingest cap + type check. - The cap here is the generous 50MB *ingest* budget, not the 20MB provider - payload cap — a 20-50MB image must survive this step so the call site can - resize it under the payload cap. See ``_MAX_INGEST_BYTES``. - - Images are typed by magic bytes. Video (opt-in via ``permitted``) is typed - by the extension table plus an mp4 container sniff: extension typing is - sufficient because every downstream consumer re-validates — the upload - gateway signs the content type into its presigned URL and the vendor - rejects undecodable input — so a wrong guess is a clean rejection there - rather than a hole here. + Images are typed by magic bytes. Video (opt-in) is typed by extension plus + an mp4 container sniff — sufficient because every downstream consumer + re-validates, so a wrong guess is a clean rejection there, not a hole here. """ from tools.vision_tools import _detect_image_mime_type_from_bytes @@ -409,9 +334,7 @@ def _finalize( return ResolvedImage(data=data, mime=sniffed, origin=origin) if "image" in permitted and b" Optional[str]: - """Video MIME from the extension table, else the mp4/mov container magic. - - The magic fallback covers extensionless sources (data: URLs, URLs with - query strings): ISO base-media files carry ``ftyp`` at offset 4. - """ + """Video MIME from the extension table, else the ISO base-media ``ftyp`` + magic at offset 4 (covers extensionless data: URLs / query-string URLs).""" from urllib.parse import urlsplit from tools.vision_tools import _detect_video_mime_type @@ -447,23 +367,11 @@ async def resolve_local_source_to_data_url( ) -> str: """Convert a path-like media source into a ``data:`` URL via the resolver. - Generation tools (image_generate / video_generate) forward model-supplied - source images to provider plugins, which historically read local paths off - the HOST filesystem regardless of terminal backend. Under a non-local - backend that is both broken (the file usually lives in the sandbox, so the - host read misses) and inconsistent with the confinement model vision/video - analysis enforce (GHSA-gpxw-6wxv-w3qq): the sandbox boundary should govern - every model-supplied path. - - This helper is the dispatch-layer chokepoint: URL-shaped sources - (http/https/data) pass through untouched; anything path-like resolves - through :func:`resolve_image_source` — media-cache host reads, bounded - in-sandbox exec-read, lazy env bring-up, credential guard, ingest cap — - and comes back as a ``data:`` URL every provider already accepts. - - Callers apply this only under a non-local terminal backend: on the local - backend providers keep their existing host-side reads (chosen posture, - zero behavior change). + Dispatch-layer chokepoint for generation tools: providers historically read + model-supplied local paths off the HOST regardless of backend — broken under + a sandbox (the file lives there) and inconsistent with the confinement model + vision enforces. URL-shaped sources (http/https/data) pass through untouched. + Callers apply this only under a non-local backend (local keeps host reads). """ s = (src or "").strip() if not s or s.lower().startswith(("http://", "https://", "data:")): diff --git a/tools/video_generation_tool.py b/tools/video_generation_tool.py index ee0170c97d..287f21dfa3 100644 --- a/tools/video_generation_tool.py +++ b/tools/video_generation_tool.py @@ -1,28 +1,11 @@ #!/usr/bin/env python3 -""" -Video Generation Tool -===================== +"""``video_generate``: one tool dispatching to a plugin-registered :class:`VideoGenProvider` +(``agent/video_gen_provider.py`` ABC, ``agent/video_gen_registry.py``, ``plugins/video_gen//``). -Single ``video_generate`` tool that dispatches to a plugin-registered -video generation provider. Mirrors the ``image_generate`` design: - -- ``agent/video_gen_provider.py`` defines the :class:`VideoGenProvider` ABC. -- ``agent/video_gen_registry.py`` holds the active providers (populated by - plugins at import time). -- Each provider lives under ``plugins/video_gen//``. - -The tool is backend-agnostic and ships **no in-tree provider** — enable a -plugin (``hermes plugins enable video_gen/``) and select it in -``hermes tools`` → Video Generation. - -One tool covers text-to-video, image-to-video and reference-to-video with a -compact schema (prompt, image_url, reference_image_urls, duration, -aspect_ratio, resolution, negative_prompt, audio, seed, model). Providers -ignore parameters they do not support: the tool layer does only lightweight -validation (type/required-prompt) and each provider clamps inside -:meth:`VideoGenProvider.generate`, so the surface stays stable as providers -with different capabilities ship. Video edit/extend are intentionally not -exposed here; providers with those workflows expose separate tools. +Ships **no in-tree provider**: enable a plugin and select it in ``hermes tools`` → Video +Generation. Covers text-, image- and reference-to-video; the tool layer only does lightweight +validation and each provider clamps/ignores unsupported params inside ``generate``, so the +surface stays stable as providers ship. Video edit/extend are deliberately not exposed here. """ from __future__ import annotations @@ -45,13 +28,9 @@ logger = logging.getLogger(__name__) VIDEO_GENERATE_SCHEMA: Dict[str, Any] = { "name": "video_generate", - # Placeholder — description AND params are rebuilt dynamically at - # get_tool_definitions() time from the active provider's declared - # capabilities() and the active model's catalog entry. Optional args - # (image_url, reference_image_urls, negative_prompt, audio, seed, - # upscale) are advertised ONLY when the active backend/model honors - # them; the handler accepts them regardless (replay compat — providers - # clamp/ignore). See _build_dynamic_video_schema(). + # Placeholder: description AND params are rebuilt at get_tool_definitions() time by + # _build_dynamic_video_schema() from capabilities() + the model's catalog entry. Optional + # args are advertised ONLY when honored; the handler accepts them regardless (replay compat). "description": "(rebuilt at get_definitions() time — see _build_dynamic_video_schema)", "parameters": { "type": "object", @@ -89,9 +68,7 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = { "``video_gen.model``. Unknown models are rejected." ), }, - # image_url / reference_image_urls / negative_prompt / audio / seed / - # upscale are added per-capability by _build_dynamic_video_schema. - # Do not re-add them statically. + # Capability-gated args are added by _build_dynamic_video_schema; never statically. }, "required": ["prompt"], }, @@ -101,8 +78,6 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = { # --------------------------------------------------------------------------- # Config readers (mirror image_generation_tool.py) # --------------------------------------------------------------------------- - - def _read_video_gen_key(key: str) -> Optional[str]: """Return the stripped ``video_gen.`` string from config.yaml, or None.""" try: @@ -127,23 +102,23 @@ def _read_configured_video_model() -> Optional[str]: return _read_video_gen_key("model") +def _discovered_registry(): + """Import the provider registry after (idempotent) plugin discovery so user-installed plugins are visible.""" + from agent import video_gen_registry + from hermes_cli.plugins import _ensure_plugins_discovered + + _ensure_plugins_discovered() + return video_gen_registry, _ensure_plugins_discovered + + # --------------------------------------------------------------------------- # Availability check + provider resolution # --------------------------------------------------------------------------- - - def check_video_generation_requirements() -> bool: - """True when at least one registered provider reports available. - - Triggers plugin discovery (idempotent) so user-installed plugins are - visible to the toolset gate. - """ + """True when at least one registered provider reports available.""" try: - from agent.video_gen_registry import list_providers - from hermes_cli.plugins import _ensure_plugins_discovered - - _ensure_plugins_discovered() - for provider in list_providers(): + registry_mod, _ = _discovered_registry() + for provider in registry_mod.list_providers(): try: if provider.is_available(): return True @@ -155,20 +130,14 @@ def check_video_generation_requirements() -> bool: def _resolve_active_provider(): - """Return the active provider object or None. - - Forces a discovery refresh on a miss — handles long-lived sessions that - started before a plugin was installed. - """ + """Active provider or None; forces a discovery refresh on a miss (long-lived sessions + that started before a plugin was installed).""" try: - from agent.video_gen_registry import get_active_provider - from hermes_cli.plugins import _ensure_plugins_discovered - - _ensure_plugins_discovered() - provider = get_active_provider() + registry_mod, ensure_discovered = _discovered_registry() + provider = registry_mod.get_active_provider() if provider is None: - _ensure_plugins_discovered(force=True) - provider = get_active_provider() + ensure_discovered(force=True) + provider = registry_mod.get_active_provider() return provider except Exception as exc: logger.debug("video_gen provider resolution failed: %s", exc) @@ -177,30 +146,28 @@ def _resolve_active_provider(): def _missing_provider_error(configured: Optional[str]) -> str: if configured: - msg = ( - f"video_gen.provider='{configured}' is set but no plugin " - f"registered that name. Run `hermes plugins list` to see " - f"installed video gen backends, or `hermes tools` → Video " - f"Generation to pick one." - ) return json.dumps(error_response( - error=msg, error_type="provider_not_registered", + error=( + f"video_gen.provider='{configured}' is set but no plugin " + f"registered that name. Run `hermes plugins list` to see " + f"installed video gen backends, or `hermes tools` → Video " + f"Generation to pick one." + ), + error_type="provider_not_registered", provider=configured, )) - msg = ( - "No video generation backend is configured. Run `hermes tools` → " - "Video Generation to enable one (xAI, FAL, or Google Veo)." - ) return json.dumps(error_response( - error=msg, error_type="no_provider_configured", + error=( + "No video generation backend is configured. Run `hermes tools` → " + "Video Generation to enable one (xAI, FAL, or Google Veo)." + ), + error_type="no_provider_configured", )) # --------------------------------------------------------------------------- # Handler # --------------------------------------------------------------------------- - - def _coerce_int(value: Any) -> Optional[int]: if value is None or value == "": return None @@ -211,16 +178,11 @@ def _coerce_int(value: Any) -> Optional[int]: def _coerce_bool(value: Any) -> Optional[bool]: - if value is None: - return None if isinstance(value, bool): return value if isinstance(value, str): - v = value.strip().lower() - if v in {"true", "1", "yes", "on"}: - return True - if v in {"false", "0", "no", "off"}: - return False + return {"true": True, "1": True, "yes": True, "on": True, + "false": False, "0": False, "no": False, "off": False}.get(value.strip().lower()) return None @@ -241,8 +203,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str: reference_image_urls = _normalize_reference_images(args.get("reference_image_urls")) task_id = _kw.get("task_id") - # Confinement chokepoint (mirrors image_generate): under a non-local - # backend, path-like source images reach providers as data: URLs. + # Confinement chokepoint (mirrors image_generate): non-local backends hand providers data: URLs. from tools.image_generation_tool import _confine_source_images image_url, reference_image_urls, confine_error = _confine_source_images( @@ -258,8 +219,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str: upscale = _coerce_bool(args.get("upscale")) model_override = (args.get("model") or "").strip() or None - # Soft validation — providers do their own. The backend may accept - # image-only on its image-to-video endpoint, but our surface always needs a prompt. + # Soft validation — providers do their own; a backend may accept image-only, our surface never does. if not prompt: return tool_error("prompt is required for video generation") if "operation" in args or "video_url" in args: @@ -291,7 +251,6 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str: } # Drop None entries so providers see clean defaults. kwargs = {k: v for k, v in kwargs.items() if v is not None} - pname = getattr(provider, "name", "?") def _err(error: str, error_type: str) -> str: @@ -303,8 +262,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str: try: result = provider.generate(prompt=prompt, **kwargs) except TypeError as exc: - # A provider that hasn't widened its signature is a plugin bug, not a - # caller error — surface a clear contract message. + # An un-widened provider signature is a plugin bug, not a caller error. logger.warning( "video_gen provider '%s' rejected kwargs (signature too narrow): %s", pname, exc, @@ -328,12 +286,33 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str: # --------------------------------------------------------------------------- # Dynamic schema — reflect the active backend's actual capabilities # --------------------------------------------------------------------------- -# The configured backend determines which modalities, aspect ratios, -# resolutions, durations and audio/negative-prompt flags are real; surfacing -# the per-model surface in the description means the model usually gets the -# call right first try. model_tools.get_tool_definitions() keys its cache on -# config.yaml mtime, so the schema rebuilds when provider/model changes. +# Surfacing the per-model surface (modalities, enums, durations, audio/negative-prompt) +# means the model usually gets the call right first try. model_tools.get_tool_definitions() +# keys its cache on config.yaml mtime, so the schema rebuilds on provider/model change. +# Optional params advertised only when the provider's capabilities() sets the flag +# (order = schema property order). +_CAPABILITY_PARAMS = ( + ("supports_negative_prompt", "negative_prompt", { + "type": "string", + "description": "Content to avoid in the output.", + }), + ("supports_audio", "audio", { + "type": "boolean", + "description": "Enable native audio generation (affects pricing tier).", + }), + ("supports_seed", "seed", { + "type": "integer", + "description": "Seed for reproducible outputs.", + }), + ("supports_upscale", "upscale", { + "type": "boolean", + "description": ( + "High-resolution pass via the backend's video upscaler " + "(~2x, extra cost/latency). Omit for native resolution." + ), + }), +) _GENERIC_DESCRIPTION = ( "Generate a video from a text prompt (text-to-video), animate a " @@ -352,14 +331,16 @@ _GENERIC_DESCRIPTION = ( ) -def _build_dynamic_video_schema() -> Dict[str, Any]: - """Render description AND params from the active backend's declared surface. +def _schema(description: str, properties: Dict[str, Any]) -> Dict[str, Any]: + return { + "description": description, + "parameters": {"type": "object", "properties": properties, "required": ["prompt"]}, + } - Optional args are advertised only when the resolved provider/model honors - them (capabilities() + the model's catalog entry); enums and duration - bounds tighten to the active model's sets. The handler still accepts - unadvertised args (replay compat): providers clamp or ignore. - """ + +def _build_dynamic_video_schema() -> Dict[str, Any]: + """Render description AND params from capabilities() + the model's catalog entry; enums and + duration bounds tighten to the active model. Unadvertised args are still accepted (replay compat).""" static_props = VIDEO_GENERATE_SCHEMA["parameters"]["properties"] parts: List[str] = [_GENERIC_DESCRIPTION] @@ -371,14 +352,7 @@ def _build_dynamic_video_schema() -> Dict[str, Any]: "\nNo video backend is available. Calls will return an error " "until the user picks one via `hermes tools` → Video Generation." ) - return { - "description": "\n".join(parts), - "parameters": { - "type": "object", - "properties": {"prompt": static_props["prompt"]}, - "required": ["prompt"], - }, - } + return _schema("\n".join(parts), {"prompt": static_props["prompt"]}) try: caps = provider.capabilities() or {} @@ -388,17 +362,11 @@ def _build_dynamic_video_schema() -> Dict[str, Any]: models = provider.list_models() or [] except Exception: models = [] - active_model = configured_model or provider.default_model() - model_meta = next( - (m for m in models if isinstance(m, dict) and m.get("id") == active_model), - {}, - ) + model_meta = next((m for m in models if isinstance(m, dict) and m.get("id") == active_model), {}) - # ---- description ------------------------------------------------- - # Model caveats surface only what differs from the backend's overall - # capabilities. FAL's plugin uses the singular ``modality`` key for - # single-modality entries. + # Model caveats surface only what differs from the backend's overall capabilities. + # FAL's plugin uses the singular ``modality`` key for single-modality entries. model_modalities = set(model_meta.get("modalities") or []) modality = model_meta.get("modality") if modality: @@ -435,7 +403,6 @@ def _build_dynamic_video_schema() -> Dict[str, Any]: if notice: parts.append(f"- storage: {notice}") - # ---- params ------------------------------------------------------ properties: Dict[str, Any] = {"prompt": static_props["prompt"]} if can_i2v: @@ -477,48 +444,17 @@ def _build_dynamic_video_schema() -> Dict[str, Any]: param["enum"] = list(caps[caps_key]) properties[key] = param - if caps.get("supports_negative_prompt"): - properties["negative_prompt"] = { - "type": "string", - "description": "Content to avoid in the output.", - } - if caps.get("supports_audio"): - properties["audio"] = { - "type": "boolean", - "description": ( - "Enable native audio generation (affects pricing tier)." - ), - } - elif caps.get("audio_always_on"): + for flag, key, param in _CAPABILITY_PARAMS: + if caps.get(flag): + properties[key] = param + if caps.get("audio_always_on") and not caps.get("supports_audio"): parts.append( "- audio: native stereo audio is generated with every video " "(always on; no toggle) — describe the desired sound in the " "prompt" ) - if caps.get("supports_seed"): - properties["seed"] = { - "type": "integer", - "description": "Seed for reproducible outputs.", - } - if caps.get("supports_upscale"): - properties["upscale"] = { - "type": "boolean", - "description": ( - "High-resolution pass via the backend's video upscaler " - "(~2x, extra cost/latency). Omit for native resolution." - ), - } - properties["model"] = static_props["model"] - - return { - "description": "\n".join(parts), - "parameters": { - "type": "object", - "properties": properties, - "required": ["prompt"], - }, - } + return _schema("\n".join(parts), properties) registry.register( diff --git a/tools/vision_tools.py b/tools/vision_tools.py index f6c1852182..50dea3a1fc 100644 --- a/tools/vision_tools.py +++ b/tools/vision_tools.py @@ -1,35 +1,13 @@ #!/usr/bin/env python3 -""" -Vision Tools Module +"""Vision tools: ``vision_analyze`` (image) and ``video_analyze``. -This module provides vision analysis tools that work with image URLs. -Uses the centralized auxiliary vision router, which can select OpenRouter, -Nous, Codex, native Anthropic, or a custom OpenAI-compatible endpoint. - -Available tools: -- vision_analyze_tool: Analyze images from URLs with custom prompts - -Features: -- Downloads images from URLs and converts to base64 for API compatibility -- Comprehensive image description -- Context-aware analysis based on user queries -- Automatic temporary file cleanup -- Proper error handling and validation -- Debug logging support - -Usage: - from vision_tools import vision_analyze_tool - import asyncio - - # Analyze an image - result = await vision_analyze_tool( - image_url="https://example.com/image.jpg", - user_prompt="What architectural style is this building?" - ) +Images resolve through :mod:`tools.image_source`, are normalized to a +provider-supported format (:mod:`tools.vision_tools_image_prep`), then either +attach natively to a vision-capable main model (multimodal tool-result +envelope) or are described by the auxiliary vision LLM router. """ import base64 -import contextlib import asyncio import json from concurrent.futures import ThreadPoolExecutor @@ -42,10 +20,9 @@ from typing import Any, Awaitable, Dict, Optional from urllib.parse import urlparse import httpx -# ``agent.auxiliary_client`` pulls credential_pool → hermes_cli.auth → httpx -# → rich (~50 ms cold); only vision handlers need it. Loaded lazily; both -# names stay module attributes so tests can keep patching -# ``tools.vision_tools.async_call_llm``. Truthy-skip: injected mocks win. +# ``agent.auxiliary_client`` costs ~50 ms cold (credential_pool → auth → rich); +# only the handlers need it. Both names stay module attributes so tests can +# patch ``tools.vision_tools.async_call_llm``; truthy-skip means injected mocks win. async_call_llm: Any = None extract_content_or_reasoning: Any = None @@ -66,78 +43,83 @@ def _load_auxiliary_client() -> None: from hermes_constants import get_hermes_dir from tools.debug_helpers import DebugSession from tools.website_policy import check_website_access -import sys +from tools.vision_tools_image_prep import ( # noqa: F401 — re-exported for tests/image_source + _ANTHROPIC_SUPPORTED_MEDIA_TYPES, + _VISION_MAX_VALIDATED_AGGREGATE_PIXELS, + _VISION_MAX_VALIDATED_FRAME_COUNT, + _crop_image_region, + _detect_image_mime_type_from_bytes, + _determine_mime_type, + _image_exceeds_dimension, + _normalize_to_supported_image, + _rasterize_svg_to_png, + _supported_media_types, + _validate_raster_image_decodable, +) logger = logging.getLogger(__name__) _debug = DebugSession("vision_tools", env_var="VISION_TOOLS_DEBUG") -# Configurable HTTP download timeout for _download_image(). -# Separate from auxiliary.vision.timeout which governs the LLM API call. -# Resolution: config.yaml auxiliary.vision.download_timeout → env var → 30s default. -def _resolve_download_timeout() -> float: - env_val = os.getenv("HERMES_VISION_DOWNLOAD_TIMEOUT", "").strip() - if env_val: + +def _read_vision_setting(env_var: str, key: str, cast, minimum=None): + """Env var → config.yaml ``auxiliary.vision.`` → None. + + Values that fail ``cast`` or fall below ``minimum`` are skipped in favor of + the next source (a cap can never be disabled by a bad value). + """ + def _accept(raw): try: - return float(env_val) - except ValueError: - pass + val = cast(raw) + except (TypeError, ValueError): + return None + return val if minimum is None or val >= minimum else None + + env_val = os.getenv(env_var, "").strip() + if env_val: + val = _accept(env_val) + if val is not None: + return val try: from hermes_cli.config import cfg_get, load_config - cfg = load_config() - val = cfg_get(cfg, "auxiliary", "vision", "download_timeout") - if val is not None: - return float(val) + raw = cfg_get(load_config(), "auxiliary", "vision", key) + if raw is not None: + return _accept(raw) except Exception: pass - return 30.0 + return None + + +def _resolve_download_timeout() -> float: + """HTTP download timeout (separate from ``auxiliary.vision.timeout``, which governs the LLM call).""" + val = _read_vision_setting("HERMES_VISION_DOWNLOAD_TIMEOUT", "download_timeout", float) + return 30.0 if val is None else val + _VISION_DOWNLOAD_TIMEOUT = _resolve_download_timeout() -# Hard cap on downloaded image file size (50 MB). Prevents OOM from -# attacker-hosted multi-gigabyte files or decompression bombs. +# Hard cap on downloaded media (50 MB): bounds memory/disk against +# attacker-hosted multi-gigabyte files. _VISION_MAX_DOWNLOAD_BYTES = 50 * 1024 * 1024 # --------------------------------------------------------------------------- # CPU-burst concurrency cap (vision encode/resize) # --------------------------------------------------------------------------- -# A single agent turn can fan out N vision_analyze calls at once (the classic -# trigger is "analyze every frame of this video" — ffmpeg explodes a clip into -# dozens of frames, the model then calls vision_analyze on each). Each call does -# a CPU-heavy base64-encode + (sometimes) Pillow resize. The tool executor runs -# concurrent tool calls on a ThreadPoolExecutor (agent.tool_executor = -# 8 workers) PER SESSION, and several agent sessions share one process (the -# dashboard runs the agent in-process). Unbounded, a video-frame fan-out across -# one or more sessions runs *every* encode at once, saturates all cores, and -# leaves no CPU to service the shared asyncio event loop that serves the -# dashboard's /api/status liveness probe — so the instance flaps to UNHEALTHY -# even though nothing has crashed (observed in prod, June 2026). -# -# The fix is NOT to cap how many vision analyses run — multi-image workflows -# ("compare these 6 screenshots", "read this 10-page scan") legitimately want -# high concurrency, and the slow part (the LLM stream) is network-bound and -# harmless to the loop. We cap ONLY the CPU burst: the encode/resize is offloaded -# to a dedicated, bounded executor sized to the host's usable core count. That -# is the resource the incident actually exhausted (cores), so bounding it to -# cores is *correct*, not an arbitrary number — excess encodes queue on the -# executor instead of all running at once, the LLM calls stay fully concurrent, -# and the loop always keeps a core. No fixed ceiling: the limit tracks the host. -# -# A threading primitive (NOT asyncio) is required: each vision call is dispatched -# through model_tools._run_async on a PER-THREAD event loop, so an asyncio -# executor/semaphore bound to one loop cannot coordinate across them. A -# ThreadPoolExecutor is loop- and thread-agnostic. +# A turn can fan out dozens of vision_analyze calls ("analyze every frame"); +# each does a CPU-heavy base64 encode + Pillow resize. Sessions share one +# process, so unbounded encodes saturate every core and starve the shared event +# loop (the dashboard liveness probe flapped UNHEALTHY in prod). We cap ONLY the +# CPU burst — the LLM calls stay fully concurrent — on a dedicated executor sized +# to the usable core count (the resource actually exhausted; no fixed ceiling). +# It must be a threading primitive: each call runs via model_tools._run_async on +# a PER-THREAD event loop, so an asyncio semaphore cannot coordinate across them. +# The default executor is NOT used: it is shared with the gateway/web server. import threading # noqa: F401 (kept for downstream importers / patch targets) def _detect_host_cpus() -> int: - """Best-effort host CPU count, honoring cgroup/affinity limits when set. - - Prefers ``os.sched_getaffinity`` (the CPUs this process may actually run - on — respects container/cpuset pinning) and falls back to - ``os.cpu_count()``. Returns at least 1. - """ + """Usable CPU count (``sched_getaffinity`` honors cpuset pinning), at least 1.""" try: return max(1, len(os.sched_getaffinity(0))) # type: ignore[attr-defined] except (AttributeError, OSError): @@ -145,50 +127,13 @@ def _detect_host_cpus() -> int: def _resolve_vision_cpu_workers() -> int: - """Resolve how many vision encode/resize bursts may run concurrently. - - Defaults to the host's usable core count (``_detect_host_cpus``) — no fixed - ceiling, because the cap tracks the actual exhausted resource (CPU cores), - not a magic number. The LLM call is NOT covered by this limit, so legitimate - multi-image fan-out keeps full request concurrency; only the simultaneous - CPU bursts are bounded so the event loop always keeps a core. - - Resolution order: HERMES_VISION_MAX_CONCURRENCY env → - config.yaml auxiliary.vision.max_concurrency → host core count. Any value - that parses to < 1 is ignored in favor of the next source so the cap can - never be disabled into an unbounded encode storm. - """ - env_val = os.getenv("HERMES_VISION_MAX_CONCURRENCY", "").strip() - if env_val: - try: - parsed = int(env_val) - if parsed >= 1: - return parsed - except ValueError: - pass - try: - from hermes_cli.config import cfg_get, load_config - cfg = load_config() - val = cfg_get(cfg, "auxiliary", "vision", "max_concurrency") - if val is not None: - parsed = int(val) - if parsed >= 1: - return parsed - except Exception: - pass - return _detect_host_cpus() + """HERMES_VISION_MAX_CONCURRENCY → ``auxiliary.vision.max_concurrency`` → host cores (values < 1 ignored).""" + val = _read_vision_setting("HERMES_VISION_MAX_CONCURRENCY", "max_concurrency", int, minimum=1) + return _detect_host_cpus() if val is None else val _VISION_CPU_WORKERS = _resolve_vision_cpu_workers() -# Dedicated, bounded executor for the CPU-bound encode/resize burst ONLY. We do -# NOT use the default executor (run_in_executor(None, ...)) — that pool is shared -# with the gateway and web server, so a fan-out would park encode work there and -# starve those callers. Sizing it to the usable core count means at most -# _VISION_CPU_WORKERS encodes run at once; further encodes queue on this -# executor's work queue, leaving cores free for the event loop. The LLM call is -# deliberately left OUTSIDE this executor so multi-image workflows keep full -# request concurrency. _vision_cpu_executor = ThreadPoolExecutor( max_workers=_VISION_CPU_WORKERS, thread_name_prefix="vision-encode", @@ -196,14 +141,7 @@ _vision_cpu_executor = ThreadPoolExecutor( async def _run_encode_on_cpu_executor(fn, *args, **kwargs): - """Run a sync encode/resize callable on the bounded vision CPU executor. - - Offloads CPU-bound image work to :data:`_vision_cpu_executor` so it (a) - never runs on the caller's event-loop thread and (b) is bounded to the - host's usable core count process-wide. Excess encodes queue on the - executor instead of all running at once, leaving cores free for the loop. - The LLM call must NOT be routed through here — only the encode/resize. - """ + """Run a sync encode/resize callable on the bounded vision CPU executor (never the LLM call).""" import functools loop = asyncio.get_running_loop() return await loop.run_in_executor( @@ -212,297 +150,41 @@ async def _run_encode_on_cpu_executor(fn, *args, **kwargs): def _image_url_shape_ok(url: str) -> bool: - """HTTP(S) shape check only (scheme, netloc). No DNS.""" - if not url or not isinstance(url, str): + """HTTP(S) shape check only (scheme + netloc; no DNS). Extension-less CDN URLs pass.""" + if not url or not isinstance(url, str) or not url.startswith(("http://", "https://")): return False - # Basic HTTP/HTTPS URL check - if not url.startswith(("http://", "https://")): - return False - # Parse to ensure we at least have a network location; still allow URLs - # without file extensions (e.g. CDN endpoints that redirect to images). - parsed = urlparse(url) - if not parsed.netloc: - return False - return True - - -def _validate_image_url(url: str) -> bool: - """Validate image URL for sync callers and tests (SSRF via sync DNS check).""" - if not _image_url_shape_ok(url): - return False - # Block private/internal addresses to prevent SSRF - from tools.url_safety import is_safe_url - return is_safe_url(url) + return bool(urlparse(url).netloc) async def _validate_image_url_async(url: str) -> bool: - """Validate remote image URL without blocking the event loop on DNS.""" + """Validate remote image URL (SSRF guard) without blocking the event loop on DNS.""" if not _image_url_shape_ok(url): return False from tools.url_safety import async_is_safe_url return await async_is_safe_url(url) -def _detect_image_mime_type_from_bytes(data: bytes) -> Optional[str]: - """Magic-byte MIME sniff on raw bytes (authoritative; no extension trust). - - Returns ``None`` for anything without a recognized image header — including - SVG, which has no magic bytes. The resolver special-cases SVG (sniffs - ``= 12 and header[:4] == b"RIFF" and header[8:12] == b"WEBP": - return "image/webp" - return None - - -# Media types the major vision providers (Anthropic in particular) accept for -# inline base64 images. Anything outside this set — SVG, BMP, TIFF, etc. — is -# rejected with a non-retryable 400. Because a vision tool-result is baked into -# immutable conversation history and re-sent every turn, embedding an -# unsupported media_type permanently wedges the session (retries re-send the -# same bad bytes). We MUST normalize to one of these before embedding. -_ANTHROPIC_SUPPORTED_MEDIA_TYPES = frozenset( - {"image/jpeg", "image/png", "image/gif", "image/webp"} -) - - -def _supported_media_types() -> frozenset: - """Formats the ACTIVE main model's server can decode. - - Cloud providers take everything in _ANTHROPIC_SUPPORTED_MEDIA_TYPES. - The managed llama-server decodes with stb_image — no WebP — and an - undecodable image part fails SILENTLY (no error; the model never sees - an image and confabulates). Narrow the set so normalization converts - those formats to PNG before they enter the request or history.""" - try: - from agent.auxiliary_client import _runtime_main_value - from hermes_cli.local_runtime.capabilities import ( - ACCEPTED_IMAGE_MIMES, - is_managed_provider, - ) - - if is_managed_provider( - str(_runtime_main_value("provider") or ""), - str(_runtime_main_value("base_url") or "")): - return ACCEPTED_IMAGE_MIMES - except Exception: # noqa: BLE001 — best-effort narrowing only - pass - return _ANTHROPIC_SUPPORTED_MEDIA_TYPES - - -def _rasterize_svg_to_png(svg_path: Path, out_path: Path) -> bool: - """Best-effort SVG → PNG rasterization. Returns True on success. - - Tries, in order: cairosvg, svglib+reportlab, then system rasterizers - (rsvg-convert, inkscape). All are soft dependencies; if none is available - we return False and the caller rejects the image with an actionable error - rather than embedding an unsupported media_type that would wedge the - session. - """ - # 1) cairosvg (pure-python-ish, most common) - try: - import cairosvg # type: ignore - cairosvg.svg2png(url=str(svg_path), write_to=str(out_path)) - return out_path.exists() and out_path.stat().st_size > 0 - except Exception: - pass - # 2) svglib + reportlab - try: - from svglib.svglib import svg2rlg # type: ignore - from reportlab.graphics import renderPM # type: ignore - drawing = svg2rlg(str(svg_path)) - if drawing is not None: - renderPM.drawToFile(drawing, str(out_path), fmt="PNG") - return out_path.exists() and out_path.stat().st_size > 0 - except Exception: - pass - # 3) system rasterizers - import shutil as _shutil - import subprocess as _subprocess - for cmd in ( - ["rsvg-convert", "-o", str(out_path), str(svg_path)], - ["inkscape", str(svg_path), "--export-type=png", - f"--export-filename={out_path}"], - ): - if _shutil.which(cmd[0]): - try: - _subprocess.run( - cmd, check=True, capture_output=True, timeout=30, - stdin=_subprocess.DEVNULL, - ) - if out_path.exists() and out_path.stat().st_size > 0: - return True - except Exception: - continue - return False - - -def _normalize_to_supported_image( - image_path: Path, detected_mime: str -) -> tuple[Optional[Path], Optional[str], Optional[str]]: - """Ensure an image is in a vision-provider-supported format. - - Returns a 3-tuple ``(path, mime, error)``: - - If ``detected_mime`` is already supported: ``(image_path, detected_mime, None)``. - - If conversion succeeds: ``(new_png_path, "image/png", None)`` — the new - path is a temp file the CALLER must clean up. - - If conversion is impossible: ``(None, None, )``. - - SVG is rasterized to PNG (best-effort, soft deps). Other raster formats - Pillow can read (BMP, TIFF, etc.) are re-encoded to PNG. This runs BEFORE - the image is base64-embedded into conversation history, so an unsupported - media_type can never reach the provider and wedge the session. - """ - if detected_mime in _supported_media_types(): - return image_path, detected_mime, None - - out_dir = get_hermes_dir("cache/vision", "temp_vision_images") - out_dir.mkdir(parents=True, exist_ok=True) - out_path = out_dir / f"converted_{uuid.uuid4()}.png" - - # SVG: needs a rasterizer (Pillow cannot render SVG). - if detected_mime == "image/svg+xml": - if _rasterize_svg_to_png(image_path, out_path): - return out_path, "image/png", None - return ( - None, - None, - "This is an SVG, which vision models cannot read directly, and no " - "SVG rasterizer is installed (tried cairosvg, svglib, rsvg-convert, " - "inkscape). Convert the SVG to PNG first — e.g. open it in a browser " - "and screenshot it, or install a rasterizer " - "(`pip install cairosvg`) — then re-run vision_analyze on the PNG.", - ) - - # Other non-supported raster formats (BMP, TIFF, ...): re-encode via Pillow. - try: - from PIL import Image as _PILImage - with _PILImage.open(image_path) as _img: - if _img.mode not in ("RGB", "RGBA", "L"): - _img = _img.convert("RGBA") - _img.save(out_path, format="PNG") - if out_path.exists() and out_path.stat().st_size > 0: - return out_path, "image/png", None - except Exception as _exc: - logger.warning("Failed to normalize %s image to PNG: %s", - detected_mime, _exc) - return ( - None, - None, - f"Image format {detected_mime!r} is not supported by the vision API " - f"and could not be converted to PNG (install Pillow for raster " - f"conversion). Convert it to PNG or JPEG and try again.", - ) - - -# Full raster validation runs on untrusted images in a shared CPU executor. -# Bound animated-image work by both iteration count and total decoded area so a -# compact file cannot monopolize a worker with an effectively unbounded number -# of frames. Images at or below both limits still have every frame decoded. -_VISION_MAX_VALIDATED_FRAME_COUNT = 100 -_VISION_MAX_VALIDATED_AGGREGATE_PIXELS = 100_000_000 - - -def _validate_raster_image_decodable(image_path: Path) -> Optional[str]: - """Return an error when Pillow cannot completely decode every image frame. - - Magic-byte MIME sniffing and ``Image.open`` only inspect container headers. - A timed-out download can therefore look like a supported PNG/JPEG/GIF/WebP - while its pixel stream is truncated. Native vision results are retained in - conversation history, so embedding those bytes poisons every later provider - request. Verify structure, then reopen and force every frame within the - validation resource limits to decode before the image can enter history. - """ - try: - from PIL import Image as _PILImage - from PIL import ImageSequence as _PILImageSequence - except ImportError: - # Pillow is optional — without it we cannot decode-validate, so pass - # the image through unvalidated rather than rejecting everything. - return None - try: - with _PILImage.open(image_path) as image: - image.verify() - with _PILImage.open(image_path) as image: - validated_pixels = 0 - for frame_number, frame in enumerate( - _PILImageSequence.Iterator(image), start=1 - ): - if frame_number > _VISION_MAX_VALIDATED_FRAME_COUNT: - return ( - "Image validation rejected animation: " - f"frame {frame_number} exceeds the maximum " - f"{_VISION_MAX_VALIDATED_FRAME_COUNT} validated frames." - ) - - frame_pixels = frame.width * frame.height - next_validated_pixels = validated_pixels + frame_pixels - if ( - next_validated_pixels - > _VISION_MAX_VALIDATED_AGGREGATE_PIXELS - ): - return ( - "Image validation rejected animation: aggregate decoded " - f"pixel count would reach {next_validated_pixels} at frame " - f"{frame_number}, exceeding the maximum " - f"{_VISION_MAX_VALIDATED_AGGREGATE_PIXELS}." - ) - - frame.load() - validated_pixels = next_validated_pixels - except Exception as exc: - return f"Image could not be fully decoded: {exc}" - return None - - def _is_retryable_download_error(error: Exception) -> bool: - """Return True only for transient image-download failures worth retrying. + """True only for transient download failures worth retrying. - Non-retryable (fail-fast): - - httpx.HTTPStatusError with a 4xx status other than 429 (404/403/410/...): - the resource is missing or forbidden; retrying can't change that. - - PermissionError: blocked by website policy / SSRF guard. - - ValueError: image too large or blocked redirect — deterministic. - - Retryable (transient): - - httpx 429 (rate limited) and 5xx (server-side) errors. - - Connection/timeout/transport errors (httpx.TransportError) and any - other unclassified exception, which may be a flaky network blip. + Fail-fast: 4xx other than 429 (missing/forbidden), PermissionError (policy + or SSRF block), ValueError (too large / blocked redirect — deterministic). + Retryable: 429, 5xx, transport errors and anything unclassified. """ if isinstance(error, (PermissionError, ValueError)): return False if isinstance(error, httpx.HTTPStatusError): status = error.response.status_code - if 400 <= status < 500 and status != 429: - return False - return True + return not (400 <= status < 500 and status != 429) return True +_DOWNLOAD_USER_AGENT = ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" +) + + async def _stream_download_to_file( client, url: str, @@ -512,24 +194,18 @@ async def _stream_download_to_file( headers: dict, media_label: str = "Image", ) -> Path: - """Stream an HTTP download to *destination* via a temp file with a running size cap. + """Stream a GET to *destination* via a temp file with a running size cap. - Uses ``client.stream("GET", ...)`` so the response body is never fully - buffered in memory — chunks are written to a temp file and the running - byte count is checked against *max_bytes* after each chunk. On success - the temp file is atomically replaced onto *destination*; on failure the - temp file is deleted. - - A ``Content-Length`` header, when present and parseable, is used for an - early rejection before any bytes are streamed, but the streaming cap is - the authoritative guard (servers can omit or lie about the header). + The body is never fully buffered: chunks go to a temp file and the running + count is checked after each one (Content-Length gives an early reject but + servers can omit or lie, so the streaming cap is authoritative). The temp + file is atomically moved onto *destination* on success, deleted on failure. """ from utils import atomic_replace async with client.stream("GET", url, headers=headers) as response: response.raise_for_status() - # Early rejection via Content-Length when present and valid. cl = response.headers.get("content-length") if cl: try: @@ -541,8 +217,7 @@ async def _stream_download_to_file( f"{media_label} too large ({declared_size} bytes, max {max_bytes})" ) - final_url = str(response.url) - blocked = check_website_access(final_url) + blocked = check_website_access(str(response.url)) if blocked: raise PermissionError(blocked["message"]) @@ -574,289 +249,135 @@ async def _stream_download_to_file( return destination -async def _download_image(image_url: str, destination: Path, max_retries: int = 3) -> Path: +async def _ssrf_redirect_guard(response): + """Re-validate each redirect target: a public URL that 302s to + http://169.254.169.254/ would otherwise bypass the pre-flight is_safe_url check. + Async because httpx.AsyncClient awaits event hooks.""" + from tools.url_safety import async_is_safe_url, redirect_target_from_response + redirect_url = redirect_target_from_response(response) + if redirect_url and not await async_is_safe_url(redirect_url): + raise ValueError( + f"Blocked redirect to private/internal address: {redirect_url}" + ) + + +async def _download_media( + url: str, + destination: Path, + max_retries: int, + *, + media_label: str, + accept: str, + max_bytes: int, + timeout: float, + retry_all: bool, +) -> Path: + """Shared SSRF-safe streaming download with exponential backoff (2s/4s/8s). + + ``retry_all=False`` (images) only retries transient errors per + :func:`_is_retryable_download_error` — a 404/403 never succeeds on retry, so + burning three backoff rounds just inflates latency. ``retry_all=True`` + (video) keeps the legacy retry-everything behavior. """ - Download an image from a URL to a local destination (async) with retry logic. - - Args: - image_url (str): The URL of the image to download - destination (Path): The path where the image should be saved - max_retries (int): Maximum number of retry attempts (default: 3) - - Returns: - Path: The path to the downloaded image - - Raises: - Exception: If download fails after all retries - """ - import asyncio - - # Create parent directories if they don't exist destination.parent.mkdir(parents=True, exist_ok=True) - - async def _ssrf_redirect_guard(response): - """Re-validate each redirect target to prevent redirect-based SSRF. - - Without this, an attacker can host a public URL that 302-redirects - to http://169.254.169.254/ and bypass the pre-flight is_safe_url check. - - Must be async because httpx.AsyncClient awaits event hooks. - """ - from tools.url_safety import async_is_safe_url, redirect_target_from_response - redirect_url = redirect_target_from_response(response) - if redirect_url and not await async_is_safe_url(redirect_url): - raise ValueError( - f"Blocked redirect to private/internal address: {redirect_url}" - ) last_error = None for attempt in range(max_retries): try: - blocked = check_website_access(image_url) + blocked = check_website_access(url) if blocked: raise PermissionError(blocked["message"]) from tools.url_safety import create_ssrf_safe_async_client - # Download the image with appropriate headers using async httpx. - # Enable follow_redirects to handle image CDNs that redirect (e.g., Imgur, Picsum). - # SSRF: the client validates DNS at TCP connect time; event_hooks - # validate each redirect target against private IP ranges. - # Streaming: body is written chunk-by-chunk to a temp file so the - # size cap bounds memory, not just disk. + # follow_redirects for CDNs; the client validates DNS at connect + # time and the hook re-validates each redirect target. async with create_ssrf_safe_async_client( - timeout=_VISION_DOWNLOAD_TIMEOUT, + timeout=timeout, follow_redirects=True, event_hooks={"response": [_ssrf_redirect_guard]}, ) as client: await _stream_download_to_file( - client, - image_url, - destination, - _VISION_MAX_DOWNLOAD_BYTES, - headers={ - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", - "Accept": "image/*,*/*;q=0.8", - }, - media_label="Image", + client, url, destination, max_bytes, + headers={"User-Agent": _DOWNLOAD_USER_AGENT, "Accept": accept}, + media_label=media_label, ) - return destination except Exception as e: last_error = e - # Error-class-aware retry: only retry transient failures. A 4xx - # client error (404/403/410, etc.) will never succeed on retry — - # the resource isn't there or we're not allowed — so burning 3 - # attempts with 2s/4s/8s backoff just inflates latency. 429 (rate - # limit) and 5xx remain retryable. PermissionError (policy block) - # and ValueError (too-large / SSRF redirect) are also terminal. - if not _is_retryable_download_error(e) or attempt >= max_retries - 1: + final = attempt >= max_retries - 1 + if final or (not retry_all and not _is_retryable_download_error(e)): logger.error( - "Image download failed after %s attempt(s): %s", - attempt + 1, - str(e)[:100], - exc_info=True, + "%s download failed after %s attempt(s): %s", + media_label, attempt + 1, str(e)[:100], exc_info=True, ) - raise - wait_time = 2 ** (attempt + 1) # 2s, 4s, 8s - logger.warning("Image download failed (attempt %s/%s): %s", attempt + 1, max_retries, str(e)[:50]) - logger.warning("Retrying in %ss...", wait_time) + if not retry_all: + raise + break + wait_time = 2 ** (attempt + 1) + logger.warning("%s download failed (attempt %s/%s): %s", + media_label, attempt + 1, max_retries, str(e)[:50]) + if not retry_all: + logger.warning("Retrying in %ss...", wait_time) await asyncio.sleep(wait_time) - # The loop always returns on success or re-raises on the final/non-retryable - # attempt, so reaching here means max_retries was non-positive. + # Reaching here means max_retries was non-positive (or video exhausted retries). if last_error is not None: raise last_error raise RuntimeError( - f"_download_image exited retry loop without attempting (max_retries={max_retries})" + f"_download_{media_label.lower()} exited retry loop without attempting (max_retries={max_retries})" ) -def _determine_mime_type(image_path: Path) -> str: - """ - Determine the MIME type of an image based on its file extension. - - Args: - image_path (Path): Path to the image file - - Returns: - str: The MIME type (defaults to image/jpeg if unknown) - """ - extension = image_path.suffix.lower() - mime_types = { - '.jpg': 'image/jpeg', - '.jpeg': 'image/jpeg', - '.png': 'image/png', - '.gif': 'image/gif', - '.bmp': 'image/bmp', - '.webp': 'image/webp', - '.svg': 'image/svg+xml' - } - return mime_types.get(extension, 'image/jpeg') +async def _download_image(image_url: str, destination: Path, max_retries: int = 3) -> Path: + """Download an image with SSRF protection and error-class-aware retry.""" + return await _download_media( + image_url, destination, max_retries, + media_label="Image", accept="image/*,*/*;q=0.8", + max_bytes=_VISION_MAX_DOWNLOAD_BYTES, timeout=_VISION_DOWNLOAD_TIMEOUT, + retry_all=False, + ) def _image_to_base64_data_url(image_path: Path, mime_type: Optional[str] = None) -> str: - """ - Convert an image file to a base64-encoded data URL. - - Args: - image_path (Path): Path to the image file - mime_type (Optional[str]): MIME type of the image (auto-detected if None) - - Returns: - str: Base64-encoded data URL (e.g., "data:image/jpeg;base64,...") - """ - # Read the image as bytes - data = image_path.read_bytes() - - # Encode to base64 - encoded = base64.b64encode(data).decode("ascii") - - # Determine MIME type - mime = mime_type or _determine_mime_type(image_path) - - # Create data URL - data_url = f"data:{mime};base64,{encoded}" - - return data_url + """``data:;base64,...`` for a file (MIME from extension when not given).""" + encoded = base64.b64encode(image_path.read_bytes()).decode("ascii") + return f"data:{mime_type or _determine_mime_type(image_path)};base64,{encoded}" -# Absolute hard ceiling for vision API payloads (20 MB) — above this, no major -# provider accepts the image and we reject outright. +# Absolute hard ceiling for vision payloads (20 MB): no major provider accepts more. _MAX_BASE64_BYTES = 20 * 1024 * 1024 -# Proactive embed cap for conversation-history reuse. Native vision_analyze -# bakes the data-URL into the tool result, which is re-sent on every later -# turn. The 20 MB hard ceiling / Anthropic 5 MB reject-cap still apply as -# safety nets; those are one-shot viewing limits, not history-reuse sizes. -# A 4 MB / 7900px embed was observed at ~400K chars and ~100–260K billed -# tokens per image (#92699), so we size for model reading instead: 256 KB -# keeps a 1568px screenshot cheap enough to ride the session (PNGs that -# exceed it are downscaled further by the byte-budget ladder), well under -# every provider's per-image limit. +# Proactive embed caps for conversation-history reuse. Native vision_analyze +# bakes the data URL into the tool result, re-sent every later turn; a 4 MB / +# 7900px embed cost ~100-260K billed tokens per image. 256 KB keeps a 1568px +# screenshot cheap enough to ride the session; Anthropic's tokenizer downsamples +# to a 1568px long edge anyway, so pixels past that cost wire bytes for no +# fidelity. The 20 MB / provider 5 MB caps remain as one-shot safety nets. _EMBED_TARGET_BYTES = 256 * 1024 - -# Proactive embed dimension cap (px, longest side). Anthropic still rejects -# above 8000px independently of the byte cap, but its tokenizer downsamples -# to a 1568px long edge — pixels past that cost wire bytes and never extra -# model fidelity. Cap at 1568 so history embeds match what the model sees. _EMBED_MAX_DIMENSION = 1568 -# Target size when auto-resizing on API failure (5 MB). After a provider -# rejects an image, we downscale to this target and retry once. +# Target when auto-resizing after a provider size rejection (retry once). _RESIZE_TARGET_BYTES = 5 * 1024 * 1024 +_SIZE_ERROR_HINTS = ( + "too large", "payload", "413", "content_too_large", + "request_too_large", "exceeds", "size limit", +) + def _is_image_size_error(error: Exception) -> bool: """Detect if an API error is related to image or payload size.""" err_str = str(error).lower() - return any(hint in err_str for hint in ( - "too large", "payload", "413", "content_too_large", - "request_too_large", "image_url", "invalid_request", - "exceeds", "size limit", - )) - - -def _image_exceeds_dimension(image_path: Path, max_dimension: int) -> bool: - """True if the image's longest side exceeds ``max_dimension`` px. - - Anthropic enforces an 8000px per-side cap independently of the 5 MB byte - cap, so a tall small-byte screenshot can pass every byte check yet trip a - non-retryable 400. Returns False (don't force a resize) when Pillow is - unavailable or the file can't be read as an image — the byte-based checks - still apply, and we never want a missing soft dependency to break the - embed path. - """ - try: - from PIL import Image as _PILImage - with _PILImage.open(image_path) as _img: - return max(_img.size) > max_dimension - except Exception: - return False - - -def _crop_image_region( - image_path: Path, - region: Any, - offset_out: Optional[dict] = None, -) -> tuple[Optional[Path], Optional[str], Optional[str]]: - """Crop ``image_path`` to ``region`` = [x1, y1, x2, y2] (original-image pixels). - - Applied BEFORE :func:`_resize_image_for_vision` so the cropped area gets - the full downscale resolution budget — a "zoom" into a detail region. - Coordinates are clamped to the image bounds; a region that clamps to zero - area (or is inverted/malformed) is rejected with an error naming the - actual image dimensions so the caller can retry with sensible values. - - Ported from: QwenLM/qwen-code zoom-image.ts (Apache-2.0). - - Returns: - (cropped_temp_path, out_mime, None) on success — the caller owns - cleanup of the temp file — or (None, None, error_message) on failure. - """ - try: - from PIL import Image - except ImportError: - return None, None, ( - "region cropping requires Pillow (`pip install Pillow`); " - "retry without the region parameter." - ) - - if ( - not isinstance(region, (list, tuple)) - or len(region) != 4 - or not all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in region) - ): - return None, None, ( - "Invalid region: expected [x1, y1, x2, y2] as four numbers " - "(pixel coordinates in the original image)." - ) - - try: - with Image.open(image_path) as img: - width, height = img.size - x1, y1, x2, y2 = (int(v) for v in region) - # Clamp to image bounds. - cx1 = max(0, min(x1, width)) - cy1 = max(0, min(y1, height)) - cx2 = max(0, min(x2, width)) - cy2 = max(0, min(y2, height)) - if cx2 <= cx1 or cy2 <= cy1: - return None, None, ( - f"Invalid region [{x1}, {y1}, {x2}, {y2}]: crops to zero " - f"area after clamping to the image bounds. The image is " - f"{width}x{height} px — pick x1 Optional[str]: - """Build a coordinate-mapping disclosure note for the analysis result. - - ``scale_info`` (from :func:`_resize_image_for_vision`) carries the - original and downscaled pixel dimensions when a downscale actually - happened. ``crop_offset`` (from :func:`_crop_image_region`) carries the - clamped crop origin when a region zoom was applied. Returns ``None`` when - neither applies — no note, no noise. - """ + """Coordinate-mapping disclosure for downscale (``scale_info``) and/or region + crop (``crop_offset``); ``None`` when neither applied — no note, no noise.""" parts = [] if scale_info: ow, oh = scale_info["orig_width"], scale_info["orig_height"] @@ -885,9 +406,28 @@ def _build_scale_note( f"{crop_offset['y']}); coordinates are relative to that crop " f"origin — add the offset to map back to the full image." ) - if not parts: + return " ".join(parts) if parts else None + + +def _import_pillow_for_resize(): + """Pillow is a lazy-installable soft dependency; return ``PIL.Image`` or None. + + ``prompt=False``: a blocking input() deadlocks the interactive CLI where + prompt_toolkit owns stdin. The install is gated by + security.allow_lazy_installs, so reaching it is already opt-in. + """ + try: + from PIL import Image + return Image + except ImportError: + pass + try: + from tools.lazy_deps import ensure as _ensure_dep + _ensure_dep("tool.vision", prompt=False) + from PIL import Image + return Image + except Exception: return None - return " ".join(parts) def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None, @@ -895,143 +435,77 @@ def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None, max_dimension: Optional[int] = None, scale_out: Optional[dict] = None, force_jpeg: bool = False) -> str: - """Convert an image to a base64 data URL, auto-resizing if too large. + """Base64 data URL, progressively downscaled with Pillow while over budget. - Tries Pillow first to progressively downscale oversized images. If Pillow - is not installed or resizing still exceeds the limit, falls back to the raw - bytes and lets the caller handle the size check. + Halves dimensions (aspect-preserving, 64px floor) up to 4 times; JPEG also + walks a quality ladder (85/70/50) at each step. Without Pillow, or if it + still doesn't fit, returns the best attempt (or raw bytes) and lets the + caller apply the size check. - Args: - max_dimension: If set, images whose longest side exceeds this pixel - count are forcibly downscaled even if they're under the byte - budget. Anthropic enforces an 8000 px per-side cap independently - of the 5 MB byte cap. - force_jpeg: Re-encode as JPEG even for PNG input when a resize is - needed. PNG has no quality ladder — its only shrink lever is - halving dimensions, which destroys text legibility on dense - screenshots. History-reuse embeds (#92699) opt in so a text-heavy - screenshot keeps its readable resolution and shrinks via JPEG - quality instead. Images already under both caps are returned - unchanged (still PNG). - - Returns the base64 data URL string. + ``max_dimension``: force a downscale above this long edge even when bytes + fit (Anthropic's 8000px cap is independent of bytes). ``force_jpeg``: + re-encode PNG input as JPEG when a resize is needed — PNG's only shrink + lever is halving dimensions, which destroys text legibility on dense + screenshots; history-reuse embeds opt in. Images under both caps return unchanged. """ - # Quick file-size estimate: base64 expands by ~4/3, plus data URL header. - # Skip the expensive full-read + encode if Pillow can resize directly. file_size = image_path.stat().st_size - estimated_b64 = (file_size * 4) // 3 + 100 # ~header overhead + estimated_b64 = (file_size * 4) // 3 + 100 # base64 ~4/3 + data URL header needs_resize_for_bytes = estimated_b64 > max_base64_bytes + needs_resize_for_dims = ( + max_dimension is not None and _image_exceeds_dimension(image_path, max_dimension) + ) - # Check pixel dimensions even if bytes are fine. - needs_resize_for_dims = False - if max_dimension is not None: - try: - from PIL import Image as _PILQuick - with _PILQuick.open(image_path) as _quick_img: - if max(_quick_img.size) > max_dimension: - needs_resize_for_dims = True - except Exception: - pass # can't check; Pillow path below will handle or skip - + data_url = None if not needs_resize_for_bytes and not needs_resize_for_dims: - # Small enough — just encode directly. data_url = _image_to_base64_data_url(image_path, mime_type=mime_type) if len(data_url) <= max_base64_bytes: return data_url - else: - data_url = None # defer full encode; try Pillow resize first - # Attempt auto-resize with Pillow (soft dependency) - try: - from PIL import Image - import io as _io - except ImportError: - # Pillow is a lazy-installable soft dependency. Try a best-effort - # install (respects security.allow_lazy_installs; no-op if disabled or - # offline), then re-import. If it still isn't importable, fall back to - # the raw bytes and let the caller raise the size error. - try: - from tools.lazy_deps import ensure as _ensure_dep - # prompt=False: never raise a blocking input() prompt mid-session. - # Under the interactive CLI prompt_toolkit owns stdin, so a bare - # input() deadlocks the terminal (#40490). The install is already - # gated by security.allow_lazy_installs, so reaching here is opt-in. - _ensure_dep("tool.vision", prompt=False) - from PIL import Image - import io as _io - except Exception: - logger.info("Pillow not installed — cannot auto-resize oversized image") - if data_url is None: - data_url = _image_to_base64_data_url(image_path, mime_type=mime_type) - return data_url # caller will raise the size error + def _raw() -> str: + return data_url or _image_to_base64_data_url(image_path, mime_type=mime_type) + + Image = _import_pillow_for_resize() + if Image is None: + logger.info("Pillow not installed — cannot auto-resize oversized image") + return _raw() # caller will raise the size error logger.info("Image file is %.1f MB (estimated base64 %.1f MB, limit %.1f MB, max_dimension=%s), auto-resizing...", file_size / (1024 * 1024), estimated_b64 / (1024 * 1024), max_base64_bytes / (1024 * 1024), max_dimension) mime = mime_type or _determine_mime_type(image_path) - # Choose output format: JPEG for photos (smaller), PNG for transparency. - # force_jpeg overrides for history-reuse embeds: a resize-needing PNG - # screenshot re-encodes as JPEG so the quality ladder can shrink bytes - # without halving resolution (text legibility, #92699). - if force_jpeg: - pil_format = "JPEG" - else: - pil_format = "PNG" if mime == "image/png" else "JPEG" + # JPEG for photos (smaller), PNG for transparency — unless force_jpeg. + pil_format = "PNG" if (mime == "image/png" and not force_jpeg) else "JPEG" out_mime = "image/png" if pil_format == "PNG" else "image/jpeg" try: img = Image.open(image_path) except Exception as exc: logger.info("Pillow cannot open image for resizing: %s", exc) - if data_url is None: - data_url = _image_to_base64_data_url(image_path, mime_type=mime_type) - return data_url # fall through to size-check in caller - # JPEG cannot encode alpha or palette/grayscale-alpha modes; normalize - # anything that isn't already plain RGB/grayscale. force_jpeg newly - # routes PNG inputs here, so exotic modes (LA/PA) must not crash save(). + return _raw() + # JPEG cannot encode alpha/palette modes (force_jpeg routes PNGs here). if pil_format == "JPEG" and img.mode not in {"RGB", "L"}: img = img.convert("RGB") - # Strategy: halve dimensions until both base64 fits AND pixel dimensions - # are within limits, up to 4 rounds. - # For JPEG, also try reducing quality at each size step. - # For PNG, quality is irrelevant — only dimension reduction helps. quality_steps = (85, 70, 50) if pil_format == "JPEG" else (None,) - orig_dims = (img.width, img.height) - prev_dims = (img.width, img.height) - candidate = None # will be set on first loop iteration + orig_dims = prev_dims = (img.width, img.height) + candidate = None def _record_scale(w: int, h: int) -> None: - """Publish the downscale into ``scale_out`` when dims changed.""" if scale_out is not None and (w, h) != orig_dims: - scale_out["orig_width"] = orig_dims[0] - scale_out["orig_height"] = orig_dims[1] - scale_out["new_width"] = w - scale_out["new_height"] = h - - def _dims_ok(w: int, h: int) -> bool: - """True if both pixel dimensions are within the limit.""" - if max_dimension is None: - return True - return max(w, h) <= max_dimension + scale_out.update(orig_width=orig_dims[0], orig_height=orig_dims[1], + new_width=w, new_height=h) for attempt in range(5): if attempt > 0: - # Proportional scaling: halve the longer side and scale the - # shorter side to preserve aspect ratio (min dimension 64). - scale = 0.5 - new_w = max(int(img.width * scale), 64) - new_h = max(int(img.height * scale), 64) - # Re-derive the scale from whichever dimension hit the floor - # so both axes shrink by the same factor. + # Halve, then re-derive the scale from whichever axis hit the 64px + # floor so both axes shrink by the same factor. + new_w = max(int(img.width * 0.5), 64) + new_h = max(int(img.height * 0.5), 64) if new_w == 64 and img.width > 0: - effective_scale = 64 / img.width - new_h = max(int(img.height * effective_scale), 64) + new_h = max(int(img.height * (64 / img.width)), 64) elif new_h == 64 and img.height > 0: - effective_scale = 64 / img.height - new_w = max(int(img.width * effective_scale), 64) - # Stop if dimensions can't shrink further + new_w = max(int(img.width * (64 / img.height)), 64) if (new_w, new_h) == prev_dims: break img = img.resize((new_w, new_h), Image.LANCZOS) @@ -1039,102 +513,64 @@ def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None, logger.info("Resized to %dx%d (attempt %d)", new_w, new_h, attempt) for q in quality_steps: - buf = _io.BytesIO() - save_kwargs = {"format": pil_format} - if q is not None: - save_kwargs["quality"] = q - img.save(buf, **save_kwargs) - encoded = base64.b64encode(buf.getvalue()).decode("ascii") - candidate = f"data:{out_mime};base64,{encoded}" - if len(candidate) <= max_base64_bytes and _dims_ok(img.width, img.height): + buf = BytesIO() + img.save(buf, format=pil_format, **({} if q is None else {"quality": q})) + candidate = f"data:{out_mime};base64,{base64.b64encode(buf.getvalue()).decode('ascii')}" + dims_ok = max_dimension is None or max(img.width, img.height) <= max_dimension + if len(candidate) <= max_base64_bytes and dims_ok: logger.info("Auto-resized image fits: %.1f MB (quality=%s, %dx%d)", len(candidate) / (1024 * 1024), q, img.width, img.height) _record_scale(img.width, img.height) return candidate - # If we still can't get it small enough, return the best attempt - # and let the caller decide if candidate is not None: logger.warning("Auto-resize could not fit image under %.1f MB (best: %.1f MB)", max_base64_bytes / (1024 * 1024), len(candidate) / (1024 * 1024)) _record_scale(img.width, img.height) return candidate - - # Shouldn't reach here, but fall back to full encode - return data_url or _image_to_base64_data_url(image_path, mime_type=mime_type) + return _raw() # --------------------------------------------------------------------------- -# Native fast path: short-circuit the auxiliary LLM when the active main model -# supports native vision. Instead of asking a separate LLM to describe the -# image and returning text, we load the image, base64-encode it, and return a -# multimodal tool-result envelope. The agent loop unwraps the envelope into an -# OpenAI-style content list on the `tool` role; provider adapters (anthropic, -# codex_responses, chat_completions) translate that into Anthropic -# tool_result image blocks / Responses input_image / OpenAI image_url tool -# content. The main model then "sees" the pixels directly on its next turn. +# Native fast path: when the active main model supports vision, skip the aux +# LLM and return the image bytes as a multimodal tool-result envelope. The +# agent loop unwraps it into an OpenAI-style content list on the `tool` role; +# provider adapters translate that per backend, so the main model "sees" the +# pixels directly on its next turn. # --------------------------------------------------------------------------- +# Providers whose tool results accept image content (spec docs verified +# Apr-2026): Anthropic Messages (+ aggregators proxying Claude — assume support, +# falling back to text would regress their frontier models), OpenAI Chat +# Completions / Responses. Gemini is gated on model: only 3.x supports +# multimodal functionResponse. +_TOOL_RESULT_MEDIA_PROVIDERS = frozenset({ + "openrouter", "nous", "vertex", "bedrock", "anthropic-vertex", "google-vertex", + "anthropic", "claude", "anthropic-direct", + "openai", "openai-chat", "openai-codex", "azure-openai", +}) +_GEMINI_PROVIDERS = frozenset({"google", "gemini", "google-gemini", "google-vertex-gemini"}) + def _supports_media_in_tool_results(provider: str, model: str) -> bool: - """Whether the given provider+model combination accepts image content - inside a tool-result message. + """Whether provider+model accepts image content inside a tool-result message. - Providers covered today (per spec docs verified Apr-2026): - - * Anthropic Messages API (``anthropic`` provider, plus aggregators that - proxy Claude — ``openrouter``, ``nous``, ``vertex``, ``bedrock``): - ``tool_result`` blocks accept ``image`` content blocks. - * OpenAI Chat Completions: tool messages accept array content with - ``image_url`` parts. - * OpenAI Responses (``openai-codex``): ``function_call_output.output`` - accepts an array of ``input_text``/``input_image`` items. - * Gemini 3 (and proxied via aggregators): supports multimodal tool - results. Older Gemini does NOT. - - For unknown / legacy providers we conservatively return False — the - caller falls back to the legacy aux-LLM text path. The check is relaxed - when the provider's ``ProviderProfile`` declares ``supports_vision=True``. + Unknown providers are conservatively False (caller falls back to the aux-LLM + text path) unless their ``ProviderProfile`` declares ``supports_vision``. """ if not isinstance(provider, str): return False p = provider.strip().lower() if not p: return False - - # Aggregators that route to multiple vendors — assume support since - # users on these aggregators are typically using vision-capable - # frontier models. Falling back to text would be a regression for - # them. - _AGGREGATORS = { - "openrouter", "nous", "vertex", "bedrock", "anthropic-vertex", - "google-vertex", - } - if p in _AGGREGATORS: + if p in _TOOL_RESULT_MEDIA_PROVIDERS: return True - - # Native Anthropic - if p in {"anthropic", "claude", "anthropic-direct"}: - return True - - # OpenAI Chat Completions and Responses - if p in {"openai", "openai-chat", "openai-codex", "azure-openai"}: - return True - - # Gemini — gate on model name; older Gemini variants did not support - # multimodal functionResponse. Gemini 3.x does. - if p in {"google", "gemini", "google-gemini", "google-vertex-gemini"}: + if p in _GEMINI_PROVIDERS: if not isinstance(model, str): return False m = model.strip().lower() - if "gemini-3" in m or "gemini-pro-3" in m or "gemini-flash-3" in m: - return True - return False - - # Check the provider's registered profile for the supports_vision flag. - # This covers vision-capable providers like xiaomi, minimax, etc. that - # aren't in the hardcoded list above. + return any(tag in m for tag in ("gemini-3", "gemini-pro-3", "gemini-flash-3")) try: from providers import get_provider_profile profile = get_provider_profile(p) @@ -1142,24 +578,13 @@ def _supports_media_in_tool_results(provider: str, model: str) -> bool: return True except Exception: pass - - # Other vision-capable provider stacks. Conservative default: False. - # Add explicit entries here as we verify each provider's tool-result - # multimodal support empirically. return False def _should_use_native_vision_fast_path() -> bool: - """Whether vision tools should attach the image to the main model directly - instead of routing through the auxiliary vision LLM. - - True when image routing resolves to ``native`` AND either the provider is - known to accept images inside tool results, or the user explicitly declared - the model vision-capable via the ``model.supports_vision`` config override. - The override is the escape hatch for custom/local providers that aren't in - the static allowlist. Best-effort: any resolution failure returns False so - the caller falls back to the legacy aux-LLM path. - """ + """True when image routing resolves to ``native`` AND the provider accepts + images in tool results, or the user set the ``model.supports_vision`` + override (escape hatch for custom/local providers). Any failure → False.""" try: from agent.auxiliary_client import _read_main_provider, _read_main_model from agent.image_routing import decide_image_input_mode, _lookup_supports_vision @@ -1186,27 +611,12 @@ def _build_native_vision_tool_result( image_size_bytes: int, scale_note: Optional[str] = None, ) -> Dict[str, Any]: - """Build the multimodal tool-result envelope returned by the fast path. + """Multimodal tool-result envelope (``_multimodal`` + ``content`` list). - Shape: - { - "_multimodal": True, - "content": [ - {"type": "text", "text": ""}, - {"type": "image_url", "image_url": {"url": "data:image/png;base64,..."}} - ], - "text_summary": "", - "meta": {"image_url": ..., "size_bytes": N}, - } - - The text part exists for two reasons: (1) it gives the model an - instruction to act on now that the pixels are in context, and - (2) providers that don't support multimodal tool results can fall back - to ``text_summary``. + The text part is intentionally minimal — the model already has the question + in context; it acknowledges the image is visible. ``text_summary`` is the + fallback for providers without multimodal tool results. """ - # The tool-result text part is intentionally minimal. The model already - # has the user's original question in context; this just acknowledges - # the image is now visible and reminds it what it was asked. text_part = ( "Image loaded into your context — you can see it natively now. " "Use your built-in vision to answer the user." @@ -1237,19 +647,99 @@ def _build_native_vision_tool_result( } -@contextlib.asynccontextmanager -async def _vision_concurrency_slot(): - """Deprecated no-op shim kept for backward compatibility. +def _unlink_quietly(path: Optional[Path]) -> None: + if path is not None: + try: + if path.exists(): + path.unlink() + except Exception: + pass - The fan-out cap was narrowed to the CPU-bound encode/resize burst only - (see :data:`_vision_cpu_executor` / :func:`_run_encode_on_cpu_executor`). - Holding a slot across the whole analysis serialized legitimate multi-image - workflows behind the slow LLM call, which is exactly what we don't want. - This context manager no longer gates anything; encode/resize is bounded - where it actually runs. Retained only so any external caller importing it - keeps working. + +class _ImagePrepError(ValueError): + """Raised by :func:`_prepare_image`; the message is user-facing.""" + + +class _PreparedImage: + """Temp image ready to encode; ``path`` is owned by the caller (delete it).""" + __slots__ = ("path", "mime", "size_bytes", "crop_offset") + + def __init__(self, path: Path, mime: Optional[str], size_bytes: int, crop_offset: dict): + self.path, self.mime, self.size_bytes, self.crop_offset = path, mime, size_bytes, crop_offset + + +async def _prepare_image( + image_url: str, task_id: Optional[str], region: Optional[list], *, validate_decode: bool, +) -> _PreparedImage: + """Resolve → materialize → normalize → (validate) → (crop). Raises ``_ImagePrepError``. + + The single resolver unifies data:/http/file/local/container sources and + enforces terminal-backend confinement; bytes land in a temp file so the + path-based encode/resize pipeline is reused. Unsupported formats (SVG, BMP) + are converted to PNG BEFORE encoding — an unsupported media_type baked into + immutable history would 400 on every resume. The crop runs BEFORE any + downscale so the region keeps the full resolution budget. Blocking + rasterizer/Pillow work is offloaded. Intermediate temp files are deleted; + on error nothing is left behind. """ - yield + from tools.image_source import ImageResolutionError, ResolveContext, resolve_image_source + + try: + resolved = await resolve_image_source(image_url, ResolveContext(task_id=task_id)) + except ImageResolutionError as exc: + raise _ImagePrepError(str(exc)) from exc + + temp_dir = get_hermes_dir("cache/vision", "temp_vision_images") + temp_dir.mkdir(parents=True, exist_ok=True) + path = temp_dir / f"temp_image_{uuid.uuid4()}.img" + await asyncio.to_thread(path.write_bytes, resolved.data) + mime = resolved.mime + size_bytes = len(resolved.data) + crop_offset: dict = {} + try: + normalized_path, mime, norm_err = await asyncio.to_thread( + _normalize_to_supported_image, path, mime, + ) + if norm_err or normalized_path is None: + raise _ImagePrepError(norm_err or "Image normalization failed.") + if normalized_path != path: + _unlink_quietly(path) + path = normalized_path + size_bytes = path.stat().st_size + + if validate_decode: + decode_error = await _run_encode_on_cpu_executor( + _validate_raster_image_decodable, path, + _VISION_MAX_VALIDATED_FRAME_COUNT, _VISION_MAX_VALIDATED_AGGREGATE_PIXELS, + ) + if decode_error: + raise _ImagePrepError(decode_error) + + if region is not None: + cropped_path, cropped_mime, crop_err = await asyncio.to_thread( + _crop_image_region, path, region, offset_out=crop_offset, + ) + if crop_err or cropped_path is None: + raise _ImagePrepError(crop_err or "Region crop failed.") + _unlink_quietly(path) + path = cropped_path + mime = cropped_mime + size_bytes = path.stat().st_size + except BaseException: + _unlink_quietly(path) + raise + return _PreparedImage(path, mime, size_bytes, crop_offset) + + +def _too_large_message(image_data_url: str) -> str: + return ( + f"Image too large for vision API: base64 payload is " + f"{len(image_data_url) / (1024 * 1024):.1f} MB " + f"(limit {_MAX_BASE64_BYTES / (1024 * 1024):.0f} MB) " + f"even after resizing. Install Pillow " + f"(`pip install Pillow`) for better auto-resize, " + f"or compress the image manually." + ) async def _vision_analyze_native( @@ -1260,145 +750,54 @@ async def _vision_analyze_native( ) -> Any: """Fast path for vision-capable main models. - Loads the image (data: / http(s) / file:// / local path / sandbox-container - path) via the unified resolver, base64-encodes it, and returns a multimodal - tool-result envelope. The agent loop unwraps it; provider adapters serialize - it into the right tool-result-with-image shape for each backend. - - Returns: - A ``_multimodal`` envelope dict on success. - A JSON error string on failure (matches the existing tool-result - contract so the agent loop displays errors normally). + Returns a ``_multimodal`` envelope dict on success, or a JSON error string + (the normal tool-result contract) on failure. """ if not isinstance(image_url, str) or not image_url.strip(): return tool_error("image_url is required", success=False) - temp_image_path: Optional[Path] = None - should_cleanup = False + prepared: Optional[_PreparedImage] = None try: from tools.interrupt import is_interrupted if is_interrupted(): return tool_error("Interrupted", success=False) - # Resolve the source to raw bytes through the single resolver (unifies - # data:/http/file/local/container and enforces terminal-backend - # confinement). Materialize to a temp file so the existing path-based - # encode/resize/embed-cap pipeline below is reused verbatim. - from tools.image_source import ( - ImageResolutionError, - ResolveContext, - resolve_image_source, - ) - try: - resolved = await resolve_image_source(image_url, ResolveContext(task_id=task_id)) - except ImageResolutionError as exc: + prepared = await _prepare_image(image_url, task_id, region, validate_decode=True) + except _ImagePrepError as exc: return tool_error(str(exc), success=False) - detected_mime_type = resolved.mime - image_size_bytes = len(resolved.data) - temp_dir = get_hermes_dir("cache/vision", "temp_vision_images") - temp_dir.mkdir(parents=True, exist_ok=True) - temp_image_path = temp_dir / f"temp_image_{uuid.uuid4()}.img" - await asyncio.to_thread(temp_image_path.write_bytes, resolved.data) - should_cleanup = True - - # Normalize unsupported formats (SVG, BMP, ...) to PNG BEFORE embedding. - # Anthropic only accepts jpeg/png/gif/webp; an unsupported media_type - # baked into immutable history wedges the session with a 400 on every - # resume. Convert here so it can never enter history. Offloaded — the - # rasterizers/Pillow are blocking. - normalized_path, detected_mime_type, _norm_err = await asyncio.to_thread( - _normalize_to_supported_image, temp_image_path, detected_mime_type, - ) - if _norm_err or normalized_path is None: - return tool_error( - _norm_err or "Image normalization failed.", success=False, - ) - if normalized_path != temp_image_path: - # We created a temp PNG — swap to it and ensure it's cleaned up. - if should_cleanup and temp_image_path.exists(): - try: - temp_image_path.unlink() - except Exception: - pass - temp_image_path = normalized_path - should_cleanup = True - image_size_bytes = temp_image_path.stat().st_size - - decode_error = await _run_encode_on_cpu_executor( - _validate_raster_image_decodable, temp_image_path, - ) - if decode_error: - return tool_error(decode_error, success=False) - - # Optional region zoom: crop BEFORE the downscale/embed-cap pipeline - # so the cropped area gets the full resolution budget. - _crop_offset: dict = {} - _scale_info: dict = {} - if region is not None: - cropped_path, cropped_mime, crop_err = await asyncio.to_thread( - _crop_image_region, temp_image_path, region, - offset_out=_crop_offset, - ) - if crop_err or cropped_path is None: - return tool_error(crop_err or "Region crop failed.", success=False) - if should_cleanup and temp_image_path.exists(): - try: - temp_image_path.unlink() - except Exception: - pass - temp_image_path = cropped_path - detected_mime_type = cropped_mime - should_cleanup = True - image_size_bytes = temp_image_path.stat().st_size - image_data_url = await _run_encode_on_cpu_executor( - _image_to_base64_data_url, - temp_image_path, mime_type=detected_mime_type, + _image_to_base64_data_url, prepared.path, mime_type=prepared.mime, ) - # Proactive embed cap: this image gets baked into conversation - # history and re-sent on every subsequent turn. Resize DOWN to the - # history-reuse target whenever the payload exceeds either the byte - # or long-edge cap, not just at the 20 MB hard ceiling. Anthropic - # still rejects >5 MB / >8000px with a non-retryable 400, but those - # are one-shot viewing limits — history embeds are sized smaller so - # repeated vision_analyze turns don't blow the context (#92699). - _over_bytes = len(image_data_url) > _EMBED_TARGET_BYTES + # Proactive embed cap: this image is re-sent on every later turn, so + # resize DOWN to the history-reuse target whenever the byte or long-edge + # cap is exceeded, not just at the 20 MB hard ceiling. + _scale_info: dict = {} _over_dims = await _run_encode_on_cpu_executor( - _image_exceeds_dimension, temp_image_path, _EMBED_MAX_DIMENSION, + _image_exceeds_dimension, prepared.path, _EMBED_MAX_DIMENSION, ) - if _over_bytes or _over_dims: + if len(image_data_url) > _EMBED_TARGET_BYTES or _over_dims: image_data_url = await _run_encode_on_cpu_executor( _resize_image_for_vision, - temp_image_path, mime_type=detected_mime_type, + prepared.path, mime_type=prepared.mime, max_base64_bytes=_EMBED_TARGET_BYTES, max_dimension=_EMBED_MAX_DIMENSION, scale_out=_scale_info, force_jpeg=True, ) - # If even resizing can't get under the absolute hard ceiling, - # there's nothing more we can do — reject rather than embed a - # session-wedging payload. + # Reject rather than embed a session-wedging payload. if len(image_data_url) > _MAX_BASE64_BYTES: - return tool_error( - f"Image too large for vision API: base64 payload is " - f"{len(image_data_url) / (1024 * 1024):.1f} MB " - f"(limit {_MAX_BASE64_BYTES / (1024 * 1024):.0f} MB) " - f"even after resizing. Install Pillow " - f"(`pip install Pillow`) for better auto-resize, " - f"or compress the image manually.", - success=False, - ) + return tool_error(_too_large_message(image_data_url), success=False) return _build_native_vision_tool_result( image_url=image_url, question=question, image_data_url=image_data_url, - image_size_bytes=image_size_bytes, + image_size_bytes=prepared.size_bytes, scale_note=_build_scale_note( - _scale_info or None, _crop_offset or None, + _scale_info or None, prepared.crop_offset or None, ), ) @@ -1407,12 +806,111 @@ async def _vision_analyze_native( return tool_error(f"Native vision failed: {exc}", success=False) finally: # Only delete temp files we created — never user-provided paths. - if should_cleanup and temp_image_path is not None: - try: - if temp_image_path.exists(): - temp_image_path.unlink() - except Exception: - pass + if prepared is not None: + _unlink_quietly(prepared.path) + + +def _read_vision_call_settings(default_timeout: float, *, min_timeout: Optional[float] = None): + """``auxiliary.vision.timeout`` / ``.temperature`` from config.yaml (defaults 120s-ish / 0.1). + + Local vision models (llama.cpp, ollama) can take well over 30s, hence the + generous defaults; ``min_timeout`` lets video enforce a floor. + """ + timeout, temperature = default_timeout, 0.1 + try: + from hermes_cli.config import cfg_get, load_config + _vision_cfg = cfg_get(load_config(), "auxiliary", "vision", default={}) + _vt = _vision_cfg.get("timeout") + if _vt is not None: + timeout = float(_vt) if min_timeout is None else max(float(_vt), min_timeout) + _vtemp = _vision_cfg.get("temperature") + if _vtemp is not None: + temperature = float(_vtemp) + except Exception: + pass + return timeout, temperature + + +# Error-message classification for the aux-LLM paths: first matching hint set +# wins (order matters — billing before capability before size/format). +_BILLING_HINTS = ("402", "insufficient", "payment required", "credits", "billing") +_IMAGE_ERROR_RULES = ( + (_BILLING_HINTS, + "Insufficient credits or payment required. Please top up your " + "API provider account and try again. Error: {e}"), + (("does not support", "not support image", "content_policy", "multimodal", + "unrecognized request argument", "image input"), + "{model} does not support vision or our request was not " + "accepted by the server. Error: {e}"), + (("invalid_request", "image_url"), + "The vision API rejected the image. This can happen when the " + "image is in an unsupported format, corrupted, or still too " + "large after auto-resize. Try a smaller JPEG/PNG and retry. " + "Error: {e}"), +) +_VIDEO_ERROR_RULES = ( + (_BILLING_HINTS, _IMAGE_ERROR_RULES[0][1]), + (("does not support", "not support video", "content_policy", "multimodal", + "unrecognized request argument", "video input", "video_url"), + "The model does not support video analysis or the request was " + "rejected. Ensure you're using a video-capable model " + "(e.g. google/gemini-2.5-flash). Error: {e}"), + (_SIZE_ERROR_HINTS, + "The video is too large for the API. Try compressing or trimming " + "the video (max ~50 MB). Error: {e}"), +) + + +def _classify_analysis_error(e: Exception, rules, fallback: str, **fmt) -> str: + err_str = str(e).lower() + for hints, template in rules: + if any(hint in err_str for hint in hints): + return template.format(e=e, **fmt) + return fallback.format(e=e) + + +def _debug_call_data(kind: str, source: str, user_prompt: str, model) -> dict: + return { + "parameters": { + f"{kind}_url": source, + "user_prompt": user_prompt[:200] + "..." if len(user_prompt) > 200 else user_prompt, + "model": model, + }, + "error": None, + "success": False, + "analysis_length": 0, + "model_used": model, + f"{kind}_size_bytes": 0, + } + + +def _finish_analysis(tool_name: str, debug_call_data: dict, result: dict) -> str: + _debug.log_call(tool_name, debug_call_data) + _debug.save() + return json.dumps(result, indent=2, ensure_ascii=False) + + +def _cleanup_temp_media(path: Optional[Path], label: str) -> None: + if path and path.exists(): + try: + path.unlink() + logger.debug("Cleaned up temporary %s file", label) + except Exception as cleanup_error: + logger.warning( + "Could not delete temporary file: %s", cleanup_error, exc_info=True + ) + + +async def _call_vision_llm(call_kwargs: dict, empty_log: str): + """Call the aux vision LLM, retrying once on empty content (reasoning-only response).""" + _load_auxiliary_client() + response = await async_call_llm(**call_kwargs) + analysis = extract_content_or_reasoning(response) + if not analysis: + logger.warning(empty_log) + response = await async_call_llm(**call_kwargs) + analysis = extract_content_or_reasoning(response) + return analysis async def vision_analyze_tool( @@ -1422,59 +920,17 @@ async def vision_analyze_tool( task_id: Optional[str] = None, region: Optional[list] = None, ) -> str: - """ - Analyze an image from a URL or local file path using vision AI. - - This tool accepts either an HTTP/HTTPS URL or a local file path. For URLs, - it downloads the image first. In both cases, the image is converted to base64 - and processed using Gemini 3 Flash Preview via OpenRouter API. - - The user_prompt parameter is expected to be pre-formatted by the calling - function (typically model_tools.py) to include both full description - requests and specific questions. - - Args: - image_url (str): The URL or local file path of the image to analyze. - Accepts http://, https:// URLs or absolute/relative file paths. - user_prompt (str): The pre-formatted prompt for the vision model - model (str): The vision model to use (default: google/gemini-3-flash-preview) - - Returns: - str: JSON string containing the analysis results with the following structure: - { - "success": bool, - "analysis": str (defaults to error message if None) - } - - Raises: - Exception: If download fails, analysis fails, or API key is not set - - Note: - - For URLs, temporary images are stored under $HERMES_HOME/cache/vision/ and cleaned up - - For local file paths, the file is used directly and NOT deleted - - Supports common image formats (JPEG, PNG, GIF, WebP, etc.) + """Describe an image (URL, local path, data: URL) with the auxiliary vision LLM. + + ``user_prompt`` is pre-formatted by the caller. Returns JSON + ``{"success": bool, "analysis": str}`` (``analysis`` carries the error + explanation on failure). Temp images live under $HERMES_HOME/cache/vision/. """ if not isinstance(user_prompt, str): user_prompt = str(user_prompt) if user_prompt is not None else "" - debug_call_data = { - "parameters": { - "image_url": image_url, - "user_prompt": user_prompt[:200] + "..." if len(user_prompt) > 200 else user_prompt, - "model": model - }, - "error": None, - "success": False, - "analysis_length": 0, - "model_used": model, - "image_size_bytes": 0 - } - - temp_image_path = None - # Track whether we should clean up the file after processing. - # Local files (e.g. from the image cache) should NOT be deleted. - should_cleanup = True - detected_mime_type = None - + debug_call_data = _debug_call_data("image", image_url, user_prompt, model) + + prepared: Optional[_PreparedImage] = None try: from tools.interrupt import is_interrupted if is_interrupted(): @@ -1483,139 +939,36 @@ async def vision_analyze_tool( logger.info("Analyzing image: %s", image_url[:60]) logger.info("User prompt: %s", user_prompt[:100]) - # Resolve the source to raw bytes through the single resolver (unifies - # data:/http/file/local/container and enforces terminal-backend - # confinement). Materialize to a temp file so the existing path-based - # encode/resize pipeline below is reused verbatim. - from tools.image_source import ( - ImageResolutionError, - ResolveContext, - resolve_image_source, - ) + prepared = await _prepare_image(image_url, task_id, region, validate_decode=False) + logger.info("Image ready (%.1f KB)", prepared.size_bytes / 1024) - try: - resolved = await resolve_image_source(image_url, ResolveContext(task_id=task_id)) - except ImageResolutionError as exc: - raise ValueError(str(exc)) - - detected_mime_type = resolved.mime - temp_dir = get_hermes_dir("cache/vision", "temp_vision_images") - temp_dir.mkdir(parents=True, exist_ok=True) - temp_image_path = temp_dir / f"temp_image_{uuid.uuid4()}.img" - await asyncio.to_thread(temp_image_path.write_bytes, resolved.data) - should_cleanup = True - - # Get image file size for logging - image_size_bytes = len(resolved.data) - image_size_kb = image_size_bytes / 1024 - logger.info("Image ready (%.1f KB)", image_size_kb) - # Normalize unsupported formats (SVG, BMP, ...) to PNG. Vision providers - # reject these media types; convert before encoding. Offloaded — the - # rasterizers/Pillow are blocking. - normalized_path, detected_mime_type, _norm_err = await asyncio.to_thread( - _normalize_to_supported_image, temp_image_path, detected_mime_type, - ) - if _norm_err or normalized_path is None: - raise ValueError(_norm_err or "Image normalization failed.") - if normalized_path != temp_image_path: - if should_cleanup and temp_image_path.exists(): - try: - temp_image_path.unlink() - except Exception: - pass - temp_image_path = normalized_path - should_cleanup = True - - # Optional region zoom: crop BEFORE the encode/downscale pipeline so - # the cropped area gets the full resolution budget. - _crop_offset: dict = {} - _scale_info: dict = {} - if region is not None: - cropped_path, cropped_mime, crop_err = await asyncio.to_thread( - _crop_image_region, temp_image_path, region, - offset_out=_crop_offset, - ) - if crop_err or cropped_path is None: - raise ValueError(crop_err or "Region crop failed.") - if should_cleanup and temp_image_path.exists(): - try: - temp_image_path.unlink() - except Exception: - pass - temp_image_path = cropped_path - detected_mime_type = cropped_mime - should_cleanup = True - - # Convert image to base64 — send at full resolution first. - # If the provider rejects it as too large, we auto-resize and retry. - # Offloaded to the bounded vision CPU executor so a fan-out of encodes - # can't saturate every core and starve the event loop. + # Send at full resolution first; on a size rejection, downscale and retry. logger.info("Converting image to base64...") image_data_url = await _run_encode_on_cpu_executor( - _image_to_base64_data_url, temp_image_path, mime_type=detected_mime_type) - data_size_kb = len(image_data_url) / 1024 - logger.info("Image converted to base64 (%.1f KB)", data_size_kb) + _image_to_base64_data_url, prepared.path, mime_type=prepared.mime) + logger.info("Image converted to base64 (%.1f KB)", len(image_data_url) / 1024) - # Hard limit (20 MB) — no provider accepts payloads this large. + _scale_info: dict = {} if len(image_data_url) > _MAX_BASE64_BYTES: - # Try to resize down to 5 MB before giving up. image_data_url = await _run_encode_on_cpu_executor( _resize_image_for_vision, - temp_image_path, mime_type=detected_mime_type, + prepared.path, mime_type=prepared.mime, scale_out=_scale_info) if len(image_data_url) > _MAX_BASE64_BYTES: - raise ValueError( - f"Image too large for vision API: base64 payload is " - f"{len(image_data_url) / (1024 * 1024):.1f} MB " - f"(limit {_MAX_BASE64_BYTES / (1024 * 1024):.0f} MB) " - f"even after resizing. " - f"Install Pillow (`pip install Pillow`) for better auto-resize, " - f"or compress the image manually." - ) + raise ValueError(_too_large_message(image_data_url)) + + debug_call_data["image_size_bytes"] = prepared.size_bytes + + messages = [{ + "role": "user", + "content": [ + {"type": "text", "text": user_prompt}, + {"type": "image_url", "image_url": {"url": image_data_url}}, + ], + }] - debug_call_data["image_size_bytes"] = image_size_bytes - - # Use the prompt as provided (model_tools.py now handles full description formatting) - comprehensive_prompt = user_prompt - - # Prepare the message with base64-encoded image - messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": comprehensive_prompt - }, - { - "type": "image_url", - "image_url": { - "url": image_data_url - } - } - ] - } - ] - logger.info("Processing image with vision model...") - - # Call the vision API via centralized router. - # Read timeout from config.yaml (auxiliary.vision.timeout), default 120s. - # Local vision models (llama.cpp, ollama) can take well over 30s. - vision_timeout = 120.0 - vision_temperature = 0.1 - try: - from hermes_cli.config import cfg_get, load_config - _cfg = load_config() - _vision_cfg = cfg_get(_cfg, "auxiliary", "vision", default={}) - _vt = _vision_cfg.get("timeout") - if _vt is not None: - vision_timeout = float(_vt) - _vtemp = _vision_cfg.get("temperature") - if _vtemp is not None: - vision_temperature = float(_vtemp) - except Exception: - pass + vision_timeout, vision_temperature = _read_vision_call_settings(120.0) call_kwargs = { "task": "vision", "messages": messages, @@ -1625,7 +978,6 @@ async def vision_analyze_tool( if model: call_kwargs["model"] = model _load_auxiliary_client() - # Try full-size image first; on size-related rejection, downscale and retry. try: response = await async_call_llm(**call_kwargs) except Exception as _api_err: @@ -1639,191 +991,80 @@ async def vision_analyze_tool( ) image_data_url = await _run_encode_on_cpu_executor( _resize_image_for_vision, - temp_image_path, mime_type=detected_mime_type, + prepared.path, mime_type=prepared.mime, scale_out=_scale_info) messages[0]["content"][1]["image_url"]["url"] = image_data_url response = await async_call_llm(**call_kwargs) else: raise - - # Extract the analysis — fall back to reasoning if content is empty - analysis = extract_content_or_reasoning(response) - # Retry once on empty content (reasoning-only response) + analysis = extract_content_or_reasoning(response) if not analysis: logger.warning("Vision LLM returned empty content, retrying once") response = await async_call_llm(**call_kwargs) analysis = extract_content_or_reasoning(response) analysis_length = len(analysis) - logger.info("Image analysis completed (%s characters)", analysis_length) - - # Prepare successful response + analysis = analysis or "There was a problem with the request and the image could not be analyzed." - scale_note = _build_scale_note( - _scale_info or None, _crop_offset or None, - ) + scale_note = _build_scale_note(_scale_info or None, prepared.crop_offset or None) result = { "success": True, "analysis": f"[{scale_note}] {analysis}" if scale_note else analysis, } if scale_note: result["scale_note"] = scale_note - + debug_call_data["success"] = True debug_call_data["analysis_length"] = analysis_length - - # Log debug information - _debug.log_call("vision_analyze_tool", debug_call_data) - _debug.save() - - return json.dumps(result, indent=2, ensure_ascii=False) - + return _finish_analysis("vision_analyze_tool", debug_call_data, result) + except Exception as e: error_msg = f"Error analyzing image: {str(e)}" logger.error("%s", error_msg, exc_info=True) - - # Detect vision capability errors — give the model a clear message - # so it can inform the user instead of a cryptic API error. - err_str = str(e).lower() - if any(hint in err_str for hint in ( - "402", "insufficient", "payment required", "credits", "billing", - )): - analysis = ( - "Insufficient credits or payment required. Please top up your " - f"API provider account and try again. Error: {e}" - ) - elif any(hint in err_str for hint in ( - "does not support", "not support image", - "content_policy", "multimodal", - "unrecognized request argument", "image input", - )): - analysis = ( - f"{model} does not support vision or our request was not " - f"accepted by the server. Error: {e}" - ) - elif "invalid_request" in err_str or "image_url" in err_str: - analysis = ( - "The vision API rejected the image. This can happen when the " - "image is in an unsupported format, corrupted, or still too " - "large after auto-resize. Try a smaller JPEG/PNG and retry. " - f"Error: {e}" - ) - else: - analysis = ( - "There was a problem with the request and the image could not " - f"be analyzed. Error: {e}" - ) - - # Prepare error response - result = { + analysis = _classify_analysis_error( + e, _IMAGE_ERROR_RULES, + "There was a problem with the request and the image could not " + "be analyzed. Error: {e}", + model=model, + ) + debug_call_data["error"] = error_msg + return _finish_analysis("vision_analyze_tool", debug_call_data, { "success": False, "error": error_msg, "analysis": analysis, - } - - debug_call_data["error"] = error_msg - _debug.log_call("vision_analyze_tool", debug_call_data) - _debug.save() - - return json.dumps(result, indent=2, ensure_ascii=False) - + }) + finally: - # Clean up temporary image file (but NOT local/cached files) - if should_cleanup and temp_image_path and temp_image_path.exists(): - try: - temp_image_path.unlink() - logger.debug("Cleaned up temporary image file") - except Exception as cleanup_error: - logger.warning( - "Could not delete temporary file: %s", cleanup_error, exc_info=True - ) + if prepared is not None: + _cleanup_temp_media(prepared.path, "image") def check_vision_requirements() -> bool: - """Check if the configured runtime vision path can resolve a client. + """True when ``call_llm(task="vision")`` could resolve a client. - Mirrors the fallback chain that ``call_llm(task="vision")`` actually uses - at runtime: first the explicit ``auxiliary.vision.provider`` (if any), - and if that fails, the auto chain (main provider → openrouter → nous). - Without the auto-fallback step the tool would disappear from the model's - tool list whenever the explicit provider name was unresolvable, even - when the auto chain would have served the request (issue #31179). + Mirrors its runtime fallback chain: explicit ``auxiliary.vision.provider``, + then the auto chain (main provider → openrouter → nous) — without the auto + step the tool would vanish whenever the explicit name was unresolvable. + Probe mode skips real SDK client construction (openai import + SSL setup) + on the tool-gating path; resolution policy is identical. """ try: from agent.auxiliary_client import aux_probe_mode, resolve_vision_provider_client except ImportError: return False try: - # Probe mode answers "is a vision client resolvable?" without paying - # for real SDK client construction (openai import + httpx/SSL setup) - # on the tool-gating path — resolution policy is identical. with aux_probe_mode(): _provider, client, _model = resolve_vision_provider_client() if client is not None: return True - # Same fallback to "auto" that call_llm performs when the configured - # provider can't be resolved. _provider, client, _model = resolve_vision_provider_client(provider="auto") return client is not None except Exception: return False - -if __name__ == "__main__": - """ - Simple test/demo when run directly - """ - print("👁️ Vision Tools Module") - print("=" * 40) - - # Check if vision model is available - api_available = check_vision_requirements() - - if not api_available: - print("❌ No auxiliary vision model available") - print("Configure a supported multimodal backend (OpenRouter, Nous, Codex, Anthropic, or a custom OpenAI-compatible endpoint).") - sys.exit(1) - else: - print("✅ Vision model available") - - print("🛠️ Vision tools ready for use!") - - # Show debug mode status - if _debug.active: - print(f"🐛 Debug mode ENABLED - Session ID: {_debug.session_id}") - print(f" Debug logs will be saved to: ./logs/vision_tools_debug_{_debug.session_id}.json") - else: - print("🐛 Debug mode disabled (set VISION_TOOLS_DEBUG=true to enable)") - - print("\nBasic usage:") - print(" from vision_tools import vision_analyze_tool") - print(" import asyncio") - print("") - print(" async def main():") - print(" result = await vision_analyze_tool(") - print(" image_url='https://example.com/image.jpg',") - print(" user_prompt='What do you see in this image?'") - print(" )") - print(" print(result)") - print(" asyncio.run(main())") - - print("\nExample prompts:") - print(" - 'What architectural style is this building?'") - print(" - 'Describe the emotions and mood in this image'") - print(" - 'What text can you read in this image?'") - print(" - 'Identify any safety hazards visible'") - print(" - 'What products or brands are shown?'") - - print("\nDebug mode:") - print(" # Enable debug logging") - print(" export VISION_TOOLS_DEBUG=true") - print(" # Debug logs capture all vision analysis calls and results") - print(" # Logs saved to: ./logs/vision_tools_debug_UUID.json") - - # --------------------------------------------------------------------------- # Registry # --------------------------------------------------------------------------- @@ -1831,12 +1072,9 @@ from tools.registry import registry, tool_error VISION_ANALYZE_SCHEMA = { "name": "vision_analyze", - # Dieted (#95681): routing mechanics (native attach vs aux-model text - # fallback) removed — the route is automatic and the native path's own - # tool result says "you can see it natively now"; the schema doesn't - # need to predict plumbing. Region keeps its flow teaching: it's - # pre-effect guidance (a model that doesn't know crops keep full - # resolution never zooms). + # Routing mechanics are deliberately absent (the route is automatic and the + # native result says so itself); region keeps its pre-effect guidance — a + # model that doesn't know crops keep full resolution never zooms. "description": ( "Load an image into the conversation so you can see it. Call it " "any time the user references an image — then answer from what " @@ -1872,23 +1110,38 @@ VISION_ANALYZE_SCHEMA = { } +def _configured_aux_model(sections: tuple, env_vars: tuple) -> Optional[str]: + """First non-empty ``auxiliary.
.model`` from config.yaml, else the + first non-empty env var (legacy override), else None.""" + try: + from hermes_cli.config import cfg_get, load_config + _cfg = load_config() + for section in sections: + _vmodel = cfg_get(_cfg, "auxiliary", section, "model") + if _vmodel: + model = str(_vmodel).strip() or None + if model: + return model + break + except Exception: + pass + for env_var in env_vars: + val = os.getenv(env_var, "").strip() + if val: + return val + return None + + async def _handle_vision_analyze(args: Dict[str, Any], **kw: Any) -> str: image_url = args.get("image_url", "") question = args.get("question", "") region = args.get("region") task_id = kw.get("task_id") - # The fan-out cap lives inside the encode/resize step (offloaded to the - # bounded _vision_cpu_executor), NOT around the whole analysis — so a - # legitimate multi-image workflow keeps full request concurrency while the - # CPU bursts that actually starve the loop are bounded to host cores. - # - # Fast path: when native image routing is in effect for the active main - # model (provider accepts images in tool results, or the user set the - # model.supports_vision override), short-circuit the auxiliary LLM and - # return the image bytes as a multimodal tool-result envelope. The main - # model sees the pixels directly on its next turn — no aux call, no - # information loss, no extra latency. + # No concurrency gate around the whole analysis — the CPU burst is bounded + # inside the encode/resize step, so multi-image fan-out keeps full request + # concurrency. Native fast path: main model sees the pixels directly, no + # aux call, no information loss. if _should_use_native_vision_fast_path(): logger.info("vision_analyze: native fast path") return await _vision_analyze_native(image_url, question, task_id=task_id, region=region) @@ -1898,18 +1151,7 @@ async def _handle_vision_analyze(args: Dict[str, Any], **kw: Any) -> str: "Fully describe and explain everything about this image, then answer the " f"following question:\n\n{question}" ) - # Prefer config.yaml auxiliary.vision.model; env var is a legacy override. - model = None - try: - from hermes_cli.config import cfg_get, load_config - _cfg = load_config() - _vmodel = cfg_get(_cfg, "auxiliary", "vision", "model") - if _vmodel: - model = str(_vmodel).strip() or None - except Exception: - pass - if not model: - model = os.getenv("AUXILIARY_VISION_MODEL", "").strip() or None + model = _configured_aux_model(("vision",), ("AUXILIARY_VISION_MODEL",)) return await vision_analyze_tool(image_url, full_prompt, model, task_id=task_id, region=region) @@ -1944,41 +1186,35 @@ _VIDEO_SIZE_WARN_BYTES = 20 * 1024 * 1024 def _detect_video_mime_type(video_path: Path) -> Optional[str]: - """Return a video MIME type based on file extension, or None if unsupported.""" - ext = video_path.suffix.lower() - return _VIDEO_MIME_TYPES.get(ext) + """Video MIME type from extension, or None if unsupported.""" + return _VIDEO_MIME_TYPES.get(video_path.suffix.lower()) + + +def _unsupported_video_format(suffix: str) -> str: + return ( + f"Unsupported video format: '{suffix}'. " + f"Supported: {', '.join(sorted(_VIDEO_MIME_TYPES.keys()))}" + ) def _video_to_base64_data_url(video_path: Path, mime_type: Optional[str] = None) -> str: - """Convert a video file to a base64-encoded data URL.""" - data = video_path.read_bytes() - encoded = base64.b64encode(data).decode("ascii") + encoded = base64.b64encode(video_path.read_bytes()).decode("ascii") mime = mime_type or _VIDEO_MIME_TYPES.get(video_path.suffix.lower(), "video/mp4") return f"data:{mime};base64,{encoded}" -def _terminal_backend_is_local() -> bool: - backend = os.getenv("TERMINAL_ENV", "local").strip().lower() - return backend in ("", "local") - - def _is_path_like_video_source(value: str) -> bool: lowered = (value or "").strip().lower() - if not lowered: - return False - return not lowered.startswith(("http://", "https://", "data:")) + return bool(lowered) and not lowered.startswith(("http://", "https://", "data:")) async def _materialize_video_from_terminal_backend(video_source: str, task_id: Optional[str]) -> Path: """Read a path via the shared media resolver into a local temp video file. - Routes through :func:`tools.image_source.resolve_image_source` with - ``permitted=("video",)`` so terminal-backend video reads get the exact - pipeline vision_analyze uses: media-cache host reads (gateway-downloaded - videos live on the host, not in the sandbox), bounded in-sandbox exec-read - (``head -c`` cap — no unbounded base64 stream, no python3 dependency in - the sandbox image), lazy env bring-up (#62825), the credential-read - guard, and the 50MB ingest cap. + ``permitted=("video",)`` gives terminal-backend video reads the exact + pipeline vision_analyze uses: media-cache host reads (gateway downloads live + on the host, not in the sandbox), bounded in-sandbox exec-read, lazy env + bring-up, the credential-read guard, and the 50MB ingest cap. """ from tools.image_source import ImageResolutionError, ResolveContext, resolve_image_source @@ -1987,10 +1223,7 @@ async def _materialize_video_from_terminal_backend(video_source: str, task_id: O source = source[len("file://"):] suffix = Path(source).suffix.lower() if suffix not in _VIDEO_MIME_TYPES: - raise ValueError( - f"Unsupported video format: '{suffix}'. " - f"Supported: {', '.join(sorted(_VIDEO_MIME_TYPES.keys()))}" - ) + raise ValueError(_unsupported_video_format(suffix)) try: resolved = await resolve_image_source( @@ -2007,63 +1240,12 @@ async def _materialize_video_from_terminal_backend(video_source: str, task_id: O async def _download_video(video_url: str, destination: Path, max_retries: int = 3) -> Path: - """Download video from URL with SSRF protection and retry.""" - import asyncio - - destination.parent.mkdir(parents=True, exist_ok=True) - - async def _ssrf_redirect_guard(response): - from tools.url_safety import async_is_safe_url, redirect_target_from_response - redirect_url = redirect_target_from_response(response) - if redirect_url and not await async_is_safe_url(redirect_url): - raise ValueError( - f"Blocked redirect to private/internal address: {redirect_url}" - ) - - last_error = None - for attempt in range(max_retries): - try: - blocked = check_website_access(video_url) - if blocked: - raise PermissionError(blocked["message"]) - - from tools.url_safety import create_ssrf_safe_async_client - - async with create_ssrf_safe_async_client( - timeout=60.0, - follow_redirects=True, - event_hooks={"response": [_ssrf_redirect_guard]}, - ) as client: - await _stream_download_to_file( - client, - video_url, - destination, - _MAX_VIDEO_BASE64_BYTES, - headers={ - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", - "Accept": "video/*,*/*;q=0.8", - }, - media_label="Video", - ) - - return destination - except Exception as e: - last_error = e - if attempt < max_retries - 1: - wait_time = 2 ** (attempt + 1) - logger.warning("Video download failed (attempt %s/%s): %s", attempt + 1, max_retries, str(e)[:50]) - await asyncio.sleep(wait_time) - else: - logger.error( - "Video download failed after %s attempts: %s", - max_retries, str(e)[:100], exc_info=True, - ) - - if last_error is None: - raise RuntimeError( - f"_download_video exited retry loop without attempting (max_retries={max_retries})" - ) - raise last_error + """Download video with SSRF protection; every failure class is retried.""" + return await _download_media( + video_url, destination, max_retries, + media_label="Video", accept="video/*,*/*;q=0.8", + max_bytes=_MAX_VIDEO_BASE64_BYTES, timeout=60.0, retry_all=True, + ) async def video_analyze_tool( @@ -2075,18 +1257,7 @@ async def video_analyze_tool( """Analyze a video via multimodal LLM. Returns JSON {success, analysis}.""" if not isinstance(user_prompt, str): user_prompt = str(user_prompt) if user_prompt is not None else "" - debug_call_data = { - "parameters": { - "video_url": video_url, - "user_prompt": user_prompt[:200] + "..." if len(user_prompt) > 200 else user_prompt, - "model": model, - }, - "error": None, - "success": False, - "analysis_length": 0, - "model_used": model, - "video_size_bytes": 0, - } + debug_call_data = _debug_call_data("video", video_url, user_prompt, model) temp_video_path = None should_cleanup = True @@ -2099,16 +1270,15 @@ async def video_analyze_tool( logger.info("Analyzing video: %s", video_url[:60]) logger.info("User prompt: %s", user_prompt[:100]) - # Resolve local path vs remote URL resolved_url = video_url if resolved_url.startswith("file://"): resolved_url = resolved_url[len("file://"):] local_path = Path(os.path.expanduser(resolved_url)) - if not _terminal_backend_is_local() and _is_path_like_video_source(video_url): + from tools.image_source import _is_local_terminal_backend + if not _is_local_terminal_backend() and _is_path_like_video_source(video_url): logger.info("Reading video source via terminal backend: %s", video_url) temp_video_path = await _materialize_video_from_terminal_backend(video_url, task_id) - should_cleanup = True elif local_path.is_file(): from agent.file_safety import raise_if_read_blocked raise_if_read_blocked(str(local_path)) @@ -2122,7 +1292,6 @@ async def video_analyze_tool( temp_dir = get_hermes_dir("cache/video", "temp_video_files") temp_video_path = temp_dir / f"temp_video_{uuid.uuid4()}.mp4" await _download_video(video_url, temp_video_path) - should_cleanup = True else: raise ValueError( "Invalid video source. Provide an HTTP/HTTPS URL or a valid local file path." @@ -2134,59 +1303,30 @@ async def video_analyze_tool( detected_mime = _detect_video_mime_type(temp_video_path) if not detected_mime: - raise ValueError( - f"Unsupported video format: '{temp_video_path.suffix}'. " - f"Supported: {', '.join(sorted(_VIDEO_MIME_TYPES.keys()))}" - ) + raise ValueError(_unsupported_video_format(temp_video_path.suffix)) if video_size_bytes > _VIDEO_SIZE_WARN_BYTES: logger.warning("Video is %.1f MB — may be slow or rejected", video_size_mb) video_data_url = _video_to_base64_data_url(temp_video_path, mime_type=detected_mime) - data_size_mb = len(video_data_url) / (1024 * 1024) - if len(video_data_url) > _MAX_VIDEO_BASE64_BYTES: raise ValueError( - f"Video too large for API: base64 payload is {data_size_mb:.1f} MB " + f"Video too large for API: base64 payload is {len(video_data_url) / (1024 * 1024):.1f} MB " f"(limit {_MAX_VIDEO_BASE64_BYTES / (1024 * 1024):.0f} MB). " f"Compress or trim the video and retry." ) debug_call_data["video_size_bytes"] = video_size_bytes - messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": user_prompt, - }, - { - "type": "video_url", - "video_url": { - "url": video_data_url, - }, - }, - ], - } - ] - - vision_timeout = 180.0 - vision_temperature = 0.1 - try: - from hermes_cli.config import cfg_get, load_config - _cfg = load_config() - _vision_cfg = cfg_get(_cfg, "auxiliary", "vision", default={}) - _vt = _vision_cfg.get("timeout") - if _vt is not None: - vision_timeout = max(float(_vt), 180.0) - _vtemp = _vision_cfg.get("temperature") - if _vtemp is not None: - vision_temperature = float(_vtemp) - except Exception: - pass + messages = [{ + "role": "user", + "content": [ + {"type": "text", "text": user_prompt}, + {"type": "video_url", "video_url": {"url": video_data_url}}, + ], + }] + vision_timeout, vision_temperature = _read_vision_call_settings(180.0, min_timeout=180.0) call_kwargs = { "task": "vision", "messages": messages, @@ -2196,14 +1336,7 @@ async def video_analyze_tool( if model: call_kwargs["model"] = model - _load_auxiliary_client() - response = await async_call_llm(**call_kwargs) - analysis = extract_content_or_reasoning(response) - - if not analysis: - logger.warning("Empty video response, retrying once") - response = await async_call_llm(**call_kwargs) - analysis = extract_content_or_reasoning(response) + analysis = await _call_vision_llm(call_kwargs, "Empty video response, retrying once") analysis_length = len(analysis) if analysis else 0 logger.info("Video analysis completed (%s characters)", analysis_length) @@ -2212,72 +1345,28 @@ async def video_analyze_tool( "success": True, "analysis": analysis or "There was a problem with the request and the video could not be analyzed.", } - debug_call_data["success"] = True debug_call_data["analysis_length"] = analysis_length - _debug.log_call("video_analyze_tool", debug_call_data) - _debug.save() - - return json.dumps(result, indent=2, ensure_ascii=False) + return _finish_analysis("video_analyze_tool", debug_call_data, result) except Exception as e: error_msg = f"Error analyzing video: {str(e)}" logger.error("%s", error_msg, exc_info=True) - - err_str = str(e).lower() - if any(hint in err_str for hint in ( - "402", "insufficient", "payment required", "credits", "billing", - )): - analysis = ( - "Insufficient credits or payment required. Please top up your " - f"API provider account and try again. Error: {e}" - ) - elif any(hint in err_str for hint in ( - "does not support", "not support video", - "content_policy", "multimodal", - "unrecognized request argument", "video input", - "video_url", - )): - analysis = ( - f"The model does not support video analysis or the request was " - f"rejected. Ensure you're using a video-capable model " - f"(e.g. google/gemini-2.5-flash). Error: {e}" - ) - elif any(hint in err_str for hint in ( - "too large", "payload", "413", "content_too_large", - "request_too_large", "exceeds", "size limit", - )): - analysis = ( - "The video is too large for the API. Try compressing or trimming " - f"the video (max ~50 MB). Error: {e}" - ) - else: - analysis = ( - "There was a problem with the request and the video could not " - f"be analyzed. Error: {e}" - ) - - result = { + analysis = _classify_analysis_error( + e, _VIDEO_ERROR_RULES, + "There was a problem with the request and the video could not " + "be analyzed. Error: {e}", + ) + debug_call_data["error"] = error_msg + return _finish_analysis("video_analyze_tool", debug_call_data, { "success": False, "error": error_msg, "analysis": analysis, - } - - debug_call_data["error"] = error_msg - _debug.log_call("video_analyze_tool", debug_call_data) - _debug.save() - - return json.dumps(result, indent=2, ensure_ascii=False) + }) finally: - if should_cleanup and temp_video_path and temp_video_path.exists(): - try: - temp_video_path.unlink() - logger.debug("Cleaned up temporary video file") - except Exception as cleanup_error: - logger.warning( - "Could not delete temporary file: %s", cleanup_error, exc_info=True - ) + if should_cleanup: + _cleanup_temp_media(temp_video_path, "video") VIDEO_ANALYZE_SCHEMA = { @@ -2314,19 +1403,9 @@ def _handle_video_analyze(args: Dict[str, Any], **kw: Any) -> Awaitable[str]: "including visual content, motion, audio cues, text overlays, and scene " f"transitions. Then answer the following question:\n\n{question}" ) - # Prefer config.yaml auxiliary.video.model (falling back to vision); - # env vars are a legacy override. - model = None - try: - from hermes_cli.config import cfg_get, load_config - _cfg = load_config() - _vmodel = cfg_get(_cfg, "auxiliary", "video", "model") or cfg_get(_cfg, "auxiliary", "vision", "model") - if _vmodel: - model = str(_vmodel).strip() or None - except Exception: - pass - if not model: - model = os.getenv("AUXILIARY_VIDEO_MODEL", "").strip() or os.getenv("AUXILIARY_VISION_MODEL", "").strip() or None + model = _configured_aux_model( + ("video", "vision"), ("AUXILIARY_VIDEO_MODEL", "AUXILIARY_VISION_MODEL"), + ) return video_analyze_tool(video_url, full_prompt, model, task_id=kw.get("task_id")) diff --git a/tools/vision_tools_image_prep.py b/tools/vision_tools_image_prep.py new file mode 100644 index 0000000000..2c67d93e58 --- /dev/null +++ b/tools/vision_tools_image_prep.py @@ -0,0 +1,313 @@ +"""Image format detection, normalization and region cropping for vision tools. + +Everything here runs BEFORE an image is base64-embedded. A vision tool result +is baked into immutable conversation history and re-sent every turn, so an +unsupported media type or corrupt bytes would wedge the session with a +non-retryable 400 on every resume — normalization must happen up front. +""" + +from __future__ import annotations + +import logging +import uuid +from io import BytesIO +from pathlib import Path +from typing import Any, Optional + +from hermes_constants import get_hermes_dir + +logger = logging.getLogger("tools.vision_tools") + +_EXTENSION_MIME_TYPES = { + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".png": "image/png", + ".gif": "image/gif", + ".bmp": "image/bmp", + ".webp": "image/webp", + ".svg": "image/svg+xml", +} + +# Media types the major vision providers (Anthropic in particular) accept +# inline. SVG/BMP/TIFF are rejected with a non-retryable 400. +_ANTHROPIC_SUPPORTED_MEDIA_TYPES = frozenset( + {"image/jpeg", "image/png", "image/gif", "image/webp"} +) + + +def _determine_mime_type(image_path: Path) -> str: + """MIME type from file extension (defaults to image/jpeg).""" + return _EXTENSION_MIME_TYPES.get(image_path.suffix.lower(), "image/jpeg") + + +def _detect_image_mime_type_from_bytes(data: bytes) -> Optional[str]: + """Magic-byte MIME sniff (authoritative; no extension trust). + + Returns ``None`` for anything without a recognized header — including SVG, + which has no magic bytes (the resolver sniffs ``= 12 and header[:4] == b"RIFF" and header[8:12] == b"WEBP": + return "image/webp" + return None + + +def _supported_media_types() -> frozenset: + """Formats the ACTIVE main model's server can decode. + + The managed llama-server decodes with stb_image — no WebP — and an + undecodable image part fails SILENTLY (the model confabulates), so the set + is narrowed there and normalization converts those formats to PNG. + """ + try: + from agent.auxiliary_client import _runtime_main_value + from hermes_cli.local_runtime.capabilities import ( + ACCEPTED_IMAGE_MIMES, + is_managed_provider, + ) + + if is_managed_provider( + str(_runtime_main_value("provider") or ""), + str(_runtime_main_value("base_url") or "")): + return ACCEPTED_IMAGE_MIMES + except Exception: # noqa: BLE001 — best-effort narrowing only + pass + return _ANTHROPIC_SUPPORTED_MEDIA_TYPES + + +def _rasterize_svg_to_png(svg_path: Path, out_path: Path) -> bool: + """Best-effort SVG → PNG via cairosvg, svglib+reportlab, rsvg-convert, inkscape (all soft deps).""" + def _ok() -> bool: + return out_path.exists() and out_path.stat().st_size > 0 + + try: + import cairosvg # type: ignore + cairosvg.svg2png(url=str(svg_path), write_to=str(out_path)) + return _ok() + except Exception: + pass + try: + from svglib.svglib import svg2rlg # type: ignore + from reportlab.graphics import renderPM # type: ignore + drawing = svg2rlg(str(svg_path)) + if drawing is not None: + renderPM.drawToFile(drawing, str(out_path), fmt="PNG") + return _ok() + except Exception: + pass + import shutil + import subprocess + for cmd in ( + ["rsvg-convert", "-o", str(out_path), str(svg_path)], + ["inkscape", str(svg_path), "--export-type=png", + f"--export-filename={out_path}"], + ): + if shutil.which(cmd[0]): + try: + subprocess.run( + cmd, check=True, capture_output=True, timeout=30, + stdin=subprocess.DEVNULL, + ) + if _ok(): + return True + except Exception: + continue + return False + + +def _normalize_to_supported_image( + image_path: Path, detected_mime: str +) -> tuple[Optional[Path], Optional[str], Optional[str]]: + """Ensure an image is in a provider-supported format. + + Returns ``(path, mime, error)``: the input unchanged when already + supported; ``(new_png_path, "image/png", None)`` after conversion — a temp + file the CALLER must clean up; ``(None, None, message)`` when impossible. + SVG is rasterized; other Pillow-readable rasters (BMP, TIFF) re-encode to PNG. + """ + if detected_mime in _supported_media_types(): + return image_path, detected_mime, None + + out_dir = get_hermes_dir("cache/vision", "temp_vision_images") + out_dir.mkdir(parents=True, exist_ok=True) + out_path = out_dir / f"converted_{uuid.uuid4()}.png" + + if detected_mime == "image/svg+xml": + if _rasterize_svg_to_png(image_path, out_path): + return out_path, "image/png", None + return ( + None, + None, + "This is an SVG, which vision models cannot read directly, and no " + "SVG rasterizer is installed (tried cairosvg, svglib, rsvg-convert, " + "inkscape). Convert the SVG to PNG first — e.g. open it in a browser " + "and screenshot it, or install a rasterizer " + "(`pip install cairosvg`) — then re-run vision_analyze on the PNG.", + ) + + try: + from PIL import Image as _PILImage + with _PILImage.open(image_path) as _img: + if _img.mode not in ("RGB", "RGBA", "L"): + _img = _img.convert("RGBA") + _img.save(out_path, format="PNG") + if out_path.exists() and out_path.stat().st_size > 0: + return out_path, "image/png", None + except Exception as _exc: + logger.warning("Failed to normalize %s image to PNG: %s", + detected_mime, _exc) + return ( + None, + None, + f"Image format {detected_mime!r} is not supported by the vision API " + f"and could not be converted to PNG (install Pillow for raster " + f"conversion). Convert it to PNG or JPEG and try again.", + ) + + +# Full raster validation runs on untrusted images in a shared CPU executor: +# bound animated-image work by frame count AND total decoded area so a compact +# file cannot monopolize a worker with unbounded frames. +_VISION_MAX_VALIDATED_FRAME_COUNT = 100 +_VISION_MAX_VALIDATED_AGGREGATE_PIXELS = 100_000_000 + + +def _validate_raster_image_decodable( + image_path: Path, + max_frames: int = _VISION_MAX_VALIDATED_FRAME_COUNT, + max_pixels: int = _VISION_MAX_VALIDATED_AGGREGATE_PIXELS, +) -> Optional[str]: + """Return an error unless Pillow can fully decode every frame. + + Header sniffing and ``Image.open`` only inspect containers: a timed-out + download can look like a valid PNG with a truncated pixel stream. Without + Pillow the image passes unvalidated rather than rejecting everything. + """ + try: + from PIL import Image as _PILImage + from PIL import ImageSequence as _PILImageSequence + except ImportError: + return None + try: + with _PILImage.open(image_path) as image: + image.verify() + with _PILImage.open(image_path) as image: + validated_pixels = 0 + for frame_number, frame in enumerate( + _PILImageSequence.Iterator(image), start=1 + ): + if frame_number > max_frames: + return ( + "Image validation rejected animation: " + f"frame {frame_number} exceeds the maximum " + f"{max_frames} validated frames." + ) + next_validated_pixels = validated_pixels + frame.width * frame.height + if next_validated_pixels > max_pixels: + return ( + "Image validation rejected animation: aggregate decoded " + f"pixel count would reach {next_validated_pixels} at frame " + f"{frame_number}, exceeding the maximum " + f"{max_pixels}." + ) + frame.load() + validated_pixels = next_validated_pixels + except Exception as exc: + return f"Image could not be fully decoded: {exc}" + return None + + +def _image_exceeds_dimension(image_path: Path, max_dimension: int) -> bool: + """True if the longest side exceeds ``max_dimension`` px. + + Anthropic enforces an 8000px per-side cap independently of the byte cap. + Returns False (no forced resize) without Pillow or on unreadable files — + a missing soft dependency must never break the embed path. + """ + try: + from PIL import Image as _PILImage + with _PILImage.open(image_path) as _img: + return max(_img.size) > max_dimension + except Exception: + return False + + +def _crop_image_region( + image_path: Path, + region: Any, + offset_out: Optional[dict] = None, +) -> tuple[Optional[Path], Optional[str], Optional[str]]: + """Crop to ``region`` = [x1, y1, x2, y2] (original-image pixels). + + Applied BEFORE downscaling so the crop gets the full resolution budget. + Coordinates clamp to the image bounds; a zero-area/inverted region is + rejected with an error naming the real dimensions. Returns + ``(cropped_temp_path, mime, None)`` — caller owns cleanup — or + ``(None, None, error)``. Ported from QwenLM/qwen-code zoom-image.ts (Apache-2.0). + """ + try: + from PIL import Image + except ImportError: + return None, None, ( + "region cropping requires Pillow (`pip install Pillow`); " + "retry without the region parameter." + ) + + if ( + not isinstance(region, (list, tuple)) + or len(region) != 4 + or not all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in region) + ): + return None, None, ( + "Invalid region: expected [x1, y1, x2, y2] as four numbers " + "(pixel coordinates in the original image)." + ) + + try: + with Image.open(image_path) as img: + width, height = img.size + x1, y1, x2, y2 = (int(v) for v in region) + cx1 = max(0, min(x1, width)) + cy1 = max(0, min(y1, height)) + cx2 = max(0, min(x2, width)) + cy2 = max(0, min(y2, height)) + if cx2 <= cx1 or cy2 <= cy1: + return None, None, ( + f"Invalid region [{x1}, {y1}, {x2}, {y2}]: crops to zero " + f"area after clamping to the image bounds. The image is " + f"{width}x{height} px — pick x1