refactor(tools/media): extract vision_tools_image_prep; dedupe image/video generation providers; compact fal/xai helpers

This commit is contained in:
Teknium
2026-09-02 13:55:34 -07:00
parent 33ae442e94
commit 9fe009f467
7 changed files with 1612 additions and 2810 deletions
+8 -23
View File
@@ -1,28 +1,13 @@
#!/usr/bin/env python3
"""Leave a mark on the page in the Hermes desktop GUI's in-app browser.
"""Persistent element annotations in the Hermes desktop GUI's in-app browser.
``drive_preview`` already draws every move it makes — the field it can reach,
a box round its target, the cursor going there — but those are transients:
each one stands for a single action and retires itself. That is right for
narrating a click and no use at all for holding a finding on screen.
This is the deliberate one. An annotation outlines an element — or, with
``hold``, the entire visible field at once — and stays until the agent takes it
down, so it can show the user what it found, flag the fields
it is about to fill, or keep its place while it works elsewhere on the page.
Named for TouchDesigner's Annotate — the labelled box you drop around part of a
network to call it out.
Annotations are bound to elements, not coordinates: they ride scrolls and
reflows, and they go when their element does, so a navigation clears them
without the agent having to.
Rides the same ``preview.act`` bridge as ``drive_preview`` rather than opening
a second channel — the renderer already resolves ``@e`` refs and owns the
overlay, so this is one more verb on a wire that exists.
Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for
desktop-sourced sessions.
``drive_preview`` draws transient marks (one per action, self-retiring). An
annotation outlines an element — or, with ``hold``, the whole visible field —
and stays until the agent removes it. Annotations bind to elements, not
coordinates: they ride scrolls/reflows and vanish with their element, so a
navigation clears them. Rides the same ``preview.act`` bridge as
``drive_preview`` (the renderer resolves ``@e`` refs and owns the overlay).
Lives in the ``desktop_ui`` toolset, enabled only for desktop-sourced sessions.
"""
import json
+272 -475
View File
@@ -11,7 +11,7 @@ faces when default-on, so upscaling is strictly per-call opt-in.
Pricing strings are as-of-commit and allowed to drift.
"""
from typing import Any, Dict
from typing import Any, Dict, Optional
_PRESET_SIZES = {
"landscape": "landscape_16_9",
@@ -19,533 +19,330 @@ _PRESET_SIZES = {
"portrait": "portrait_16_9",
}
_ASPECT_SIZES = {"landscape": "16:9", "square": "1:1", "portrait": "9:16"}
_DEFAULT_SIZES = {"image_size_preset": _PRESET_SIZES, "aspect_ratio": _ASPECT_SIZES}
def _model(
display: str,
speed: str,
strengths: str,
price: str,
*,
style: str = "image_size_preset",
sizes: Optional[Dict[str, Any]] = None,
defaults: Dict[str, Any],
supports: set,
edit_endpoint: Optional[str] = None,
edit_supports: Optional[set] = None,
max_reference_images: Optional[int] = None,
) -> Dict[str, Any]:
"""Build one catalog entry; edit keys are present only for edit-capable models."""
entry: Dict[str, Any] = {
"display": display,
"speed": speed,
"strengths": strengths,
"price": price,
"size_style": style,
"sizes": sizes if sizes is not None else _DEFAULT_SIZES[style],
"defaults": defaults,
"supports": supports,
"upscale": False,
}
if edit_endpoint:
entry["edit_endpoint"] = edit_endpoint
entry["edit_supports"] = edit_supports
entry["max_reference_images"] = max_reference_images
return entry
FAL_MODELS: Dict[str, Dict[str, Any]] = {
"fal-ai/flux-2/klein/9b": {
"display": "FLUX 2 Klein 9B",
"speed": "<1s",
"strengths": "Fast, crisp text",
"price": "$0.006/MP",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"num_inference_steps": 4,
"output_format": "png",
"enable_safety_checker": False,
"fal-ai/flux-2/klein/9b": _model(
"FLUX 2 Klein 9B", "<1s", "Fast, crisp text", "$0.006/MP",
defaults={
"num_inference_steps": 4, "output_format": "png", "enable_safety_checker": False,
},
"supports": {
"prompt", "image_size", "num_inference_steps", "seed",
"output_format", "enable_safety_checker",
supports={
"prompt", "image_size", "num_inference_steps", "seed", "output_format", "enable_safety_checker",
},
"upscale": False,
# Image-to-image / editing: FLUX.2 [klein] 9B edit endpoint takes
# `image_urls` (list). Natural-language edits, multi-ref.
"edit_endpoint": "fal-ai/flux-2/klein/9b/edit",
"edit_supports": {
"prompt", "image_urls", "num_inference_steps", "seed",
"output_format", "enable_safety_checker",
edit_endpoint="fal-ai/flux-2/klein/9b/edit",
edit_supports={
"prompt", "image_urls", "num_inference_steps", "seed", "output_format", "enable_safety_checker",
},
"max_reference_images": 9,
},
"fal-ai/flux-2-pro": {
"display": "FLUX 2 Pro",
"speed": "~6s",
"strengths": "Studio photorealism",
"price": "$0.03/MP",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"num_inference_steps": 50,
"guidance_scale": 4.5,
"num_images": 1,
"output_format": "png",
"enable_safety_checker": False,
"safety_tolerance": "5",
max_reference_images=9,
),
"fal-ai/flux-2-pro": _model(
"FLUX 2 Pro", "~6s", "Studio photorealism", "$0.03/MP",
defaults={
"num_inference_steps": 50, "guidance_scale": 4.5, "num_images": 1,
"output_format": "png", "enable_safety_checker": False, "safety_tolerance": "5",
"sync_mode": True,
},
"supports": {
"prompt", "image_size", "num_inference_steps", "guidance_scale",
"num_images", "output_format", "enable_safety_checker",
"safety_tolerance", "sync_mode", "seed",
supports={
"prompt", "image_size", "num_inference_steps", "guidance_scale", "num_images", "output_format",
"enable_safety_checker", "safety_tolerance", "sync_mode", "seed",
},
"upscale": False,
# Edit endpoint accepts up to 9 reference images.
"edit_endpoint": "fal-ai/flux-2-pro/edit",
"edit_supports": {
"prompt", "image_urls", "num_inference_steps", "guidance_scale",
"num_images", "output_format", "enable_safety_checker",
"safety_tolerance", "sync_mode", "seed",
edit_endpoint="fal-ai/flux-2-pro/edit",
edit_supports={
"prompt", "image_urls", "num_inference_steps", "guidance_scale", "num_images", "output_format",
"enable_safety_checker", "safety_tolerance", "sync_mode", "seed",
},
"max_reference_images": 9,
},
"fal-ai/z-image/turbo": {
"display": "Z-Image Turbo",
"speed": "~2s",
"strengths": "Bilingual EN/CN, 6B",
"price": "$0.005/MP",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"num_inference_steps": 8,
"num_images": 1,
"output_format": "png",
"enable_safety_checker": False,
"enable_prompt_expansion": False, # avoid the extra per-request charge
max_reference_images=9,
),
"fal-ai/z-image/turbo": _model(
"Z-Image Turbo", "~2s", "Bilingual EN/CN, 6B", "$0.005/MP",
defaults={ # prompt expansion off: avoids the extra per-request charge
"num_inference_steps": 8, "num_images": 1, "output_format": "png",
"enable_safety_checker": False, "enable_prompt_expansion": False,
},
"supports": {
"prompt", "image_size", "num_inference_steps", "num_images",
"seed", "output_format", "enable_safety_checker",
"enable_prompt_expansion",
supports={
"prompt", "image_size", "num_inference_steps", "num_images", "seed", "output_format",
"enable_safety_checker", "enable_prompt_expansion",
},
"upscale": False,
},
"fal-ai/nano-banana-pro": {
"display": "Nano Banana Pro (Gemini 3 Pro Image)",
"speed": "~8s",
"strengths": "Gemini 3 Pro, reasoning depth, text rendering",
"price": "$0.15/image (1K)",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"num_images": 1,
"output_format": "png",
"safety_tolerance": "5",
# "1K" is the cheapest tier; 4K doubles the per-image cost.
# Users on Nous Subscription should stay at 1K for predictable billing.
),
"fal-ai/nano-banana-pro": _model(
"Nano Banana Pro (Gemini 3 Pro Image)", "~8s", "Gemini 3 Pro, reasoning depth, text rendering", "$0.15/image (1K)",
style="aspect_ratio",
# "1K" is the cheapest tier; 4K doubles the per-image cost (Nous Subscription billing).
defaults={
"num_images": 1, "output_format": "png", "safety_tolerance": "5",
"resolution": "1K",
},
"supports": {
"prompt", "aspect_ratio", "num_images", "output_format",
"safety_tolerance", "seed", "sync_mode", "resolution",
"enable_web_search", "limit_generations",
},
"upscale": False,
# Nano Banana Pro edit (Gemini 3 Pro Image): natural-language edits
# with up to 2 reference images via `image_urls`.
"edit_endpoint": "fal-ai/nano-banana-pro/edit",
"edit_supports": {
"prompt", "image_urls", "aspect_ratio", "num_images",
"output_format", "safety_tolerance", "seed", "sync_mode",
supports={
"prompt", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", "sync_mode",
"resolution", "enable_web_search", "limit_generations",
},
"max_reference_images": 2,
},
"fal-ai/nano-banana-2": {
"display": "Nano Banana 2 (Gemini 3.1 Flash Image)",
"speed": "~3s",
"strengths": "Fast reasoning, multilingual text, infographics",
"price": "Lower-cost Flash tier",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"num_images": 1,
"output_format": "png",
"safety_tolerance": "4",
"resolution": "1K",
"limit_generations": True,
edit_endpoint="fal-ai/nano-banana-pro/edit",
edit_supports={
"prompt", "image_urls", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed",
"sync_mode", "resolution", "enable_web_search", "limit_generations",
},
"supports": {
"prompt", "aspect_ratio", "num_images", "output_format",
"safety_tolerance", "seed", "sync_mode", "system_prompt",
"resolution", "enable_web_search", "limit_generations",
max_reference_images=2,
),
"fal-ai/nano-banana-2": _model(
"Nano Banana 2 (Gemini 3.1 Flash Image)", "~3s", "Fast reasoning, multilingual text, infographics", "Lower-cost Flash tier",
style="aspect_ratio",
defaults={
"num_images": 1, "output_format": "png", "safety_tolerance": "4",
"resolution": "1K", "limit_generations": True,
},
supports={
"prompt", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", "sync_mode",
"system_prompt", "resolution", "enable_web_search", "limit_generations", "thinking_level",
},
edit_endpoint="fal-ai/nano-banana-2/edit",
edit_supports={
"prompt", "image_urls", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed",
"sync_mode", "system_prompt", "resolution", "enable_web_search", "limit_generations",
"thinking_level",
},
"upscale": False,
"edit_endpoint": "fal-ai/nano-banana-2/edit",
"edit_supports": {
"prompt", "image_urls", "aspect_ratio", "num_images",
"output_format", "safety_tolerance", "seed", "sync_mode",
"system_prompt", "resolution", "enable_web_search",
"limit_generations", "thinking_level",
max_reference_images=14,
),
"fal-ai/gpt-image-1.5": _model(
"GPT Image 1.5", "~15s", "Prompt adherence", "$0.034/image",
style="gpt_literal", sizes={
"landscape": "1536x1024", "square": "1024x1024", "portrait": "1024x1536",
},
"max_reference_images": 14,
},
"fal-ai/gpt-image-1.5": {
"display": "GPT Image 1.5",
"speed": "~15s",
"strengths": "Prompt adherence",
"price": "$0.034/image",
"size_style": "gpt_literal",
"sizes": {
"landscape": "1536x1024",
"square": "1024x1024",
"portrait": "1024x1536",
# quality pinned to medium (also for gpt-image-2) so portal billing stays
# predictable: low is too rough, high is 3-6x the per-image cost.
defaults={"quality": "medium", "num_images": 1, "output_format": "png"},
supports={
"prompt", "image_size", "quality", "num_images", "output_format", "background", "sync_mode",
},
"defaults": {
# Quality is pinned to medium to keep portal billing predictable
# across all users (low is too rough, high is 4-6x more expensive).
"quality": "medium",
"num_images": 1,
"output_format": "png",
edit_endpoint="fal-ai/gpt-image-1.5/edit",
edit_supports={
"prompt", "image_urls", "image_size", "quality", "num_images", "output_format", "sync_mode",
},
"supports": {
"prompt", "image_size", "quality", "num_images", "output_format",
"background", "sync_mode",
max_reference_images=16,
),
# GPT Image 2 uses FAL's preset enum (unlike 1.5's literal dims) mapped to the
# 4:3 variants: the 16:9 presets (1024x576) fall below its 655,360 min-pixel
# requirement. openai_api_key (BYOK) is deliberately not in `supports` — all
# users go through the shared FAL billing path. Its edit endpoint lives under
# the OpenAI namespace (NOT fal-ai/) and auto-infers size, so no image_size.
"fal-ai/gpt-image-2": _model(
"GPT Image 2", "~20s", "SOTA text rendering + CJK, world-aware photorealism", "$0.04–0.06/image",
style="image_size_preset", sizes={
"landscape": "landscape_4_3", "square": "square_hd", "portrait": "portrait_4_3",
},
"upscale": False,
# Edit endpoint: high-fidelity edits preserving composition/lighting.
"edit_endpoint": "fal-ai/gpt-image-1.5/edit",
"edit_supports": {
"prompt", "image_urls", "image_size", "quality", "num_images",
"output_format", "sync_mode",
defaults={"quality": "medium", "num_images": 1, "output_format": "png"},
supports={
"prompt", "image_size", "quality", "num_images", "output_format", "sync_mode",
},
"max_reference_images": 16,
},
"fal-ai/gpt-image-2": {
"display": "GPT Image 2",
"speed": "~20s",
"strengths": "SOTA text rendering + CJK, world-aware photorealism",
"price": "$0.04–0.06/image",
# GPT Image 2 uses FAL's standard preset enum (unlike 1.5's literal
# dimensions). We map to the 4:3 variants — the 16:9 presets
# (1024x576) fall below GPT-Image-2's 655,360 min-pixel requirement
# and would be rejected. 4:3 keeps us above the minimum on all
# three aspect ratios.
"size_style": "image_size_preset",
"sizes": {
"landscape": "landscape_4_3", # 1024x768
"square": "square_hd", # 1024x1024
"portrait": "portrait_4_3", # 768x1024
edit_endpoint="openai/gpt-image-2/edit",
edit_supports={
"prompt", "image_urls", "quality", "num_images", "output_format", "sync_mode", "mask_image_url",
},
"defaults": {
# Same quality pinning as gpt-image-1.5: medium keeps Nous
# Portal billing predictable. "high" is 3-4x the per-image
# cost at the same size; "low" is too rough for production use.
"quality": "medium",
"num_images": 1,
"output_format": "png",
max_reference_images=16,
),
"fal-ai/ideogram/v3": _model(
"Ideogram V3", "~5s", "Best typography", "$0.03-0.09/image",
defaults={"rendering_speed": "BALANCED", "expand_prompt": True, "style": "AUTO"},
supports={
"prompt", "image_size", "rendering_speed", "expand_prompt", "style", "seed",
},
"supports": {
"prompt", "image_size", "quality", "num_images", "output_format",
"sync_mode",
# openai_api_key (BYOK) intentionally omitted — all users go
# through the shared FAL billing path.
edit_endpoint="fal-ai/ideogram/v3/edit",
edit_supports={
"prompt", "image_urls", "rendering_speed", "expand_prompt", "style", "seed",
},
"upscale": False,
# GPT Image 2 edit endpoint lives under the OpenAI namespace on FAL
# (NOT fal-ai/). Takes `image_urls` (list) + optional mask. We don't
# send `image_size` on edit so the model auto-infers from input.
"edit_endpoint": "openai/gpt-image-2/edit",
"edit_supports": {
"prompt", "image_urls", "quality", "num_images", "output_format",
"sync_mode", "mask_image_url",
max_reference_images=1,
),
"fal-ai/recraft/v4/pro/text-to-image": _model(
"Recraft V4 Pro", "~8s", "Design, brand systems, production-ready", "$0.25/image",
defaults={"enable_safety_checker": False}, # V4 Pro dropped V3's required `style` enum
supports={
"prompt", "image_size", "enable_safety_checker", "colors", "background_color",
},
"max_reference_images": 16,
},
"fal-ai/ideogram/v3": {
"display": "Ideogram V3",
"speed": "~5s",
"strengths": "Best typography",
"price": "$0.03-0.09/image",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"rendering_speed": "BALANCED",
"expand_prompt": True,
"style": "AUTO",
),
"fal-ai/qwen-image": _model(
"Qwen Image", "~12s", "LLM-based, complex text", "$0.02/MP",
defaults={
"num_inference_steps": 30, "guidance_scale": 2.5, "num_images": 1,
"output_format": "png", "acceleration": "regular",
},
"supports": {
"prompt", "image_size", "rendering_speed", "expand_prompt",
"style", "seed",
supports={
"prompt", "image_size", "num_inference_steps", "guidance_scale", "num_images", "output_format",
"acceleration", "seed", "sync_mode",
},
"upscale": False,
# Ideogram V3 edit endpoint takes `image_urls` (list).
"edit_endpoint": "fal-ai/ideogram/v3/edit",
"edit_supports": {
"prompt", "image_urls", "rendering_speed", "expand_prompt",
"style", "seed",
edit_endpoint="fal-ai/qwen-image-2/pro/edit",
edit_supports={
"prompt", "image_urls", "num_inference_steps", "guidance_scale", "num_images", "output_format",
"acceleration", "seed", "sync_mode",
},
"max_reference_images": 1,
},
"fal-ai/recraft/v4/pro/text-to-image": {
"display": "Recraft V4 Pro",
"speed": "~8s",
"strengths": "Design, brand systems, production-ready",
"price": "$0.25/image",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
# V4 Pro dropped V3's required `style` enum — defaults handle taste now.
"enable_safety_checker": False,
max_reference_images=3,
),
# Krea 2 on FAL — same family as ``plugins/image_gen/krea`` but billed through
# FAL / the FAL managed gateway. Native ``krea-2-*`` ids route to the plugin.
"fal-ai/krea/v2/medium/text-to-image": _model(
"Krea 2 Medium", "~15-25s", "Illustration, anime, painting, expressive/artistic styles", "$0.030 (text) / $0.035 (style refs)",
style="aspect_ratio",
defaults={"creativity": "medium"},
supports={
"prompt", "aspect_ratio", "creativity", "seed", "image_style_references",
},
"supports": {
"prompt", "image_size", "enable_safety_checker",
"colors", "background_color",
),
"fal-ai/krea/v2/large/text-to-image": _model(
"Krea 2 Large", "~25-60s", "Photorealism, raw textured looks (motion blur, grain, film)", "$0.060 (text) / $0.065 (style refs)",
style="aspect_ratio",
defaults={"creativity": "medium"},
supports={
"prompt", "aspect_ratio", "creativity", "seed", "image_style_references",
},
"upscale": False,
},
"fal-ai/qwen-image": {
"display": "Qwen Image",
"speed": "~12s",
"strengths": "LLM-based, complex text",
"price": "$0.02/MP",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"num_inference_steps": 30,
"guidance_scale": 2.5,
"num_images": 1,
"output_format": "png",
"acceleration": "regular",
},
"supports": {
"prompt", "image_size", "num_inference_steps", "guidance_scale",
"num_images", "output_format", "acceleration", "seed", "sync_mode",
},
"upscale": False,
# Qwen edit uses the Qwen Image 2.0 Pro editing endpoint, which takes
# `image_urls` (list) + natural-language edit instructions.
"edit_endpoint": "fal-ai/qwen-image-2/pro/edit",
"edit_supports": {
"prompt", "image_urls", "num_inference_steps", "guidance_scale",
"num_images", "output_format", "acceleration", "seed", "sync_mode",
},
"max_reference_images": 3,
},
# Krea 2 on FAL — same model family as ``plugins/image_gen/krea``, but billed
# through FAL / the FAL managed gateway. Native ``krea-2-*`` ids route to the
# dedicated Krea plugin instead.
"fal-ai/krea/v2/medium/text-to-image": {
"display": "Krea 2 Medium",
"speed": "~15-25s",
"strengths": "Illustration, anime, painting, expressive/artistic styles",
"price": "$0.030 (text) / $0.035 (style refs)",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"creativity": "medium",
},
"supports": {
"prompt", "aspect_ratio", "creativity", "seed",
"image_style_references",
},
"upscale": False,
},
"fal-ai/krea/v2/large/text-to-image": {
"display": "Krea 2 Large",
"speed": "~25-60s",
"strengths": "Photorealism, raw textured looks (motion blur, grain, film)",
"price": "$0.060 (text) / $0.065 (style refs)",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"creativity": "medium",
},
"supports": {
"prompt", "aspect_ratio", "creativity", "seed",
"image_style_references",
},
"upscale": False,
},
# ─── Aug 2026 catalog expansion ────────────────────────────────────────
# Endpoint ids, `supports` whitelists and enum defaults below are taken
# from each model's FAL OpenAPI schema, so a key we send is a key the
# vendor declares. Paired `/edit` apps hang off their text-to-image entry
# rather than appearing as separate picker rows.
"bytedance/seedream/v5/pro/text-to-image": {
"display": "Seedream 5.0 Pro",
"speed": "~10s",
"strengths": "ByteDance flagship, dense layouts, native text in 14 languages",
"price": "$0.0675/image (≤1536²)",
"size_style": "image_size_preset",
# Pro requires total pixels between 1024x1024 and 2048x2048 —
# explicit ImageSize dicts keep every aspect inside that window.
"sizes": {
),
# Entries below take endpoint ids, `supports` whitelists and enum defaults from
# each model's FAL OpenAPI schema; paired `/edit` apps hang off their
# text-to-image entry rather than appearing as separate picker rows.
# Seedream Pro requires total pixels between 1024² and 2048² — explicit
# ImageSize dicts keep every aspect inside that window.
"bytedance/seedream/v5/pro/text-to-image": _model(
"Seedream 5.0 Pro", "~10s", "ByteDance flagship, dense layouts, native text in 14 languages", "$0.0675/image (≤1536²)",
style="image_size_preset", sizes={
"landscape": {"width": 2048, "height": 1152},
"square": {"width": 1536, "height": 1536},
"portrait": {"width": 1152, "height": 2048},
},
"defaults": {
"num_images": 1,
"output_format": "png",
defaults={
"num_images": 1, "output_format": "png", "enable_safety_checker": False,
},
supports={
"prompt", "image_size", "num_images", "output_format", "sync_mode", "enable_safety_checker",
},
edit_endpoint="bytedance/seedream/v5/pro/edit",
edit_supports={
"prompt", "image_urls", "image_size", "num_images", "output_format", "sync_mode",
"enable_safety_checker",
},
max_reference_images=10,
),
# Lite wants 2560x1440..4096x4096 total pixels: use the documented presets (FAL
# auto-scales under the floor) rather than hand-rolled dicts that drift.
"bytedance/seedream/v5/lite/text-to-image": _model(
"Seedream 5.0 Lite", "~5s", "Fast/cheap Seedream tier, high-res output", "$0.035/image",
defaults={"num_images": 1, "enable_safety_checker": False},
supports={
"prompt", "image_size", "num_images", "max_images", "sync_mode", "enable_safety_checker",
},
),
"ideogram/v4/instant": _model(
"Ideogram V4 (Instant)", "<1s", "Latest Ideogram typography, posters/logos, instant", "$0.0075/MP",
defaults={
"expansion_model": "Medium", "output_format": "png",
"enable_safety_checker": False,
},
"supports": {
"prompt", "image_size", "num_images", "output_format",
"sync_mode", "enable_safety_checker",
supports={
"prompt", "image_size", "expansion_model", "num_images", "seed", "sync_mode",
"enable_safety_checker", "output_format",
},
"upscale": False,
# Region-precise editing with up to 10 reference images.
"edit_endpoint": "bytedance/seedream/v5/pro/edit",
"edit_supports": {
"prompt", "image_urls", "image_size", "num_images",
"output_format", "sync_mode", "enable_safety_checker",
),
"ideogram/v4/fast": _model(
"Ideogram V4 (Fast)", "~1s", "Ideogram V4 quality tiers via rendering_speed", "$0.005-0.018/MP",
defaults={"expansion_model": "Medium", "rendering_speed": "BALANCED"},
supports={
"prompt", "image_size", "expansion_model", "rendering_speed", "num_images", "seed", "sync_mode",
},
"max_reference_images": 10,
},
"bytedance/seedream/v5/lite/text-to-image": {
"display": "Seedream 5.0 Lite",
"speed": "~5s",
"strengths": "Fast/cheap Seedream tier, high-res output",
"price": "$0.035/image",
"size_style": "image_size_preset",
# Lite wants total pixels between 2560x1440 and 4096x4096. Use the
# documented presets (FAL auto-scales if a preset is under the floor)
# instead of hand-rolled ImageSize dicts that drift from the schema.
"sizes": _PRESET_SIZES,
"defaults": {
"num_images": 1,
),
"alibaba/qwen-image-3/text-to-image": _model(
"Qwen Image 3", "~8s", "Complex CN/EN text rendering, prompt-guided resolution", "$0.04 (1K) / $0.075 (2K) per image",
defaults={
"num_images": 1, "output_format": "png", "enable_prompt_expansion": False,
"enable_safety_checker": False,
},
"supports": {
"prompt", "image_size", "num_images", "max_images",
"sync_mode", "enable_safety_checker",
},
"upscale": False,
},
"ideogram/v4/instant": {
"display": "Ideogram V4 (Instant)",
"speed": "<1s",
"strengths": "Latest Ideogram typography, posters/logos, instant",
"price": "$0.0075/MP",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"expansion_model": "Medium",
"output_format": "png",
"enable_safety_checker": False,
},
"supports": {
"prompt", "image_size", "expansion_model", "num_images",
"seed", "sync_mode", "enable_safety_checker", "output_format",
},
"upscale": False,
},
"ideogram/v4/fast": {
"display": "Ideogram V4 (Fast)",
"speed": "~1s",
"strengths": "Ideogram V4 quality tiers via rendering_speed",
"price": "$0.005-0.018/MP",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"expansion_model": "Medium",
"rendering_speed": "BALANCED",
},
"supports": {
"prompt", "image_size", "expansion_model", "rendering_speed",
"num_images", "seed", "sync_mode",
},
"upscale": False,
},
"alibaba/qwen-image-3/text-to-image": {
"display": "Qwen Image 3",
"speed": "~8s",
"strengths": "Complex CN/EN text rendering, prompt-guided resolution",
"price": "$0.04 (1K) / $0.075 (2K) per image",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"num_images": 1,
"output_format": "png",
"enable_prompt_expansion": False, # avoid the LLM rewrite surprise
"enable_safety_checker": False,
},
"supports": {
"prompt", "negative_prompt", "image_size", "num_images",
"seed", "sync_mode", "output_format",
supports={
"prompt", "negative_prompt", "image_size", "num_images", "seed", "sync_mode", "output_format",
"enable_prompt_expansion", "enable_safety_checker",
},
"upscale": False,
# Qwen Image 3 edit: 1-3 reference images, identity-preserving edits.
"edit_endpoint": "alibaba/qwen-image-3/edit",
"edit_supports": {
"prompt", "image_urls", "negative_prompt", "num_images",
"seed", "sync_mode", "output_format",
edit_endpoint="alibaba/qwen-image-3/edit",
edit_supports={
"prompt", "image_urls", "negative_prompt", "num_images", "seed", "sync_mode", "output_format",
"enable_prompt_expansion", "enable_safety_checker",
},
"max_reference_images": 3,
},
"microsoft/mai-image-2.5-pro": {
"display": "MAI Image 2.5 Pro",
"speed": "~10s",
"strengths": "Microsoft flagship, hero imagery, precise typography",
"price": "~$0.17/image",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"num_images": 1,
"output_format": "png",
},
"supports": {
"prompt", "aspect_ratio", "num_images", "output_format",
"sync_mode",
},
"upscale": False,
},
"google/nano-banana-2-lite": {
"display": "Nano Banana 2 Lite",
"speed": "<2s",
"strengths": "Gemini image family, sub-2s, 14 aspect ratios incl. extreme",
"price": "~$0.04/image (1K fixed)",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"num_images": 1,
"output_format": "png",
"safety_tolerance": "5",
},
"supports": {
"prompt", "aspect_ratio", "num_images", "seed",
"output_format", "safety_tolerance", "sync_mode",
max_reference_images=3,
),
"microsoft/mai-image-2.5-pro": _model(
"MAI Image 2.5 Pro", "~10s", "Microsoft flagship, hero imagery, precise typography", "~$0.17/image",
style="aspect_ratio",
defaults={"num_images": 1, "output_format": "png"},
supports={"prompt", "aspect_ratio", "num_images", "output_format", "sync_mode"},
),
"google/nano-banana-2-lite": _model(
"Nano Banana 2 Lite", "<2s", "Gemini image family, sub-2s, 14 aspect ratios incl. extreme", "~$0.04/image (1K fixed)",
style="aspect_ratio",
defaults={"num_images": 1, "output_format": "png", "safety_tolerance": "5"},
supports={
"prompt", "aspect_ratio", "num_images", "seed", "output_format", "safety_tolerance", "sync_mode",
"system_prompt", "limit_generations", "thinking_level",
},
"upscale": False,
# Fast multi-turn local edits with reference images via `image_urls`.
"edit_endpoint": "google/nano-banana-2-lite/edit",
"edit_supports": {
"prompt", "image_urls", "aspect_ratio", "num_images",
"seed", "output_format", "safety_tolerance", "sync_mode",
"system_prompt",
edit_endpoint="google/nano-banana-2-lite/edit",
edit_supports={
"prompt", "image_urls", "aspect_ratio", "num_images", "seed", "output_format", "safety_tolerance",
"sync_mode", "system_prompt",
},
"max_reference_images": 4,
},
"fal-ai/recraft/v4.1/text-to-image": {
"display": "Recraft V4.1",
"speed": "~8s",
"strengths": "Design-first raster, brand systems, editorial",
"price": "$0.035/image",
"size_style": "image_size_preset",
"sizes": _PRESET_SIZES,
"defaults": {
"enable_safety_checker": False,
max_reference_images=4,
),
"fal-ai/recraft/v4.1/text-to-image": _model(
"Recraft V4.1", "~8s", "Design-first raster, brand systems, editorial", "$0.035/image",
defaults={"enable_safety_checker": False},
supports={
"prompt", "image_size", "enable_safety_checker", "colors", "background_color",
},
"supports": {
"prompt", "image_size", "enable_safety_checker",
"colors", "background_color",
),
"xai/grok-imagine-image/v2.0/text-to-image": _model(
"Grok Imagine Image 2.0", "~5s", "xAI. Design-grade typography/layout, instruction following", "$0.06/image (1K medium)",
style="aspect_ratio",
# 1k + medium is the cheapest sensible tier; 2k is roughly +33%/image. 1k native
# is sub-2MP — pass upscale=true per call when needed. Edits omit aspect_ratio
# (defaults to "auto", following the first input image).
defaults={
"num_images": 1, "output_format": "png", "resolution": "1k", "quality": "medium",
},
"upscale": False,
},
"xai/grok-imagine-image/v2.0/text-to-image": {
"display": "Grok Imagine Image 2.0",
"speed": "~5s",
"strengths": "xAI. Design-grade typography/layout, instruction following",
"price": "$0.06/image (1K medium)",
"size_style": "aspect_ratio",
"sizes": _ASPECT_SIZES,
"defaults": {
"num_images": 1,
"output_format": "png",
# 1k + medium is the cheapest sensible tier ($0.06/image);
# 2k roughly +33% per image.
"resolution": "1k",
"quality": "medium",
supports={
"prompt", "aspect_ratio", "num_images", "output_format", "resolution", "quality", "sync_mode",
},
"supports": {
"prompt", "aspect_ratio", "num_images", "output_format",
"resolution", "quality", "sync_mode",
edit_endpoint="xai/grok-imagine-image/v2.0/edit",
edit_supports={
"prompt", "image_urls", "num_images", "output_format", "resolution", "quality", "sync_mode",
},
"upscale": False,
# Edit endpoint takes `image_urls` (max 3) + the same knobs;
# aspect_ratio defaults to "auto" (follows the first input image),
# so we don't send it on edits.
"edit_endpoint": "xai/grok-imagine-image/v2.0/edit",
"edit_supports": {
"prompt", "image_urls", "num_images", "output_format",
"resolution", "quality", "sync_mode",
},
"max_reference_images": 3,
},
max_reference_images=3,
),
}
File diff suppressed because it is too large Load Diff
+63 -155
View File
@@ -2,33 +2,24 @@
All source handling (data:/http(s)/file/local/container) funnels through
:func:`resolve_image_source` so size and magic-byte checks are enforced exactly
once. Returns raw bytes (not a path): the downstream step is base64 -> data URL
(RFC 2397) and provider base64 content blocks.
once. Returns raw bytes (not a path): the downstream step is base64 -> data URL.
Images are the default and the historical purpose. Callers whose argument
takes video opt in via ``permitted=("video",)`` — the same confinement and
credential-guard pipeline applies, and only the type check at the end differs
(extension-table typing plus an mp4 magic sniff, rather than image magic
bytes). Every existing call site keeps the image-only default unchanged.
Images are the default. Callers whose argument takes video opt in via
``permitted=("video",)``: same confinement and credential-guard pipeline, only
the final type check differs (extension table + mp4 magic sniff).
Security (terminal-backend confinement, GHSA-gpxw-6wxv-w3qq): under a non-local
terminal backend the file tools are confined to the sandbox (SECURITY.md 2.2),
but vision read images host-side. This resolver enforces the same boundary:
backend the file tools are confined to the sandbox, so vision must be too:
* local backend -> read any host path (chosen posture, unchanged)
* local backend -> read any host path (chosen posture)
* non-local backend:
path in a media cache -> host-read (the gateway/download caches live on
the host and are bind-mounted into the sandbox)
path anywhere else -> read the bytes *inside the sandbox* via exec-read
(the agent can already ``cat`` any container file;
this stays within the sandbox boundary and never
reaches the host's ``/etc/passwd`` / ``~/.ssh``).
path in a media cache -> host-read (gateway/download caches live on the
host and are bind-mounted into the sandbox)
path anywhere else -> read *inside the sandbox* via exec-read (the
agent can already ``cat`` any container file)
So a prompt-injected ``vision_analyze('/etc/passwd')`` under Docker reads the
*container's* file (what every other tool sees), not the host's — no escape —
while container-only images (tmpfs ``/workspace``, root-owned) are still
deliverable. This is the unified delivery + confinement model: the same
mechanism that fixes "vision can't see container files" also closes the escape.
container's file, never the host's, while container-only images stay deliverable.
"""
from __future__ import annotations
@@ -40,12 +31,9 @@ from dataclasses import dataclass
from pathlib import Path
from typing import Optional
# Raw-bytes INGEST budget — what the resolver will load before handing off.
# This is deliberately the 50MB download cap (tools/vision_tools._VISION_MAX_DOWNLOAD_BYTES),
# NOT the 20MB provider payload cap. The 20MB cap (_MAX_BASE64_BYTES) is a
# *post-resize* limit enforced at the call sites: an oversized raw image must
# still reach the resizer so it can be downscaled under the payload cap. Capping
# raw bytes at 20MB here would reject every 20-50MB photo before resize can run.
# Raw-bytes INGEST budget: deliberately the 50MB download cap, NOT the 20MB
# provider payload cap — that one is enforced post-resize at the call sites, and
# a 20-50MB photo must still reach the resizer.
_MAX_INGEST_BYTES = 50 * 1024 * 1024
@@ -87,8 +75,7 @@ class ResolvedImage:
origin: str # one of: data | http | file | local | container
# Explicit URL scheme, e.g. "ftp://", "s3://". Bare Windows drive paths
# ("C:\x.png") don't match because they lack the "//".
# Explicit URL scheme ("ftp://", "s3://"). Bare Windows drive paths lack the "//".
_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*://")
@@ -118,23 +105,15 @@ async def resolve_image_source(
)
# Everything else is a filesystem path — including bare relative names
# like "pic.png" (accepted on main; a path-shape gate here regressed them).
# like "pic.png" (a path-shape gate here regressed them once).
candidate = s[len("file://"):] if s.lower().startswith("file://") else s
p = Path(os.path.expanduser(candidate))
# Confinement decision (see module docstring). Under a non-local backend
# a path is host-readable ONLY if it lands in a media cache (after
# translating a container-visible cache path back to its host mount);
# every other path is read inside the sandbox via exec-read, so a host
# path outside the caches never yields the host's bytes.
host_target = _permitted_host_read_target(p, ctx)
if host_target is not None and host_target.is_file():
# Shared credential-read guard (agent.file_safety, #57698): refuse
# secret-bearing files (.env, auth.json, ...) with an intentional,
# specific error instead of relying on the magic-byte sniff to
# reject them incidentally. Same chokepoint the image-gen/video-gen
# provider plugins enforce on model-supplied local paths. Import is
# best-effort (guard unavailability must not break image loading);
# a real block always propagates.
# Shared credential-read guard: refuse secret-bearing files (.env,
# auth.json) with a specific error rather than relying on the magic
# sniff to reject them incidentally. Guard import is best-effort; a
# real block always propagates.
try:
from agent.file_safety import raise_if_read_blocked
except Exception: # noqa: BLE001 — guard unavailable: proceed
@@ -147,12 +126,8 @@ async def resolve_image_source(
data = await asyncio.to_thread(host_target.read_bytes)
return _finalize(data, "", "file", s, permitted)
if _is_local_terminal_backend():
# Local backend: any path was host-readable, so a miss simply means
# the file doesn't exist — no sandbox to fall back to.
# Any path was host-readable, so a miss means the file doesn't exist.
raise SourceNotFound(f"media file not found: '{p}'", src=s, origin="file")
# Not a permitted host read (or the host file is absent) -> read the
# bytes inside the sandbox. Under a sandbox this reads the container's
# filesystem, never the host's.
return await _resolve_container_fallback(p, ctx, s, permitted)
@@ -172,15 +147,9 @@ def _resolve_data_url(s: str) -> tuple[bytes, str]:
def _http_block_reason(url: str) -> Optional[str]:
"""Return a human-readable block reason, or None when the URL is allowed.
Pre-flight short-circuit: policy-blocked URLs are refused BEFORE any
network I/O. ``_download_image`` re-checks policy internally (per attempt
and against the final redirect target) — that second evaluation is
intentional, not redundant: this one guarantees no bytes move for a
blocked URL; the inner one covers redirects and non-resolver callers.
Preserves the specific website-policy message so the agent sees *why*.
"""
"""Block reason, or None when allowed. Refuses policy-blocked URLs BEFORE any
network I/O; ``_download_image`` re-checks per attempt and against the final
redirect target — the second evaluation is intentional, not redundant."""
from tools.url_safety import is_safe_url
from tools.website_policy import check_website_access
@@ -210,27 +179,19 @@ async def _download_to_bytes(url: str) -> bytes:
def _is_local_terminal_backend() -> bool:
"""True when the terminal backend runs directly on the host.
Mirrors ``tools.browser_tool._is_local_backend`` and terminal_tool's own
dispatch, which key off ``TERMINAL_ENV``.
"""
"""True when the terminal backend runs directly on the host (keys off ``TERMINAL_ENV``)."""
return os.getenv("TERMINAL_ENV", "local").strip().lower() in ("local", "")
def _media_cache_roots() -> list:
"""Agent-managed media cache directories under HERMES_HOME (host side).
The only host paths vision may read under a non-local backend: gateway-
downloaded inbound media and the tools' own URL-download temp dirs. Covers
the consolidated ``cache/`` layout and the legacy flat directories.
"""
"""Host-side media caches: the only host paths vision may read under a
non-local backend (gateway inbound media + the tools' own download temp dirs)."""
from hermes_constants import get_hermes_home
home = get_hermes_home()
return [
home / "cache", # cache/images, cache/vision, cache/video(s), cache/audio
home / "images", # desktop/clipboard/PDF uploads (tui_gateway) — #69575
home / "images", # desktop/clipboard/PDF uploads (tui_gateway)
home / "image_cache",
home / "audio_cache",
home / "video_cache",
@@ -240,14 +201,12 @@ def _media_cache_roots() -> list:
def _permitted_host_read_target(p: Path, ctx: ResolveContext) -> Optional[Path]:
"""Return the host path to read, or ``None`` if a host read is not permitted.
"""Host path to read, or ``None`` if a host read is not permitted.
- Local backend: any path is permitted (chosen posture). Returns ``p``.
- Non-local backend: permitted only if the path resolves inside a media
cache root. A container-visible cache path (e.g. ``/root/.hermes/cache/
images/x.png``) is first translated back to its host mount; anything that
is not under a cache returns ``None`` so the caller routes it to the
in-sandbox exec-read instead of reading the host filesystem.
Local backend: any path. Non-local: only paths resolving inside a media
cache root (a container-visible cache path is first translated back to its
host mount); anything else returns ``None`` so the caller exec-reads inside
the sandbox instead of touching the host filesystem.
"""
if _is_local_terminal_backend():
try:
@@ -283,13 +242,12 @@ def _get_active_env(task_id: Optional[str]):
def _ensure_container_env(task_id: Optional[str]) -> None:
"""Lazily bring up the sandbox (SSH/Docker/…) before an in-sandbox read.
"""Lazily bring up the sandbox before an in-sandbox read.
Unlike the terminal tool, vision never triggered environment creation, so a
session whose first action is ``vision_analyze`` on a container-only path
under a non-local backend found no active env and failed — until a terminal
command happened to create one (issue #62825). Best-effort: any failure just
leaves the env absent and the caller hits the existing fail-closed error.
Vision never triggered environment creation, so a session whose first action
was ``vision_analyze`` on a container-only path found no active env until a
terminal command created one. Best-effort: failure leaves the env absent and
the caller hits the fail-closed error.
"""
if not task_id:
return
@@ -304,35 +262,17 @@ def _ensure_container_env(task_id: Optional[str]) -> None:
async def _resolve_container_fallback(
p: Path, ctx: ResolveContext, src: str, permitted: tuple = ("image",)
) -> ResolvedImage:
"""Read the image bytes inside the sandbox (fail-closed when none exists).
"""Read the bytes inside the sandbox; fail-closed when no env exists (a
non-cache host path under a sandbox must never leak via a host fallback).
Reached when a host read is not permitted or the host file is absent. The
agent can already ``cat`` any container file (file_operations.py reads
root-owned mode-600 files this way), so this stays within the same sandbox
boundary and never touches the host filesystem. ``--`` stops a leading-dash
path from being parsed as a ``base64`` option; ``base64 -w0`` is GNU-only,
so pipe through ``tr -d`` for BusyBox.
Fail-closed: if there is no active sandbox env we refuse rather than falling
back to a host read, so a non-cache host path under a sandbox never leaks.
Cold-start retry: under Docker the very first exec against a freshly
started container can fail (empty pipe / partial setup) while an identical
second call succeeds. We retry once with a short delay before giving up,
so callers don't see "could not read inside the sandbox" on a file that is
verifiably readable on the immediate retry. See #76566.
Diagnostic: when every attempt fails, the container's own output (stderr
+ stdout) is folded into the raised error so the user can distinguish
"no such file" from "permission denied" from "container never came up"
instead of staring at one opaque message.
Cold-start retry: under Docker the first exec against a fresh container can
fail (empty pipe) while an identical second call succeeds, so retry once
after a short delay. On final failure the container's own output is folded
into the error so "no such file" / "permission denied" / "never came up"
are distinguishable.
"""
import asyncio
import shlex
# Bring the sandbox up on demand: without this, the first vision_analyze of
# a session (before any terminal command) has no active env to read from
# under a non-local backend (issue #62825).
_ensure_container_env(ctx.task_id)
env = _get_active_env(ctx.task_id)
@@ -342,14 +282,11 @@ async def _resolve_container_fallback(
f"session is available to read it",
src=src, origin="container")
# Bound the read INSIDE the sandbox: head -c caps at ingest-limit+1 bytes
# so a huge file (or /dev/zero) can't stream unbounded base64 into host
# memory — the +1 byte lets us distinguish "exactly at the cap" from
# "over the cap" after decode. The input redirect (< path) avoids argv
# entirely, so leading-dash paths can't be parsed as options; base64
# -w0 is GNU-only, so pipe through tr -d for BusyBox.
# env.execute is a blocking backend exec; keep it off the event loop so a
# multi-MB base64 read doesn't stall every other coroutine.
# Bound the read INSIDE the sandbox: head -c caps at ingest-limit+1 so
# /dev/zero can't stream unbounded base64 into host memory (the +1
# distinguishes "at the cap" from "over"). The input redirect avoids argv, so
# leading-dash paths can't parse as options; base64 -w0 is GNU-only, hence
# tr -d for BusyBox. env.execute blocks — keep it off the event loop.
qp = shlex.quote(str(p))
cmd = f"head -c {_MAX_INGEST_BYTES + 1} < {qp} | base64 | tr -d '\\n'"
@@ -359,14 +296,9 @@ async def _resolve_container_fallback(
if last_res.get("returncode", 1) == 0:
break
if attempt == 0:
# Cold-start: give the container a moment to settle its pipes
# before retrying. 150ms covers Docker exec warm-up in practice
# without making a real failure feel sluggish.
await asyncio.sleep(0.15)
await asyncio.sleep(0.15) # covers Docker exec warm-up in practice
if last_res.get("returncode", 1) != 0:
diag = (last_res.get("output") or "").strip().splitlines()
# Keep the diagnostic small and noise-free: first non-empty line,
# trimmed to a sane length so it slots into the agent's error UI.
first = next((ln.strip() for ln in diag if ln.strip()), "")
suffix = f" ({first[:200]})" if first else ""
raise SourceNotFound(
@@ -384,18 +316,11 @@ async def _resolve_container_fallback(
def _finalize(
data: bytes, declared_mime: str, origin: str, src: str, permitted: tuple = ("image",)
) -> ResolvedImage:
"""Intrinsic-correctness chokepoint: ingest byte cap + type check.
"""Chokepoint: 50MB ingest cap + type check.
The cap here is the generous 50MB *ingest* budget, not the 20MB provider
payload cap — a 20-50MB image must survive this step so the call site can
resize it under the payload cap. See ``_MAX_INGEST_BYTES``.
Images are typed by magic bytes. Video (opt-in via ``permitted``) is typed
by the extension table plus an mp4 container sniff: extension typing is
sufficient because every downstream consumer re-validates — the upload
gateway signs the content type into its presigned URL and the vendor
rejects undecodable input — so a wrong guess is a clean rejection there
rather than a hole here.
Images are typed by magic bytes. Video (opt-in) is typed by extension plus
an mp4 container sniff — sufficient because every downstream consumer
re-validates, so a wrong guess is a clean rejection there, not a hole here.
"""
from tools.vision_tools import _detect_image_mime_type_from_bytes
@@ -409,9 +334,7 @@ def _finalize(
return ResolvedImage(data=data, mime=sniffed, origin=origin)
if "image" in permitted and b"<svg" in data[:4096].lower():
# Pass SVG through — the vision call sites rasterize it to PNG
# via _normalize_to_supported_image before embedding (providers
# only ingest raster images).
# Pass SVG through — call sites rasterize it to PNG before embedding.
return ResolvedImage(data=data, mime="image/svg+xml", origin=origin)
if "video" in permitted:
@@ -424,11 +347,8 @@ def _finalize(
def _detect_video_mime(data: bytes, src: str) -> Optional[str]:
"""Video MIME from the extension table, else the mp4/mov container magic.
The magic fallback covers extensionless sources (data: URLs, URLs with
query strings): ISO base-media files carry ``ftyp`` at offset 4.
"""
"""Video MIME from the extension table, else the ISO base-media ``ftyp``
magic at offset 4 (covers extensionless data: URLs / query-string URLs)."""
from urllib.parse import urlsplit
from tools.vision_tools import _detect_video_mime_type
@@ -447,23 +367,11 @@ async def resolve_local_source_to_data_url(
) -> str:
"""Convert a path-like media source into a ``data:`` URL via the resolver.
Generation tools (image_generate / video_generate) forward model-supplied
source images to provider plugins, which historically read local paths off
the HOST filesystem regardless of terminal backend. Under a non-local
backend that is both broken (the file usually lives in the sandbox, so the
host read misses) and inconsistent with the confinement model vision/video
analysis enforce (GHSA-gpxw-6wxv-w3qq): the sandbox boundary should govern
every model-supplied path.
This helper is the dispatch-layer chokepoint: URL-shaped sources
(http/https/data) pass through untouched; anything path-like resolves
through :func:`resolve_image_source` — media-cache host reads, bounded
in-sandbox exec-read, lazy env bring-up, credential guard, ingest cap —
and comes back as a ``data:`` URL every provider already accepts.
Callers apply this only under a non-local terminal backend: on the local
backend providers keep their existing host-side reads (chosen posture,
zero behavior change).
Dispatch-layer chokepoint for generation tools: providers historically read
model-supplied local paths off the HOST regardless of backend — broken under
a sandbox (the file lives there) and inconsistent with the confinement model
vision enforces. URL-shaped sources (http/https/data) pass through untouched.
Callers apply this only under a non-local backend (local keeps host reads).
"""
s = (src or "").strip()
if not s or s.lower().startswith(("http://", "https://", "data:")):
+89 -153
View File
@@ -1,28 +1,11 @@
#!/usr/bin/env python3
"""
Video Generation Tool
=====================
"""``video_generate``: one tool dispatching to a plugin-registered :class:`VideoGenProvider`
(``agent/video_gen_provider.py`` ABC, ``agent/video_gen_registry.py``, ``plugins/video_gen/<name>/``).
Single ``video_generate`` tool that dispatches to a plugin-registered
video generation provider. Mirrors the ``image_generate`` design:
- ``agent/video_gen_provider.py`` defines the :class:`VideoGenProvider` ABC.
- ``agent/video_gen_registry.py`` holds the active providers (populated by
plugins at import time).
- Each provider lives under ``plugins/video_gen/<name>/``.
The tool is backend-agnostic and ships **no in-tree provider** — enable a
plugin (``hermes plugins enable video_gen/<name>``) and select it in
``hermes tools`` → Video Generation.
One tool covers text-to-video, image-to-video and reference-to-video with a
compact schema (prompt, image_url, reference_image_urls, duration,
aspect_ratio, resolution, negative_prompt, audio, seed, model). Providers
ignore parameters they do not support: the tool layer does only lightweight
validation (type/required-prompt) and each provider clamps inside
:meth:`VideoGenProvider.generate`, so the surface stays stable as providers
with different capabilities ship. Video edit/extend are intentionally not
exposed here; providers with those workflows expose separate tools.
Ships **no in-tree provider**: enable a plugin and select it in ``hermes tools`` → Video
Generation. Covers text-, image- and reference-to-video; the tool layer only does lightweight
validation and each provider clamps/ignores unsupported params inside ``generate``, so the
surface stays stable as providers ship. Video edit/extend are deliberately not exposed here.
"""
from __future__ import annotations
@@ -45,13 +28,9 @@ logger = logging.getLogger(__name__)
VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
"name": "video_generate",
# Placeholder — description AND params are rebuilt dynamically at
# get_tool_definitions() time from the active provider's declared
# capabilities() and the active model's catalog entry. Optional args
# (image_url, reference_image_urls, negative_prompt, audio, seed,
# upscale) are advertised ONLY when the active backend/model honors
# them; the handler accepts them regardless (replay compat — providers
# clamp/ignore). See _build_dynamic_video_schema().
# Placeholder: description AND params are rebuilt at get_tool_definitions() time by
# _build_dynamic_video_schema() from capabilities() + the model's catalog entry. Optional
# args are advertised ONLY when honored; the handler accepts them regardless (replay compat).
"description": "(rebuilt at get_definitions() time — see _build_dynamic_video_schema)",
"parameters": {
"type": "object",
@@ -89,9 +68,7 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
"``video_gen.model``. Unknown models are rejected."
),
},
# image_url / reference_image_urls / negative_prompt / audio / seed /
# upscale are added per-capability by _build_dynamic_video_schema.
# Do not re-add them statically.
# Capability-gated args are added by _build_dynamic_video_schema; never statically.
},
"required": ["prompt"],
},
@@ -101,8 +78,6 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
# ---------------------------------------------------------------------------
# Config readers (mirror image_generation_tool.py)
# ---------------------------------------------------------------------------
def _read_video_gen_key(key: str) -> Optional[str]:
"""Return the stripped ``video_gen.<key>`` string from config.yaml, or None."""
try:
@@ -127,23 +102,23 @@ def _read_configured_video_model() -> Optional[str]:
return _read_video_gen_key("model")
def _discovered_registry():
"""Import the provider registry after (idempotent) plugin discovery so user-installed plugins are visible."""
from agent import video_gen_registry
from hermes_cli.plugins import _ensure_plugins_discovered
_ensure_plugins_discovered()
return video_gen_registry, _ensure_plugins_discovered
# ---------------------------------------------------------------------------
# Availability check + provider resolution
# ---------------------------------------------------------------------------
def check_video_generation_requirements() -> bool:
"""True when at least one registered provider reports available.
Triggers plugin discovery (idempotent) so user-installed plugins are
visible to the toolset gate.
"""
"""True when at least one registered provider reports available."""
try:
from agent.video_gen_registry import list_providers
from hermes_cli.plugins import _ensure_plugins_discovered
_ensure_plugins_discovered()
for provider in list_providers():
registry_mod, _ = _discovered_registry()
for provider in registry_mod.list_providers():
try:
if provider.is_available():
return True
@@ -155,20 +130,14 @@ def check_video_generation_requirements() -> bool:
def _resolve_active_provider():
"""Return the active provider object or None.
Forces a discovery refresh on a miss — handles long-lived sessions that
started before a plugin was installed.
"""
"""Active provider or None; forces a discovery refresh on a miss (long-lived sessions
that started before a plugin was installed)."""
try:
from agent.video_gen_registry import get_active_provider
from hermes_cli.plugins import _ensure_plugins_discovered
_ensure_plugins_discovered()
provider = get_active_provider()
registry_mod, ensure_discovered = _discovered_registry()
provider = registry_mod.get_active_provider()
if provider is None:
_ensure_plugins_discovered(force=True)
provider = get_active_provider()
ensure_discovered(force=True)
provider = registry_mod.get_active_provider()
return provider
except Exception as exc:
logger.debug("video_gen provider resolution failed: %s", exc)
@@ -177,30 +146,28 @@ def _resolve_active_provider():
def _missing_provider_error(configured: Optional[str]) -> str:
if configured:
msg = (
f"video_gen.provider='{configured}' is set but no plugin "
f"registered that name. Run `hermes plugins list` to see "
f"installed video gen backends, or `hermes tools` → Video "
f"Generation to pick one."
)
return json.dumps(error_response(
error=msg, error_type="provider_not_registered",
error=(
f"video_gen.provider='{configured}' is set but no plugin "
f"registered that name. Run `hermes plugins list` to see "
f"installed video gen backends, or `hermes tools` → Video "
f"Generation to pick one."
),
error_type="provider_not_registered",
provider=configured,
))
msg = (
"No video generation backend is configured. Run `hermes tools` → "
"Video Generation to enable one (xAI, FAL, or Google Veo)."
)
return json.dumps(error_response(
error=msg, error_type="no_provider_configured",
error=(
"No video generation backend is configured. Run `hermes tools` → "
"Video Generation to enable one (xAI, FAL, or Google Veo)."
),
error_type="no_provider_configured",
))
# ---------------------------------------------------------------------------
# Handler
# ---------------------------------------------------------------------------
def _coerce_int(value: Any) -> Optional[int]:
if value is None or value == "":
return None
@@ -211,16 +178,11 @@ def _coerce_int(value: Any) -> Optional[int]:
def _coerce_bool(value: Any) -> Optional[bool]:
if value is None:
return None
if isinstance(value, bool):
return value
if isinstance(value, str):
v = value.strip().lower()
if v in {"true", "1", "yes", "on"}:
return True
if v in {"false", "0", "no", "off"}:
return False
return {"true": True, "1": True, "yes": True, "on": True,
"false": False, "0": False, "no": False, "off": False}.get(value.strip().lower())
return None
@@ -241,8 +203,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
reference_image_urls = _normalize_reference_images(args.get("reference_image_urls"))
task_id = _kw.get("task_id")
# Confinement chokepoint (mirrors image_generate): under a non-local
# backend, path-like source images reach providers as data: URLs.
# Confinement chokepoint (mirrors image_generate): non-local backends hand providers data: URLs.
from tools.image_generation_tool import _confine_source_images
image_url, reference_image_urls, confine_error = _confine_source_images(
@@ -258,8 +219,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
upscale = _coerce_bool(args.get("upscale"))
model_override = (args.get("model") or "").strip() or None
# Soft validation — providers do their own. The backend may accept
# image-only on its image-to-video endpoint, but our surface always needs a prompt.
# Soft validation — providers do their own; a backend may accept image-only, our surface never does.
if not prompt:
return tool_error("prompt is required for video generation")
if "operation" in args or "video_url" in args:
@@ -291,7 +251,6 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
}
# Drop None entries so providers see clean defaults.
kwargs = {k: v for k, v in kwargs.items() if v is not None}
pname = getattr(provider, "name", "?")
def _err(error: str, error_type: str) -> str:
@@ -303,8 +262,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
try:
result = provider.generate(prompt=prompt, **kwargs)
except TypeError as exc:
# A provider that hasn't widened its signature is a plugin bug, not a
# caller error — surface a clear contract message.
# An un-widened provider signature is a plugin bug, not a caller error.
logger.warning(
"video_gen provider '%s' rejected kwargs (signature too narrow): %s",
pname, exc,
@@ -328,12 +286,33 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
# ---------------------------------------------------------------------------
# Dynamic schema — reflect the active backend's actual capabilities
# ---------------------------------------------------------------------------
# The configured backend determines which modalities, aspect ratios,
# resolutions, durations and audio/negative-prompt flags are real; surfacing
# the per-model surface in the description means the model usually gets the
# call right first try. model_tools.get_tool_definitions() keys its cache on
# config.yaml mtime, so the schema rebuilds when provider/model changes.
# Surfacing the per-model surface (modalities, enums, durations, audio/negative-prompt)
# means the model usually gets the call right first try. model_tools.get_tool_definitions()
# keys its cache on config.yaml mtime, so the schema rebuilds on provider/model change.
# Optional params advertised only when the provider's capabilities() sets the flag
# (order = schema property order).
_CAPABILITY_PARAMS = (
("supports_negative_prompt", "negative_prompt", {
"type": "string",
"description": "Content to avoid in the output.",
}),
("supports_audio", "audio", {
"type": "boolean",
"description": "Enable native audio generation (affects pricing tier).",
}),
("supports_seed", "seed", {
"type": "integer",
"description": "Seed for reproducible outputs.",
}),
("supports_upscale", "upscale", {
"type": "boolean",
"description": (
"High-resolution pass via the backend's video upscaler "
"(~2x, extra cost/latency). Omit for native resolution."
),
}),
)
_GENERIC_DESCRIPTION = (
"Generate a video from a text prompt (text-to-video), animate a "
@@ -352,14 +331,16 @@ _GENERIC_DESCRIPTION = (
)
def _build_dynamic_video_schema() -> Dict[str, Any]:
"""Render description AND params from the active backend's declared surface.
def _schema(description: str, properties: Dict[str, Any]) -> Dict[str, Any]:
return {
"description": description,
"parameters": {"type": "object", "properties": properties, "required": ["prompt"]},
}
Optional args are advertised only when the resolved provider/model honors
them (capabilities() + the model's catalog entry); enums and duration
bounds tighten to the active model's sets. The handler still accepts
unadvertised args (replay compat): providers clamp or ignore.
"""
def _build_dynamic_video_schema() -> Dict[str, Any]:
"""Render description AND params from capabilities() + the model's catalog entry; enums and
duration bounds tighten to the active model. Unadvertised args are still accepted (replay compat)."""
static_props = VIDEO_GENERATE_SCHEMA["parameters"]["properties"]
parts: List[str] = [_GENERIC_DESCRIPTION]
@@ -371,14 +352,7 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
"\nNo video backend is available. Calls will return an error "
"until the user picks one via `hermes tools` → Video Generation."
)
return {
"description": "\n".join(parts),
"parameters": {
"type": "object",
"properties": {"prompt": static_props["prompt"]},
"required": ["prompt"],
},
}
return _schema("\n".join(parts), {"prompt": static_props["prompt"]})
try:
caps = provider.capabilities() or {}
@@ -388,17 +362,11 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
models = provider.list_models() or []
except Exception:
models = []
active_model = configured_model or provider.default_model()
model_meta = next(
(m for m in models if isinstance(m, dict) and m.get("id") == active_model),
{},
)
model_meta = next((m for m in models if isinstance(m, dict) and m.get("id") == active_model), {})
# ---- description -------------------------------------------------
# Model caveats surface only what differs from the backend's overall
# capabilities. FAL's plugin uses the singular ``modality`` key for
# single-modality entries.
# Model caveats surface only what differs from the backend's overall capabilities.
# FAL's plugin uses the singular ``modality`` key for single-modality entries.
model_modalities = set(model_meta.get("modalities") or [])
modality = model_meta.get("modality")
if modality:
@@ -435,7 +403,6 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
if notice:
parts.append(f"- storage: {notice}")
# ---- params ------------------------------------------------------
properties: Dict[str, Any] = {"prompt": static_props["prompt"]}
if can_i2v:
@@ -477,48 +444,17 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
param["enum"] = list(caps[caps_key])
properties[key] = param
if caps.get("supports_negative_prompt"):
properties["negative_prompt"] = {
"type": "string",
"description": "Content to avoid in the output.",
}
if caps.get("supports_audio"):
properties["audio"] = {
"type": "boolean",
"description": (
"Enable native audio generation (affects pricing tier)."
),
}
elif caps.get("audio_always_on"):
for flag, key, param in _CAPABILITY_PARAMS:
if caps.get(flag):
properties[key] = param
if caps.get("audio_always_on") and not caps.get("supports_audio"):
parts.append(
"- audio: native stereo audio is generated with every video "
"(always on; no toggle) — describe the desired sound in the "
"prompt"
)
if caps.get("supports_seed"):
properties["seed"] = {
"type": "integer",
"description": "Seed for reproducible outputs.",
}
if caps.get("supports_upscale"):
properties["upscale"] = {
"type": "boolean",
"description": (
"High-resolution pass via the backend's video upscaler "
"(~2x, extra cost/latency). Omit for native resolution."
),
}
properties["model"] = static_props["model"]
return {
"description": "\n".join(parts),
"parameters": {
"type": "object",
"properties": properties,
"required": ["prompt"],
},
}
return _schema("\n".join(parts), properties)
registry.register(
+602 -1523
View File
File diff suppressed because it is too large Load Diff
+313
View File
@@ -0,0 +1,313 @@
"""Image format detection, normalization and region cropping for vision tools.
Everything here runs BEFORE an image is base64-embedded. A vision tool result
is baked into immutable conversation history and re-sent every turn, so an
unsupported media type or corrupt bytes would wedge the session with a
non-retryable 400 on every resume — normalization must happen up front.
"""
from __future__ import annotations
import logging
import uuid
from io import BytesIO
from pathlib import Path
from typing import Any, Optional
from hermes_constants import get_hermes_dir
logger = logging.getLogger("tools.vision_tools")
_EXTENSION_MIME_TYPES = {
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".png": "image/png",
".gif": "image/gif",
".bmp": "image/bmp",
".webp": "image/webp",
".svg": "image/svg+xml",
}
# Media types the major vision providers (Anthropic in particular) accept
# inline. SVG/BMP/TIFF are rejected with a non-retryable 400.
_ANTHROPIC_SUPPORTED_MEDIA_TYPES = frozenset(
{"image/jpeg", "image/png", "image/gif", "image/webp"}
)
def _determine_mime_type(image_path: Path) -> str:
"""MIME type from file extension (defaults to image/jpeg)."""
return _EXTENSION_MIME_TYPES.get(image_path.suffix.lower(), "image/jpeg")
def _detect_image_mime_type_from_bytes(data: bytes) -> Optional[str]:
"""Magic-byte MIME sniff (authoritative; no extension trust).
Returns ``None`` for anything without a recognized header — including SVG,
which has no magic bytes (the resolver sniffs ``<svg`` and passes it
through for rasterization).
"""
header = data[:64]
if header.startswith(b"\x89PNG\r\n\x1a\n"):
# Magic bytes alone are insufficient: reject corrupt PNGs before they
# can be embedded. Pillow is optional — without it fall back to
# header-only sniffing; only an actual failed verify() rejects.
try:
from PIL import Image
except ImportError:
return "image/png"
try:
with Image.open(BytesIO(data)) as image:
image.verify()
except Exception:
return None
return "image/png"
if header.startswith(b"\xff\xd8\xff"):
return "image/jpeg"
if header.startswith((b"GIF87a", b"GIF89a")):
return "image/gif"
if header.startswith(b"BM"):
return "image/bmp"
if len(header) >= 12 and header[:4] == b"RIFF" and header[8:12] == b"WEBP":
return "image/webp"
return None
def _supported_media_types() -> frozenset:
"""Formats the ACTIVE main model's server can decode.
The managed llama-server decodes with stb_image — no WebP — and an
undecodable image part fails SILENTLY (the model confabulates), so the set
is narrowed there and normalization converts those formats to PNG.
"""
try:
from agent.auxiliary_client import _runtime_main_value
from hermes_cli.local_runtime.capabilities import (
ACCEPTED_IMAGE_MIMES,
is_managed_provider,
)
if is_managed_provider(
str(_runtime_main_value("provider") or ""),
str(_runtime_main_value("base_url") or "")):
return ACCEPTED_IMAGE_MIMES
except Exception: # noqa: BLE001 — best-effort narrowing only
pass
return _ANTHROPIC_SUPPORTED_MEDIA_TYPES
def _rasterize_svg_to_png(svg_path: Path, out_path: Path) -> bool:
"""Best-effort SVG → PNG via cairosvg, svglib+reportlab, rsvg-convert, inkscape (all soft deps)."""
def _ok() -> bool:
return out_path.exists() and out_path.stat().st_size > 0
try:
import cairosvg # type: ignore
cairosvg.svg2png(url=str(svg_path), write_to=str(out_path))
return _ok()
except Exception:
pass
try:
from svglib.svglib import svg2rlg # type: ignore
from reportlab.graphics import renderPM # type: ignore
drawing = svg2rlg(str(svg_path))
if drawing is not None:
renderPM.drawToFile(drawing, str(out_path), fmt="PNG")
return _ok()
except Exception:
pass
import shutil
import subprocess
for cmd in (
["rsvg-convert", "-o", str(out_path), str(svg_path)],
["inkscape", str(svg_path), "--export-type=png",
f"--export-filename={out_path}"],
):
if shutil.which(cmd[0]):
try:
subprocess.run(
cmd, check=True, capture_output=True, timeout=30,
stdin=subprocess.DEVNULL,
)
if _ok():
return True
except Exception:
continue
return False
def _normalize_to_supported_image(
image_path: Path, detected_mime: str
) -> tuple[Optional[Path], Optional[str], Optional[str]]:
"""Ensure an image is in a provider-supported format.
Returns ``(path, mime, error)``: the input unchanged when already
supported; ``(new_png_path, "image/png", None)`` after conversion — a temp
file the CALLER must clean up; ``(None, None, message)`` when impossible.
SVG is rasterized; other Pillow-readable rasters (BMP, TIFF) re-encode to PNG.
"""
if detected_mime in _supported_media_types():
return image_path, detected_mime, None
out_dir = get_hermes_dir("cache/vision", "temp_vision_images")
out_dir.mkdir(parents=True, exist_ok=True)
out_path = out_dir / f"converted_{uuid.uuid4()}.png"
if detected_mime == "image/svg+xml":
if _rasterize_svg_to_png(image_path, out_path):
return out_path, "image/png", None
return (
None,
None,
"This is an SVG, which vision models cannot read directly, and no "
"SVG rasterizer is installed (tried cairosvg, svglib, rsvg-convert, "
"inkscape). Convert the SVG to PNG first — e.g. open it in a browser "
"and screenshot it, or install a rasterizer "
"(`pip install cairosvg`) — then re-run vision_analyze on the PNG.",
)
try:
from PIL import Image as _PILImage
with _PILImage.open(image_path) as _img:
if _img.mode not in ("RGB", "RGBA", "L"):
_img = _img.convert("RGBA")
_img.save(out_path, format="PNG")
if out_path.exists() and out_path.stat().st_size > 0:
return out_path, "image/png", None
except Exception as _exc:
logger.warning("Failed to normalize %s image to PNG: %s",
detected_mime, _exc)
return (
None,
None,
f"Image format {detected_mime!r} is not supported by the vision API "
f"and could not be converted to PNG (install Pillow for raster "
f"conversion). Convert it to PNG or JPEG and try again.",
)
# Full raster validation runs on untrusted images in a shared CPU executor:
# bound animated-image work by frame count AND total decoded area so a compact
# file cannot monopolize a worker with unbounded frames.
_VISION_MAX_VALIDATED_FRAME_COUNT = 100
_VISION_MAX_VALIDATED_AGGREGATE_PIXELS = 100_000_000
def _validate_raster_image_decodable(
image_path: Path,
max_frames: int = _VISION_MAX_VALIDATED_FRAME_COUNT,
max_pixels: int = _VISION_MAX_VALIDATED_AGGREGATE_PIXELS,
) -> Optional[str]:
"""Return an error unless Pillow can fully decode every frame.
Header sniffing and ``Image.open`` only inspect containers: a timed-out
download can look like a valid PNG with a truncated pixel stream. Without
Pillow the image passes unvalidated rather than rejecting everything.
"""
try:
from PIL import Image as _PILImage
from PIL import ImageSequence as _PILImageSequence
except ImportError:
return None
try:
with _PILImage.open(image_path) as image:
image.verify()
with _PILImage.open(image_path) as image:
validated_pixels = 0
for frame_number, frame in enumerate(
_PILImageSequence.Iterator(image), start=1
):
if frame_number > max_frames:
return (
"Image validation rejected animation: "
f"frame {frame_number} exceeds the maximum "
f"{max_frames} validated frames."
)
next_validated_pixels = validated_pixels + frame.width * frame.height
if next_validated_pixels > max_pixels:
return (
"Image validation rejected animation: aggregate decoded "
f"pixel count would reach {next_validated_pixels} at frame "
f"{frame_number}, exceeding the maximum "
f"{max_pixels}."
)
frame.load()
validated_pixels = next_validated_pixels
except Exception as exc:
return f"Image could not be fully decoded: {exc}"
return None
def _image_exceeds_dimension(image_path: Path, max_dimension: int) -> bool:
"""True if the longest side exceeds ``max_dimension`` px.
Anthropic enforces an 8000px per-side cap independently of the byte cap.
Returns False (no forced resize) without Pillow or on unreadable files —
a missing soft dependency must never break the embed path.
"""
try:
from PIL import Image as _PILImage
with _PILImage.open(image_path) as _img:
return max(_img.size) > max_dimension
except Exception:
return False
def _crop_image_region(
image_path: Path,
region: Any,
offset_out: Optional[dict] = None,
) -> tuple[Optional[Path], Optional[str], Optional[str]]:
"""Crop to ``region`` = [x1, y1, x2, y2] (original-image pixels).
Applied BEFORE downscaling so the crop gets the full resolution budget.
Coordinates clamp to the image bounds; a zero-area/inverted region is
rejected with an error naming the real dimensions. Returns
``(cropped_temp_path, mime, None)`` — caller owns cleanup — or
``(None, None, error)``. Ported from QwenLM/qwen-code zoom-image.ts (Apache-2.0).
"""
try:
from PIL import Image
except ImportError:
return None, None, (
"region cropping requires Pillow (`pip install Pillow`); "
"retry without the region parameter."
)
if (
not isinstance(region, (list, tuple))
or len(region) != 4
or not all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in region)
):
return None, None, (
"Invalid region: expected [x1, y1, x2, y2] as four numbers "
"(pixel coordinates in the original image)."
)
try:
with Image.open(image_path) as img:
width, height = img.size
x1, y1, x2, y2 = (int(v) for v in region)
cx1 = max(0, min(x1, width))
cy1 = max(0, min(y1, height))
cx2 = max(0, min(x2, width))
cy2 = max(0, min(y2, height))
if cx2 <= cx1 or cy2 <= cy1:
return None, None, (
f"Invalid region [{x1}, {y1}, {x2}, {y2}]: crops to zero "
f"area after clamping to the image bounds. The image is "
f"{width}x{height} px — pick x1<x2 and y1<y2 inside "
f"[0, 0, {width}, {height}]."
)
cropped = img.crop((cx1, cy1, cx2, cy2))
if offset_out is not None:
offset_out.update(x=cx1, y=cy1, width=cx2 - cx1, height=cy2 - cy1)
out_path = image_path.with_name(
f"{image_path.stem}_region_{uuid.uuid4().hex[:8]}.png"
)
if cropped.mode not in ("RGB", "RGBA", "L", "LA", "P"):
cropped = cropped.convert("RGB")
cropped.save(out_path, format="PNG")
return out_path, "image/png", None
except Exception as exc:
return None, None, f"Failed to crop region: {exc}"