refactor(tools/media): extract vision_tools_image_prep; dedupe image/video generation providers; compact fal/xai helpers
This commit is contained in:
@@ -1,28 +1,13 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Leave a mark on the page in the Hermes desktop GUI's in-app browser.
|
||||
"""Persistent element annotations in the Hermes desktop GUI's in-app browser.
|
||||
|
||||
``drive_preview`` already draws every move it makes — the field it can reach,
|
||||
a box round its target, the cursor going there — but those are transients:
|
||||
each one stands for a single action and retires itself. That is right for
|
||||
narrating a click and no use at all for holding a finding on screen.
|
||||
|
||||
This is the deliberate one. An annotation outlines an element — or, with
|
||||
``hold``, the entire visible field at once — and stays until the agent takes it
|
||||
down, so it can show the user what it found, flag the fields
|
||||
it is about to fill, or keep its place while it works elsewhere on the page.
|
||||
Named for TouchDesigner's Annotate — the labelled box you drop around part of a
|
||||
network to call it out.
|
||||
|
||||
Annotations are bound to elements, not coordinates: they ride scrolls and
|
||||
reflows, and they go when their element does, so a navigation clears them
|
||||
without the agent having to.
|
||||
|
||||
Rides the same ``preview.act`` bridge as ``drive_preview`` rather than opening
|
||||
a second channel — the renderer already resolves ``@e`` refs and owns the
|
||||
overlay, so this is one more verb on a wire that exists.
|
||||
|
||||
Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for
|
||||
desktop-sourced sessions.
|
||||
``drive_preview`` draws transient marks (one per action, self-retiring). An
|
||||
annotation outlines an element — or, with ``hold``, the whole visible field —
|
||||
and stays until the agent removes it. Annotations bind to elements, not
|
||||
coordinates: they ride scrolls/reflows and vanish with their element, so a
|
||||
navigation clears them. Rides the same ``preview.act`` bridge as
|
||||
``drive_preview`` (the renderer resolves ``@e`` refs and owns the overlay).
|
||||
Lives in the ``desktop_ui`` toolset, enabled only for desktop-sourced sessions.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
+272
-475
@@ -11,7 +11,7 @@ faces when default-on, so upscaling is strictly per-call opt-in.
|
||||
Pricing strings are as-of-commit and allowed to drift.
|
||||
"""
|
||||
|
||||
from typing import Any, Dict
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
_PRESET_SIZES = {
|
||||
"landscape": "landscape_16_9",
|
||||
@@ -19,533 +19,330 @@ _PRESET_SIZES = {
|
||||
"portrait": "portrait_16_9",
|
||||
}
|
||||
_ASPECT_SIZES = {"landscape": "16:9", "square": "1:1", "portrait": "9:16"}
|
||||
_DEFAULT_SIZES = {"image_size_preset": _PRESET_SIZES, "aspect_ratio": _ASPECT_SIZES}
|
||||
|
||||
|
||||
def _model(
|
||||
display: str,
|
||||
speed: str,
|
||||
strengths: str,
|
||||
price: str,
|
||||
*,
|
||||
style: str = "image_size_preset",
|
||||
sizes: Optional[Dict[str, Any]] = None,
|
||||
defaults: Dict[str, Any],
|
||||
supports: set,
|
||||
edit_endpoint: Optional[str] = None,
|
||||
edit_supports: Optional[set] = None,
|
||||
max_reference_images: Optional[int] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Build one catalog entry; edit keys are present only for edit-capable models."""
|
||||
entry: Dict[str, Any] = {
|
||||
"display": display,
|
||||
"speed": speed,
|
||||
"strengths": strengths,
|
||||
"price": price,
|
||||
"size_style": style,
|
||||
"sizes": sizes if sizes is not None else _DEFAULT_SIZES[style],
|
||||
"defaults": defaults,
|
||||
"supports": supports,
|
||||
"upscale": False,
|
||||
}
|
||||
if edit_endpoint:
|
||||
entry["edit_endpoint"] = edit_endpoint
|
||||
entry["edit_supports"] = edit_supports
|
||||
entry["max_reference_images"] = max_reference_images
|
||||
return entry
|
||||
|
||||
|
||||
FAL_MODELS: Dict[str, Dict[str, Any]] = {
|
||||
"fal-ai/flux-2/klein/9b": {
|
||||
"display": "FLUX 2 Klein 9B",
|
||||
"speed": "<1s",
|
||||
"strengths": "Fast, crisp text",
|
||||
"price": "$0.006/MP",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"num_inference_steps": 4,
|
||||
"output_format": "png",
|
||||
"enable_safety_checker": False,
|
||||
"fal-ai/flux-2/klein/9b": _model(
|
||||
"FLUX 2 Klein 9B", "<1s", "Fast, crisp text", "$0.006/MP",
|
||||
defaults={
|
||||
"num_inference_steps": 4, "output_format": "png", "enable_safety_checker": False,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "num_inference_steps", "seed",
|
||||
"output_format", "enable_safety_checker",
|
||||
supports={
|
||||
"prompt", "image_size", "num_inference_steps", "seed", "output_format", "enable_safety_checker",
|
||||
},
|
||||
"upscale": False,
|
||||
# Image-to-image / editing: FLUX.2 [klein] 9B edit endpoint takes
|
||||
# `image_urls` (list). Natural-language edits, multi-ref.
|
||||
"edit_endpoint": "fal-ai/flux-2/klein/9b/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "num_inference_steps", "seed",
|
||||
"output_format", "enable_safety_checker",
|
||||
edit_endpoint="fal-ai/flux-2/klein/9b/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "num_inference_steps", "seed", "output_format", "enable_safety_checker",
|
||||
},
|
||||
"max_reference_images": 9,
|
||||
},
|
||||
"fal-ai/flux-2-pro": {
|
||||
"display": "FLUX 2 Pro",
|
||||
"speed": "~6s",
|
||||
"strengths": "Studio photorealism",
|
||||
"price": "$0.03/MP",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"num_inference_steps": 50,
|
||||
"guidance_scale": 4.5,
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"enable_safety_checker": False,
|
||||
"safety_tolerance": "5",
|
||||
max_reference_images=9,
|
||||
),
|
||||
"fal-ai/flux-2-pro": _model(
|
||||
"FLUX 2 Pro", "~6s", "Studio photorealism", "$0.03/MP",
|
||||
defaults={
|
||||
"num_inference_steps": 50, "guidance_scale": 4.5, "num_images": 1,
|
||||
"output_format": "png", "enable_safety_checker": False, "safety_tolerance": "5",
|
||||
"sync_mode": True,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "num_inference_steps", "guidance_scale",
|
||||
"num_images", "output_format", "enable_safety_checker",
|
||||
"safety_tolerance", "sync_mode", "seed",
|
||||
supports={
|
||||
"prompt", "image_size", "num_inference_steps", "guidance_scale", "num_images", "output_format",
|
||||
"enable_safety_checker", "safety_tolerance", "sync_mode", "seed",
|
||||
},
|
||||
"upscale": False,
|
||||
# Edit endpoint accepts up to 9 reference images.
|
||||
"edit_endpoint": "fal-ai/flux-2-pro/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "num_inference_steps", "guidance_scale",
|
||||
"num_images", "output_format", "enable_safety_checker",
|
||||
"safety_tolerance", "sync_mode", "seed",
|
||||
edit_endpoint="fal-ai/flux-2-pro/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "num_inference_steps", "guidance_scale", "num_images", "output_format",
|
||||
"enable_safety_checker", "safety_tolerance", "sync_mode", "seed",
|
||||
},
|
||||
"max_reference_images": 9,
|
||||
},
|
||||
"fal-ai/z-image/turbo": {
|
||||
"display": "Z-Image Turbo",
|
||||
"speed": "~2s",
|
||||
"strengths": "Bilingual EN/CN, 6B",
|
||||
"price": "$0.005/MP",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"num_inference_steps": 8,
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"enable_safety_checker": False,
|
||||
"enable_prompt_expansion": False, # avoid the extra per-request charge
|
||||
max_reference_images=9,
|
||||
),
|
||||
"fal-ai/z-image/turbo": _model(
|
||||
"Z-Image Turbo", "~2s", "Bilingual EN/CN, 6B", "$0.005/MP",
|
||||
defaults={ # prompt expansion off: avoids the extra per-request charge
|
||||
"num_inference_steps": 8, "num_images": 1, "output_format": "png",
|
||||
"enable_safety_checker": False, "enable_prompt_expansion": False,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "num_inference_steps", "num_images",
|
||||
"seed", "output_format", "enable_safety_checker",
|
||||
"enable_prompt_expansion",
|
||||
supports={
|
||||
"prompt", "image_size", "num_inference_steps", "num_images", "seed", "output_format",
|
||||
"enable_safety_checker", "enable_prompt_expansion",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"fal-ai/nano-banana-pro": {
|
||||
"display": "Nano Banana Pro (Gemini 3 Pro Image)",
|
||||
"speed": "~8s",
|
||||
"strengths": "Gemini 3 Pro, reasoning depth, text rendering",
|
||||
"price": "$0.15/image (1K)",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"safety_tolerance": "5",
|
||||
# "1K" is the cheapest tier; 4K doubles the per-image cost.
|
||||
# Users on Nous Subscription should stay at 1K for predictable billing.
|
||||
),
|
||||
"fal-ai/nano-banana-pro": _model(
|
||||
"Nano Banana Pro (Gemini 3 Pro Image)", "~8s", "Gemini 3 Pro, reasoning depth, text rendering", "$0.15/image (1K)",
|
||||
style="aspect_ratio",
|
||||
# "1K" is the cheapest tier; 4K doubles the per-image cost (Nous Subscription billing).
|
||||
defaults={
|
||||
"num_images": 1, "output_format": "png", "safety_tolerance": "5",
|
||||
"resolution": "1K",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "num_images", "output_format",
|
||||
"safety_tolerance", "seed", "sync_mode", "resolution",
|
||||
"enable_web_search", "limit_generations",
|
||||
},
|
||||
"upscale": False,
|
||||
# Nano Banana Pro edit (Gemini 3 Pro Image): natural-language edits
|
||||
# with up to 2 reference images via `image_urls`.
|
||||
"edit_endpoint": "fal-ai/nano-banana-pro/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "aspect_ratio", "num_images",
|
||||
"output_format", "safety_tolerance", "seed", "sync_mode",
|
||||
supports={
|
||||
"prompt", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", "sync_mode",
|
||||
"resolution", "enable_web_search", "limit_generations",
|
||||
},
|
||||
"max_reference_images": 2,
|
||||
},
|
||||
"fal-ai/nano-banana-2": {
|
||||
"display": "Nano Banana 2 (Gemini 3.1 Flash Image)",
|
||||
"speed": "~3s",
|
||||
"strengths": "Fast reasoning, multilingual text, infographics",
|
||||
"price": "Lower-cost Flash tier",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"safety_tolerance": "4",
|
||||
"resolution": "1K",
|
||||
"limit_generations": True,
|
||||
edit_endpoint="fal-ai/nano-banana-pro/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed",
|
||||
"sync_mode", "resolution", "enable_web_search", "limit_generations",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "num_images", "output_format",
|
||||
"safety_tolerance", "seed", "sync_mode", "system_prompt",
|
||||
"resolution", "enable_web_search", "limit_generations",
|
||||
max_reference_images=2,
|
||||
),
|
||||
"fal-ai/nano-banana-2": _model(
|
||||
"Nano Banana 2 (Gemini 3.1 Flash Image)", "~3s", "Fast reasoning, multilingual text, infographics", "Lower-cost Flash tier",
|
||||
style="aspect_ratio",
|
||||
defaults={
|
||||
"num_images": 1, "output_format": "png", "safety_tolerance": "4",
|
||||
"resolution": "1K", "limit_generations": True,
|
||||
},
|
||||
supports={
|
||||
"prompt", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed", "sync_mode",
|
||||
"system_prompt", "resolution", "enable_web_search", "limit_generations", "thinking_level",
|
||||
},
|
||||
edit_endpoint="fal-ai/nano-banana-2/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "aspect_ratio", "num_images", "output_format", "safety_tolerance", "seed",
|
||||
"sync_mode", "system_prompt", "resolution", "enable_web_search", "limit_generations",
|
||||
"thinking_level",
|
||||
},
|
||||
"upscale": False,
|
||||
"edit_endpoint": "fal-ai/nano-banana-2/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "aspect_ratio", "num_images",
|
||||
"output_format", "safety_tolerance", "seed", "sync_mode",
|
||||
"system_prompt", "resolution", "enable_web_search",
|
||||
"limit_generations", "thinking_level",
|
||||
max_reference_images=14,
|
||||
),
|
||||
"fal-ai/gpt-image-1.5": _model(
|
||||
"GPT Image 1.5", "~15s", "Prompt adherence", "$0.034/image",
|
||||
style="gpt_literal", sizes={
|
||||
"landscape": "1536x1024", "square": "1024x1024", "portrait": "1024x1536",
|
||||
},
|
||||
"max_reference_images": 14,
|
||||
},
|
||||
"fal-ai/gpt-image-1.5": {
|
||||
"display": "GPT Image 1.5",
|
||||
"speed": "~15s",
|
||||
"strengths": "Prompt adherence",
|
||||
"price": "$0.034/image",
|
||||
"size_style": "gpt_literal",
|
||||
"sizes": {
|
||||
"landscape": "1536x1024",
|
||||
"square": "1024x1024",
|
||||
"portrait": "1024x1536",
|
||||
# quality pinned to medium (also for gpt-image-2) so portal billing stays
|
||||
# predictable: low is too rough, high is 3-6x the per-image cost.
|
||||
defaults={"quality": "medium", "num_images": 1, "output_format": "png"},
|
||||
supports={
|
||||
"prompt", "image_size", "quality", "num_images", "output_format", "background", "sync_mode",
|
||||
},
|
||||
"defaults": {
|
||||
# Quality is pinned to medium to keep portal billing predictable
|
||||
# across all users (low is too rough, high is 4-6x more expensive).
|
||||
"quality": "medium",
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
edit_endpoint="fal-ai/gpt-image-1.5/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "image_size", "quality", "num_images", "output_format", "sync_mode",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "quality", "num_images", "output_format",
|
||||
"background", "sync_mode",
|
||||
max_reference_images=16,
|
||||
),
|
||||
# GPT Image 2 uses FAL's preset enum (unlike 1.5's literal dims) mapped to the
|
||||
# 4:3 variants: the 16:9 presets (1024x576) fall below its 655,360 min-pixel
|
||||
# requirement. openai_api_key (BYOK) is deliberately not in `supports` — all
|
||||
# users go through the shared FAL billing path. Its edit endpoint lives under
|
||||
# the OpenAI namespace (NOT fal-ai/) and auto-infers size, so no image_size.
|
||||
"fal-ai/gpt-image-2": _model(
|
||||
"GPT Image 2", "~20s", "SOTA text rendering + CJK, world-aware photorealism", "$0.04–0.06/image",
|
||||
style="image_size_preset", sizes={
|
||||
"landscape": "landscape_4_3", "square": "square_hd", "portrait": "portrait_4_3",
|
||||
},
|
||||
"upscale": False,
|
||||
# Edit endpoint: high-fidelity edits preserving composition/lighting.
|
||||
"edit_endpoint": "fal-ai/gpt-image-1.5/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "image_size", "quality", "num_images",
|
||||
"output_format", "sync_mode",
|
||||
defaults={"quality": "medium", "num_images": 1, "output_format": "png"},
|
||||
supports={
|
||||
"prompt", "image_size", "quality", "num_images", "output_format", "sync_mode",
|
||||
},
|
||||
"max_reference_images": 16,
|
||||
},
|
||||
"fal-ai/gpt-image-2": {
|
||||
"display": "GPT Image 2",
|
||||
"speed": "~20s",
|
||||
"strengths": "SOTA text rendering + CJK, world-aware photorealism",
|
||||
"price": "$0.04–0.06/image",
|
||||
# GPT Image 2 uses FAL's standard preset enum (unlike 1.5's literal
|
||||
# dimensions). We map to the 4:3 variants — the 16:9 presets
|
||||
# (1024x576) fall below GPT-Image-2's 655,360 min-pixel requirement
|
||||
# and would be rejected. 4:3 keeps us above the minimum on all
|
||||
# three aspect ratios.
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": {
|
||||
"landscape": "landscape_4_3", # 1024x768
|
||||
"square": "square_hd", # 1024x1024
|
||||
"portrait": "portrait_4_3", # 768x1024
|
||||
edit_endpoint="openai/gpt-image-2/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "quality", "num_images", "output_format", "sync_mode", "mask_image_url",
|
||||
},
|
||||
"defaults": {
|
||||
# Same quality pinning as gpt-image-1.5: medium keeps Nous
|
||||
# Portal billing predictable. "high" is 3-4x the per-image
|
||||
# cost at the same size; "low" is too rough for production use.
|
||||
"quality": "medium",
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
max_reference_images=16,
|
||||
),
|
||||
"fal-ai/ideogram/v3": _model(
|
||||
"Ideogram V3", "~5s", "Best typography", "$0.03-0.09/image",
|
||||
defaults={"rendering_speed": "BALANCED", "expand_prompt": True, "style": "AUTO"},
|
||||
supports={
|
||||
"prompt", "image_size", "rendering_speed", "expand_prompt", "style", "seed",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "quality", "num_images", "output_format",
|
||||
"sync_mode",
|
||||
# openai_api_key (BYOK) intentionally omitted — all users go
|
||||
# through the shared FAL billing path.
|
||||
edit_endpoint="fal-ai/ideogram/v3/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "rendering_speed", "expand_prompt", "style", "seed",
|
||||
},
|
||||
"upscale": False,
|
||||
# GPT Image 2 edit endpoint lives under the OpenAI namespace on FAL
|
||||
# (NOT fal-ai/). Takes `image_urls` (list) + optional mask. We don't
|
||||
# send `image_size` on edit so the model auto-infers from input.
|
||||
"edit_endpoint": "openai/gpt-image-2/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "quality", "num_images", "output_format",
|
||||
"sync_mode", "mask_image_url",
|
||||
max_reference_images=1,
|
||||
),
|
||||
"fal-ai/recraft/v4/pro/text-to-image": _model(
|
||||
"Recraft V4 Pro", "~8s", "Design, brand systems, production-ready", "$0.25/image",
|
||||
defaults={"enable_safety_checker": False}, # V4 Pro dropped V3's required `style` enum
|
||||
supports={
|
||||
"prompt", "image_size", "enable_safety_checker", "colors", "background_color",
|
||||
},
|
||||
"max_reference_images": 16,
|
||||
},
|
||||
"fal-ai/ideogram/v3": {
|
||||
"display": "Ideogram V3",
|
||||
"speed": "~5s",
|
||||
"strengths": "Best typography",
|
||||
"price": "$0.03-0.09/image",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"rendering_speed": "BALANCED",
|
||||
"expand_prompt": True,
|
||||
"style": "AUTO",
|
||||
),
|
||||
"fal-ai/qwen-image": _model(
|
||||
"Qwen Image", "~12s", "LLM-based, complex text", "$0.02/MP",
|
||||
defaults={
|
||||
"num_inference_steps": 30, "guidance_scale": 2.5, "num_images": 1,
|
||||
"output_format": "png", "acceleration": "regular",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "rendering_speed", "expand_prompt",
|
||||
"style", "seed",
|
||||
supports={
|
||||
"prompt", "image_size", "num_inference_steps", "guidance_scale", "num_images", "output_format",
|
||||
"acceleration", "seed", "sync_mode",
|
||||
},
|
||||
"upscale": False,
|
||||
# Ideogram V3 edit endpoint takes `image_urls` (list).
|
||||
"edit_endpoint": "fal-ai/ideogram/v3/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "rendering_speed", "expand_prompt",
|
||||
"style", "seed",
|
||||
edit_endpoint="fal-ai/qwen-image-2/pro/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "num_inference_steps", "guidance_scale", "num_images", "output_format",
|
||||
"acceleration", "seed", "sync_mode",
|
||||
},
|
||||
"max_reference_images": 1,
|
||||
},
|
||||
"fal-ai/recraft/v4/pro/text-to-image": {
|
||||
"display": "Recraft V4 Pro",
|
||||
"speed": "~8s",
|
||||
"strengths": "Design, brand systems, production-ready",
|
||||
"price": "$0.25/image",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
# V4 Pro dropped V3's required `style` enum — defaults handle taste now.
|
||||
"enable_safety_checker": False,
|
||||
max_reference_images=3,
|
||||
),
|
||||
# Krea 2 on FAL — same family as ``plugins/image_gen/krea`` but billed through
|
||||
# FAL / the FAL managed gateway. Native ``krea-2-*`` ids route to the plugin.
|
||||
"fal-ai/krea/v2/medium/text-to-image": _model(
|
||||
"Krea 2 Medium", "~15-25s", "Illustration, anime, painting, expressive/artistic styles", "$0.030 (text) / $0.035 (style refs)",
|
||||
style="aspect_ratio",
|
||||
defaults={"creativity": "medium"},
|
||||
supports={
|
||||
"prompt", "aspect_ratio", "creativity", "seed", "image_style_references",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "enable_safety_checker",
|
||||
"colors", "background_color",
|
||||
),
|
||||
"fal-ai/krea/v2/large/text-to-image": _model(
|
||||
"Krea 2 Large", "~25-60s", "Photorealism, raw textured looks (motion blur, grain, film)", "$0.060 (text) / $0.065 (style refs)",
|
||||
style="aspect_ratio",
|
||||
defaults={"creativity": "medium"},
|
||||
supports={
|
||||
"prompt", "aspect_ratio", "creativity", "seed", "image_style_references",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"fal-ai/qwen-image": {
|
||||
"display": "Qwen Image",
|
||||
"speed": "~12s",
|
||||
"strengths": "LLM-based, complex text",
|
||||
"price": "$0.02/MP",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"num_inference_steps": 30,
|
||||
"guidance_scale": 2.5,
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"acceleration": "regular",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "num_inference_steps", "guidance_scale",
|
||||
"num_images", "output_format", "acceleration", "seed", "sync_mode",
|
||||
},
|
||||
"upscale": False,
|
||||
# Qwen edit uses the Qwen Image 2.0 Pro editing endpoint, which takes
|
||||
# `image_urls` (list) + natural-language edit instructions.
|
||||
"edit_endpoint": "fal-ai/qwen-image-2/pro/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "num_inference_steps", "guidance_scale",
|
||||
"num_images", "output_format", "acceleration", "seed", "sync_mode",
|
||||
},
|
||||
"max_reference_images": 3,
|
||||
},
|
||||
# Krea 2 on FAL — same model family as ``plugins/image_gen/krea``, but billed
|
||||
# through FAL / the FAL managed gateway. Native ``krea-2-*`` ids route to the
|
||||
# dedicated Krea plugin instead.
|
||||
"fal-ai/krea/v2/medium/text-to-image": {
|
||||
"display": "Krea 2 Medium",
|
||||
"speed": "~15-25s",
|
||||
"strengths": "Illustration, anime, painting, expressive/artistic styles",
|
||||
"price": "$0.030 (text) / $0.035 (style refs)",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"creativity": "medium",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "creativity", "seed",
|
||||
"image_style_references",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"fal-ai/krea/v2/large/text-to-image": {
|
||||
"display": "Krea 2 Large",
|
||||
"speed": "~25-60s",
|
||||
"strengths": "Photorealism, raw textured looks (motion blur, grain, film)",
|
||||
"price": "$0.060 (text) / $0.065 (style refs)",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"creativity": "medium",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "creativity", "seed",
|
||||
"image_style_references",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
# ─── Aug 2026 catalog expansion ────────────────────────────────────────
|
||||
# Endpoint ids, `supports` whitelists and enum defaults below are taken
|
||||
# from each model's FAL OpenAPI schema, so a key we send is a key the
|
||||
# vendor declares. Paired `/edit` apps hang off their text-to-image entry
|
||||
# rather than appearing as separate picker rows.
|
||||
"bytedance/seedream/v5/pro/text-to-image": {
|
||||
"display": "Seedream 5.0 Pro",
|
||||
"speed": "~10s",
|
||||
"strengths": "ByteDance flagship, dense layouts, native text in 14 languages",
|
||||
"price": "$0.0675/image (≤1536²)",
|
||||
"size_style": "image_size_preset",
|
||||
# Pro requires total pixels between 1024x1024 and 2048x2048 —
|
||||
# explicit ImageSize dicts keep every aspect inside that window.
|
||||
"sizes": {
|
||||
),
|
||||
# Entries below take endpoint ids, `supports` whitelists and enum defaults from
|
||||
# each model's FAL OpenAPI schema; paired `/edit` apps hang off their
|
||||
# text-to-image entry rather than appearing as separate picker rows.
|
||||
# Seedream Pro requires total pixels between 1024² and 2048² — explicit
|
||||
# ImageSize dicts keep every aspect inside that window.
|
||||
"bytedance/seedream/v5/pro/text-to-image": _model(
|
||||
"Seedream 5.0 Pro", "~10s", "ByteDance flagship, dense layouts, native text in 14 languages", "$0.0675/image (≤1536²)",
|
||||
style="image_size_preset", sizes={
|
||||
"landscape": {"width": 2048, "height": 1152},
|
||||
"square": {"width": 1536, "height": 1536},
|
||||
"portrait": {"width": 1152, "height": 2048},
|
||||
},
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
defaults={
|
||||
"num_images": 1, "output_format": "png", "enable_safety_checker": False,
|
||||
},
|
||||
supports={
|
||||
"prompt", "image_size", "num_images", "output_format", "sync_mode", "enable_safety_checker",
|
||||
},
|
||||
edit_endpoint="bytedance/seedream/v5/pro/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "image_size", "num_images", "output_format", "sync_mode",
|
||||
"enable_safety_checker",
|
||||
},
|
||||
max_reference_images=10,
|
||||
),
|
||||
# Lite wants 2560x1440..4096x4096 total pixels: use the documented presets (FAL
|
||||
# auto-scales under the floor) rather than hand-rolled dicts that drift.
|
||||
"bytedance/seedream/v5/lite/text-to-image": _model(
|
||||
"Seedream 5.0 Lite", "~5s", "Fast/cheap Seedream tier, high-res output", "$0.035/image",
|
||||
defaults={"num_images": 1, "enable_safety_checker": False},
|
||||
supports={
|
||||
"prompt", "image_size", "num_images", "max_images", "sync_mode", "enable_safety_checker",
|
||||
},
|
||||
),
|
||||
"ideogram/v4/instant": _model(
|
||||
"Ideogram V4 (Instant)", "<1s", "Latest Ideogram typography, posters/logos, instant", "$0.0075/MP",
|
||||
defaults={
|
||||
"expansion_model": "Medium", "output_format": "png",
|
||||
"enable_safety_checker": False,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "num_images", "output_format",
|
||||
"sync_mode", "enable_safety_checker",
|
||||
supports={
|
||||
"prompt", "image_size", "expansion_model", "num_images", "seed", "sync_mode",
|
||||
"enable_safety_checker", "output_format",
|
||||
},
|
||||
"upscale": False,
|
||||
# Region-precise editing with up to 10 reference images.
|
||||
"edit_endpoint": "bytedance/seedream/v5/pro/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "image_size", "num_images",
|
||||
"output_format", "sync_mode", "enable_safety_checker",
|
||||
),
|
||||
"ideogram/v4/fast": _model(
|
||||
"Ideogram V4 (Fast)", "~1s", "Ideogram V4 quality tiers via rendering_speed", "$0.005-0.018/MP",
|
||||
defaults={"expansion_model": "Medium", "rendering_speed": "BALANCED"},
|
||||
supports={
|
||||
"prompt", "image_size", "expansion_model", "rendering_speed", "num_images", "seed", "sync_mode",
|
||||
},
|
||||
"max_reference_images": 10,
|
||||
},
|
||||
"bytedance/seedream/v5/lite/text-to-image": {
|
||||
"display": "Seedream 5.0 Lite",
|
||||
"speed": "~5s",
|
||||
"strengths": "Fast/cheap Seedream tier, high-res output",
|
||||
"price": "$0.035/image",
|
||||
"size_style": "image_size_preset",
|
||||
# Lite wants total pixels between 2560x1440 and 4096x4096. Use the
|
||||
# documented presets (FAL auto-scales if a preset is under the floor)
|
||||
# instead of hand-rolled ImageSize dicts that drift from the schema.
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
),
|
||||
"alibaba/qwen-image-3/text-to-image": _model(
|
||||
"Qwen Image 3", "~8s", "Complex CN/EN text rendering, prompt-guided resolution", "$0.04 (1K) / $0.075 (2K) per image",
|
||||
defaults={
|
||||
"num_images": 1, "output_format": "png", "enable_prompt_expansion": False,
|
||||
"enable_safety_checker": False,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "num_images", "max_images",
|
||||
"sync_mode", "enable_safety_checker",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"ideogram/v4/instant": {
|
||||
"display": "Ideogram V4 (Instant)",
|
||||
"speed": "<1s",
|
||||
"strengths": "Latest Ideogram typography, posters/logos, instant",
|
||||
"price": "$0.0075/MP",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"expansion_model": "Medium",
|
||||
"output_format": "png",
|
||||
"enable_safety_checker": False,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "expansion_model", "num_images",
|
||||
"seed", "sync_mode", "enable_safety_checker", "output_format",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"ideogram/v4/fast": {
|
||||
"display": "Ideogram V4 (Fast)",
|
||||
"speed": "~1s",
|
||||
"strengths": "Ideogram V4 quality tiers via rendering_speed",
|
||||
"price": "$0.005-0.018/MP",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"expansion_model": "Medium",
|
||||
"rendering_speed": "BALANCED",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "expansion_model", "rendering_speed",
|
||||
"num_images", "seed", "sync_mode",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"alibaba/qwen-image-3/text-to-image": {
|
||||
"display": "Qwen Image 3",
|
||||
"speed": "~8s",
|
||||
"strengths": "Complex CN/EN text rendering, prompt-guided resolution",
|
||||
"price": "$0.04 (1K) / $0.075 (2K) per image",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"enable_prompt_expansion": False, # avoid the LLM rewrite surprise
|
||||
"enable_safety_checker": False,
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "negative_prompt", "image_size", "num_images",
|
||||
"seed", "sync_mode", "output_format",
|
||||
supports={
|
||||
"prompt", "negative_prompt", "image_size", "num_images", "seed", "sync_mode", "output_format",
|
||||
"enable_prompt_expansion", "enable_safety_checker",
|
||||
},
|
||||
"upscale": False,
|
||||
# Qwen Image 3 edit: 1-3 reference images, identity-preserving edits.
|
||||
"edit_endpoint": "alibaba/qwen-image-3/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "negative_prompt", "num_images",
|
||||
"seed", "sync_mode", "output_format",
|
||||
edit_endpoint="alibaba/qwen-image-3/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "negative_prompt", "num_images", "seed", "sync_mode", "output_format",
|
||||
"enable_prompt_expansion", "enable_safety_checker",
|
||||
},
|
||||
"max_reference_images": 3,
|
||||
},
|
||||
"microsoft/mai-image-2.5-pro": {
|
||||
"display": "MAI Image 2.5 Pro",
|
||||
"speed": "~10s",
|
||||
"strengths": "Microsoft flagship, hero imagery, precise typography",
|
||||
"price": "~$0.17/image",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "num_images", "output_format",
|
||||
"sync_mode",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"google/nano-banana-2-lite": {
|
||||
"display": "Nano Banana 2 Lite",
|
||||
"speed": "<2s",
|
||||
"strengths": "Gemini image family, sub-2s, 14 aspect ratios incl. extreme",
|
||||
"price": "~$0.04/image (1K fixed)",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
"safety_tolerance": "5",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "num_images", "seed",
|
||||
"output_format", "safety_tolerance", "sync_mode",
|
||||
max_reference_images=3,
|
||||
),
|
||||
"microsoft/mai-image-2.5-pro": _model(
|
||||
"MAI Image 2.5 Pro", "~10s", "Microsoft flagship, hero imagery, precise typography", "~$0.17/image",
|
||||
style="aspect_ratio",
|
||||
defaults={"num_images": 1, "output_format": "png"},
|
||||
supports={"prompt", "aspect_ratio", "num_images", "output_format", "sync_mode"},
|
||||
),
|
||||
"google/nano-banana-2-lite": _model(
|
||||
"Nano Banana 2 Lite", "<2s", "Gemini image family, sub-2s, 14 aspect ratios incl. extreme", "~$0.04/image (1K fixed)",
|
||||
style="aspect_ratio",
|
||||
defaults={"num_images": 1, "output_format": "png", "safety_tolerance": "5"},
|
||||
supports={
|
||||
"prompt", "aspect_ratio", "num_images", "seed", "output_format", "safety_tolerance", "sync_mode",
|
||||
"system_prompt", "limit_generations", "thinking_level",
|
||||
},
|
||||
"upscale": False,
|
||||
# Fast multi-turn local edits with reference images via `image_urls`.
|
||||
"edit_endpoint": "google/nano-banana-2-lite/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "aspect_ratio", "num_images",
|
||||
"seed", "output_format", "safety_tolerance", "sync_mode",
|
||||
"system_prompt",
|
||||
edit_endpoint="google/nano-banana-2-lite/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "aspect_ratio", "num_images", "seed", "output_format", "safety_tolerance",
|
||||
"sync_mode", "system_prompt",
|
||||
},
|
||||
"max_reference_images": 4,
|
||||
},
|
||||
"fal-ai/recraft/v4.1/text-to-image": {
|
||||
"display": "Recraft V4.1",
|
||||
"speed": "~8s",
|
||||
"strengths": "Design-first raster, brand systems, editorial",
|
||||
"price": "$0.035/image",
|
||||
"size_style": "image_size_preset",
|
||||
"sizes": _PRESET_SIZES,
|
||||
"defaults": {
|
||||
"enable_safety_checker": False,
|
||||
max_reference_images=4,
|
||||
),
|
||||
"fal-ai/recraft/v4.1/text-to-image": _model(
|
||||
"Recraft V4.1", "~8s", "Design-first raster, brand systems, editorial", "$0.035/image",
|
||||
defaults={"enable_safety_checker": False},
|
||||
supports={
|
||||
"prompt", "image_size", "enable_safety_checker", "colors", "background_color",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "image_size", "enable_safety_checker",
|
||||
"colors", "background_color",
|
||||
),
|
||||
"xai/grok-imagine-image/v2.0/text-to-image": _model(
|
||||
"Grok Imagine Image 2.0", "~5s", "xAI. Design-grade typography/layout, instruction following", "$0.06/image (1K medium)",
|
||||
style="aspect_ratio",
|
||||
# 1k + medium is the cheapest sensible tier; 2k is roughly +33%/image. 1k native
|
||||
# is sub-2MP — pass upscale=true per call when needed. Edits omit aspect_ratio
|
||||
# (defaults to "auto", following the first input image).
|
||||
defaults={
|
||||
"num_images": 1, "output_format": "png", "resolution": "1k", "quality": "medium",
|
||||
},
|
||||
"upscale": False,
|
||||
},
|
||||
"xai/grok-imagine-image/v2.0/text-to-image": {
|
||||
"display": "Grok Imagine Image 2.0",
|
||||
"speed": "~5s",
|
||||
"strengths": "xAI. Design-grade typography/layout, instruction following",
|
||||
"price": "$0.06/image (1K medium)",
|
||||
"size_style": "aspect_ratio",
|
||||
"sizes": _ASPECT_SIZES,
|
||||
"defaults": {
|
||||
"num_images": 1,
|
||||
"output_format": "png",
|
||||
# 1k + medium is the cheapest sensible tier ($0.06/image);
|
||||
# 2k roughly +33% per image.
|
||||
"resolution": "1k",
|
||||
"quality": "medium",
|
||||
supports={
|
||||
"prompt", "aspect_ratio", "num_images", "output_format", "resolution", "quality", "sync_mode",
|
||||
},
|
||||
"supports": {
|
||||
"prompt", "aspect_ratio", "num_images", "output_format",
|
||||
"resolution", "quality", "sync_mode",
|
||||
edit_endpoint="xai/grok-imagine-image/v2.0/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "num_images", "output_format", "resolution", "quality", "sync_mode",
|
||||
},
|
||||
"upscale": False,
|
||||
# Edit endpoint takes `image_urls` (max 3) + the same knobs;
|
||||
# aspect_ratio defaults to "auto" (follows the first input image),
|
||||
# so we don't send it on edits.
|
||||
"edit_endpoint": "xai/grok-imagine-image/v2.0/edit",
|
||||
"edit_supports": {
|
||||
"prompt", "image_urls", "num_images", "output_format",
|
||||
"resolution", "quality", "sync_mode",
|
||||
},
|
||||
"max_reference_images": 3,
|
||||
},
|
||||
max_reference_images=3,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
|
||||
+265
-481
File diff suppressed because it is too large
Load Diff
+63
-155
@@ -2,33 +2,24 @@
|
||||
|
||||
All source handling (data:/http(s)/file/local/container) funnels through
|
||||
:func:`resolve_image_source` so size and magic-byte checks are enforced exactly
|
||||
once. Returns raw bytes (not a path): the downstream step is base64 -> data URL
|
||||
(RFC 2397) and provider base64 content blocks.
|
||||
once. Returns raw bytes (not a path): the downstream step is base64 -> data URL.
|
||||
|
||||
Images are the default and the historical purpose. Callers whose argument
|
||||
takes video opt in via ``permitted=("video",)`` — the same confinement and
|
||||
credential-guard pipeline applies, and only the type check at the end differs
|
||||
(extension-table typing plus an mp4 magic sniff, rather than image magic
|
||||
bytes). Every existing call site keeps the image-only default unchanged.
|
||||
Images are the default. Callers whose argument takes video opt in via
|
||||
``permitted=("video",)``: same confinement and credential-guard pipeline, only
|
||||
the final type check differs (extension table + mp4 magic sniff).
|
||||
|
||||
Security (terminal-backend confinement, GHSA-gpxw-6wxv-w3qq): under a non-local
|
||||
terminal backend the file tools are confined to the sandbox (SECURITY.md 2.2),
|
||||
but vision read images host-side. This resolver enforces the same boundary:
|
||||
backend the file tools are confined to the sandbox, so vision must be too:
|
||||
|
||||
* local backend -> read any host path (chosen posture, unchanged)
|
||||
* local backend -> read any host path (chosen posture)
|
||||
* non-local backend:
|
||||
path in a media cache -> host-read (the gateway/download caches live on
|
||||
the host and are bind-mounted into the sandbox)
|
||||
path anywhere else -> read the bytes *inside the sandbox* via exec-read
|
||||
(the agent can already ``cat`` any container file;
|
||||
this stays within the sandbox boundary and never
|
||||
reaches the host's ``/etc/passwd`` / ``~/.ssh``).
|
||||
path in a media cache -> host-read (gateway/download caches live on the
|
||||
host and are bind-mounted into the sandbox)
|
||||
path anywhere else -> read *inside the sandbox* via exec-read (the
|
||||
agent can already ``cat`` any container file)
|
||||
|
||||
So a prompt-injected ``vision_analyze('/etc/passwd')`` under Docker reads the
|
||||
*container's* file (what every other tool sees), not the host's — no escape —
|
||||
while container-only images (tmpfs ``/workspace``, root-owned) are still
|
||||
deliverable. This is the unified delivery + confinement model: the same
|
||||
mechanism that fixes "vision can't see container files" also closes the escape.
|
||||
container's file, never the host's, while container-only images stay deliverable.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -40,12 +31,9 @@ from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
# Raw-bytes INGEST budget — what the resolver will load before handing off.
|
||||
# This is deliberately the 50MB download cap (tools/vision_tools._VISION_MAX_DOWNLOAD_BYTES),
|
||||
# NOT the 20MB provider payload cap. The 20MB cap (_MAX_BASE64_BYTES) is a
|
||||
# *post-resize* limit enforced at the call sites: an oversized raw image must
|
||||
# still reach the resizer so it can be downscaled under the payload cap. Capping
|
||||
# raw bytes at 20MB here would reject every 20-50MB photo before resize can run.
|
||||
# Raw-bytes INGEST budget: deliberately the 50MB download cap, NOT the 20MB
|
||||
# provider payload cap — that one is enforced post-resize at the call sites, and
|
||||
# a 20-50MB photo must still reach the resizer.
|
||||
_MAX_INGEST_BYTES = 50 * 1024 * 1024
|
||||
|
||||
|
||||
@@ -87,8 +75,7 @@ class ResolvedImage:
|
||||
origin: str # one of: data | http | file | local | container
|
||||
|
||||
|
||||
# Explicit URL scheme, e.g. "ftp://", "s3://". Bare Windows drive paths
|
||||
# ("C:\x.png") don't match because they lack the "//".
|
||||
# Explicit URL scheme ("ftp://", "s3://"). Bare Windows drive paths lack the "//".
|
||||
_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*://")
|
||||
|
||||
|
||||
@@ -118,23 +105,15 @@ async def resolve_image_source(
|
||||
)
|
||||
|
||||
# Everything else is a filesystem path — including bare relative names
|
||||
# like "pic.png" (accepted on main; a path-shape gate here regressed them).
|
||||
# like "pic.png" (a path-shape gate here regressed them once).
|
||||
candidate = s[len("file://"):] if s.lower().startswith("file://") else s
|
||||
p = Path(os.path.expanduser(candidate))
|
||||
# Confinement decision (see module docstring). Under a non-local backend
|
||||
# a path is host-readable ONLY if it lands in a media cache (after
|
||||
# translating a container-visible cache path back to its host mount);
|
||||
# every other path is read inside the sandbox via exec-read, so a host
|
||||
# path outside the caches never yields the host's bytes.
|
||||
host_target = _permitted_host_read_target(p, ctx)
|
||||
if host_target is not None and host_target.is_file():
|
||||
# Shared credential-read guard (agent.file_safety, #57698): refuse
|
||||
# secret-bearing files (.env, auth.json, ...) with an intentional,
|
||||
# specific error instead of relying on the magic-byte sniff to
|
||||
# reject them incidentally. Same chokepoint the image-gen/video-gen
|
||||
# provider plugins enforce on model-supplied local paths. Import is
|
||||
# best-effort (guard unavailability must not break image loading);
|
||||
# a real block always propagates.
|
||||
# Shared credential-read guard: refuse secret-bearing files (.env,
|
||||
# auth.json) with a specific error rather than relying on the magic
|
||||
# sniff to reject them incidentally. Guard import is best-effort; a
|
||||
# real block always propagates.
|
||||
try:
|
||||
from agent.file_safety import raise_if_read_blocked
|
||||
except Exception: # noqa: BLE001 — guard unavailable: proceed
|
||||
@@ -147,12 +126,8 @@ async def resolve_image_source(
|
||||
data = await asyncio.to_thread(host_target.read_bytes)
|
||||
return _finalize(data, "", "file", s, permitted)
|
||||
if _is_local_terminal_backend():
|
||||
# Local backend: any path was host-readable, so a miss simply means
|
||||
# the file doesn't exist — no sandbox to fall back to.
|
||||
# Any path was host-readable, so a miss means the file doesn't exist.
|
||||
raise SourceNotFound(f"media file not found: '{p}'", src=s, origin="file")
|
||||
# Not a permitted host read (or the host file is absent) -> read the
|
||||
# bytes inside the sandbox. Under a sandbox this reads the container's
|
||||
# filesystem, never the host's.
|
||||
return await _resolve_container_fallback(p, ctx, s, permitted)
|
||||
|
||||
|
||||
@@ -172,15 +147,9 @@ def _resolve_data_url(s: str) -> tuple[bytes, str]:
|
||||
|
||||
|
||||
def _http_block_reason(url: str) -> Optional[str]:
|
||||
"""Return a human-readable block reason, or None when the URL is allowed.
|
||||
|
||||
Pre-flight short-circuit: policy-blocked URLs are refused BEFORE any
|
||||
network I/O. ``_download_image`` re-checks policy internally (per attempt
|
||||
and against the final redirect target) — that second evaluation is
|
||||
intentional, not redundant: this one guarantees no bytes move for a
|
||||
blocked URL; the inner one covers redirects and non-resolver callers.
|
||||
Preserves the specific website-policy message so the agent sees *why*.
|
||||
"""
|
||||
"""Block reason, or None when allowed. Refuses policy-blocked URLs BEFORE any
|
||||
network I/O; ``_download_image`` re-checks per attempt and against the final
|
||||
redirect target — the second evaluation is intentional, not redundant."""
|
||||
from tools.url_safety import is_safe_url
|
||||
from tools.website_policy import check_website_access
|
||||
|
||||
@@ -210,27 +179,19 @@ async def _download_to_bytes(url: str) -> bytes:
|
||||
|
||||
|
||||
def _is_local_terminal_backend() -> bool:
|
||||
"""True when the terminal backend runs directly on the host.
|
||||
|
||||
Mirrors ``tools.browser_tool._is_local_backend`` and terminal_tool's own
|
||||
dispatch, which key off ``TERMINAL_ENV``.
|
||||
"""
|
||||
"""True when the terminal backend runs directly on the host (keys off ``TERMINAL_ENV``)."""
|
||||
return os.getenv("TERMINAL_ENV", "local").strip().lower() in ("local", "")
|
||||
|
||||
|
||||
def _media_cache_roots() -> list:
|
||||
"""Agent-managed media cache directories under HERMES_HOME (host side).
|
||||
|
||||
The only host paths vision may read under a non-local backend: gateway-
|
||||
downloaded inbound media and the tools' own URL-download temp dirs. Covers
|
||||
the consolidated ``cache/`` layout and the legacy flat directories.
|
||||
"""
|
||||
"""Host-side media caches: the only host paths vision may read under a
|
||||
non-local backend (gateway inbound media + the tools' own download temp dirs)."""
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
home = get_hermes_home()
|
||||
return [
|
||||
home / "cache", # cache/images, cache/vision, cache/video(s), cache/audio
|
||||
home / "images", # desktop/clipboard/PDF uploads (tui_gateway) — #69575
|
||||
home / "images", # desktop/clipboard/PDF uploads (tui_gateway)
|
||||
home / "image_cache",
|
||||
home / "audio_cache",
|
||||
home / "video_cache",
|
||||
@@ -240,14 +201,12 @@ def _media_cache_roots() -> list:
|
||||
|
||||
|
||||
def _permitted_host_read_target(p: Path, ctx: ResolveContext) -> Optional[Path]:
|
||||
"""Return the host path to read, or ``None`` if a host read is not permitted.
|
||||
"""Host path to read, or ``None`` if a host read is not permitted.
|
||||
|
||||
- Local backend: any path is permitted (chosen posture). Returns ``p``.
|
||||
- Non-local backend: permitted only if the path resolves inside a media
|
||||
cache root. A container-visible cache path (e.g. ``/root/.hermes/cache/
|
||||
images/x.png``) is first translated back to its host mount; anything that
|
||||
is not under a cache returns ``None`` so the caller routes it to the
|
||||
in-sandbox exec-read instead of reading the host filesystem.
|
||||
Local backend: any path. Non-local: only paths resolving inside a media
|
||||
cache root (a container-visible cache path is first translated back to its
|
||||
host mount); anything else returns ``None`` so the caller exec-reads inside
|
||||
the sandbox instead of touching the host filesystem.
|
||||
"""
|
||||
if _is_local_terminal_backend():
|
||||
try:
|
||||
@@ -283,13 +242,12 @@ def _get_active_env(task_id: Optional[str]):
|
||||
|
||||
|
||||
def _ensure_container_env(task_id: Optional[str]) -> None:
|
||||
"""Lazily bring up the sandbox (SSH/Docker/…) before an in-sandbox read.
|
||||
"""Lazily bring up the sandbox before an in-sandbox read.
|
||||
|
||||
Unlike the terminal tool, vision never triggered environment creation, so a
|
||||
session whose first action is ``vision_analyze`` on a container-only path
|
||||
under a non-local backend found no active env and failed — until a terminal
|
||||
command happened to create one (issue #62825). Best-effort: any failure just
|
||||
leaves the env absent and the caller hits the existing fail-closed error.
|
||||
Vision never triggered environment creation, so a session whose first action
|
||||
was ``vision_analyze`` on a container-only path found no active env until a
|
||||
terminal command created one. Best-effort: failure leaves the env absent and
|
||||
the caller hits the fail-closed error.
|
||||
"""
|
||||
if not task_id:
|
||||
return
|
||||
@@ -304,35 +262,17 @@ def _ensure_container_env(task_id: Optional[str]) -> None:
|
||||
async def _resolve_container_fallback(
|
||||
p: Path, ctx: ResolveContext, src: str, permitted: tuple = ("image",)
|
||||
) -> ResolvedImage:
|
||||
"""Read the image bytes inside the sandbox (fail-closed when none exists).
|
||||
"""Read the bytes inside the sandbox; fail-closed when no env exists (a
|
||||
non-cache host path under a sandbox must never leak via a host fallback).
|
||||
|
||||
Reached when a host read is not permitted or the host file is absent. The
|
||||
agent can already ``cat`` any container file (file_operations.py reads
|
||||
root-owned mode-600 files this way), so this stays within the same sandbox
|
||||
boundary and never touches the host filesystem. ``--`` stops a leading-dash
|
||||
path from being parsed as a ``base64`` option; ``base64 -w0`` is GNU-only,
|
||||
so pipe through ``tr -d`` for BusyBox.
|
||||
|
||||
Fail-closed: if there is no active sandbox env we refuse rather than falling
|
||||
back to a host read, so a non-cache host path under a sandbox never leaks.
|
||||
|
||||
Cold-start retry: under Docker the very first exec against a freshly
|
||||
started container can fail (empty pipe / partial setup) while an identical
|
||||
second call succeeds. We retry once with a short delay before giving up,
|
||||
so callers don't see "could not read inside the sandbox" on a file that is
|
||||
verifiably readable on the immediate retry. See #76566.
|
||||
|
||||
Diagnostic: when every attempt fails, the container's own output (stderr
|
||||
+ stdout) is folded into the raised error so the user can distinguish
|
||||
"no such file" from "permission denied" from "container never came up"
|
||||
instead of staring at one opaque message.
|
||||
Cold-start retry: under Docker the first exec against a fresh container can
|
||||
fail (empty pipe) while an identical second call succeeds, so retry once
|
||||
after a short delay. On final failure the container's own output is folded
|
||||
into the error so "no such file" / "permission denied" / "never came up"
|
||||
are distinguishable.
|
||||
"""
|
||||
import asyncio
|
||||
import shlex
|
||||
|
||||
# Bring the sandbox up on demand: without this, the first vision_analyze of
|
||||
# a session (before any terminal command) has no active env to read from
|
||||
# under a non-local backend (issue #62825).
|
||||
_ensure_container_env(ctx.task_id)
|
||||
|
||||
env = _get_active_env(ctx.task_id)
|
||||
@@ -342,14 +282,11 @@ async def _resolve_container_fallback(
|
||||
f"session is available to read it",
|
||||
src=src, origin="container")
|
||||
|
||||
# Bound the read INSIDE the sandbox: head -c caps at ingest-limit+1 bytes
|
||||
# so a huge file (or /dev/zero) can't stream unbounded base64 into host
|
||||
# memory — the +1 byte lets us distinguish "exactly at the cap" from
|
||||
# "over the cap" after decode. The input redirect (< path) avoids argv
|
||||
# entirely, so leading-dash paths can't be parsed as options; base64
|
||||
# -w0 is GNU-only, so pipe through tr -d for BusyBox.
|
||||
# env.execute is a blocking backend exec; keep it off the event loop so a
|
||||
# multi-MB base64 read doesn't stall every other coroutine.
|
||||
# Bound the read INSIDE the sandbox: head -c caps at ingest-limit+1 so
|
||||
# /dev/zero can't stream unbounded base64 into host memory (the +1
|
||||
# distinguishes "at the cap" from "over"). The input redirect avoids argv, so
|
||||
# leading-dash paths can't parse as options; base64 -w0 is GNU-only, hence
|
||||
# tr -d for BusyBox. env.execute blocks — keep it off the event loop.
|
||||
qp = shlex.quote(str(p))
|
||||
cmd = f"head -c {_MAX_INGEST_BYTES + 1} < {qp} | base64 | tr -d '\\n'"
|
||||
|
||||
@@ -359,14 +296,9 @@ async def _resolve_container_fallback(
|
||||
if last_res.get("returncode", 1) == 0:
|
||||
break
|
||||
if attempt == 0:
|
||||
# Cold-start: give the container a moment to settle its pipes
|
||||
# before retrying. 150ms covers Docker exec warm-up in practice
|
||||
# without making a real failure feel sluggish.
|
||||
await asyncio.sleep(0.15)
|
||||
await asyncio.sleep(0.15) # covers Docker exec warm-up in practice
|
||||
if last_res.get("returncode", 1) != 0:
|
||||
diag = (last_res.get("output") or "").strip().splitlines()
|
||||
# Keep the diagnostic small and noise-free: first non-empty line,
|
||||
# trimmed to a sane length so it slots into the agent's error UI.
|
||||
first = next((ln.strip() for ln in diag if ln.strip()), "")
|
||||
suffix = f" ({first[:200]})" if first else ""
|
||||
raise SourceNotFound(
|
||||
@@ -384,18 +316,11 @@ async def _resolve_container_fallback(
|
||||
def _finalize(
|
||||
data: bytes, declared_mime: str, origin: str, src: str, permitted: tuple = ("image",)
|
||||
) -> ResolvedImage:
|
||||
"""Intrinsic-correctness chokepoint: ingest byte cap + type check.
|
||||
"""Chokepoint: 50MB ingest cap + type check.
|
||||
|
||||
The cap here is the generous 50MB *ingest* budget, not the 20MB provider
|
||||
payload cap — a 20-50MB image must survive this step so the call site can
|
||||
resize it under the payload cap. See ``_MAX_INGEST_BYTES``.
|
||||
|
||||
Images are typed by magic bytes. Video (opt-in via ``permitted``) is typed
|
||||
by the extension table plus an mp4 container sniff: extension typing is
|
||||
sufficient because every downstream consumer re-validates — the upload
|
||||
gateway signs the content type into its presigned URL and the vendor
|
||||
rejects undecodable input — so a wrong guess is a clean rejection there
|
||||
rather than a hole here.
|
||||
Images are typed by magic bytes. Video (opt-in) is typed by extension plus
|
||||
an mp4 container sniff — sufficient because every downstream consumer
|
||||
re-validates, so a wrong guess is a clean rejection there, not a hole here.
|
||||
"""
|
||||
from tools.vision_tools import _detect_image_mime_type_from_bytes
|
||||
|
||||
@@ -409,9 +334,7 @@ def _finalize(
|
||||
return ResolvedImage(data=data, mime=sniffed, origin=origin)
|
||||
|
||||
if "image" in permitted and b"<svg" in data[:4096].lower():
|
||||
# Pass SVG through — the vision call sites rasterize it to PNG
|
||||
# via _normalize_to_supported_image before embedding (providers
|
||||
# only ingest raster images).
|
||||
# Pass SVG through — call sites rasterize it to PNG before embedding.
|
||||
return ResolvedImage(data=data, mime="image/svg+xml", origin=origin)
|
||||
|
||||
if "video" in permitted:
|
||||
@@ -424,11 +347,8 @@ def _finalize(
|
||||
|
||||
|
||||
def _detect_video_mime(data: bytes, src: str) -> Optional[str]:
|
||||
"""Video MIME from the extension table, else the mp4/mov container magic.
|
||||
|
||||
The magic fallback covers extensionless sources (data: URLs, URLs with
|
||||
query strings): ISO base-media files carry ``ftyp`` at offset 4.
|
||||
"""
|
||||
"""Video MIME from the extension table, else the ISO base-media ``ftyp``
|
||||
magic at offset 4 (covers extensionless data: URLs / query-string URLs)."""
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from tools.vision_tools import _detect_video_mime_type
|
||||
@@ -447,23 +367,11 @@ async def resolve_local_source_to_data_url(
|
||||
) -> str:
|
||||
"""Convert a path-like media source into a ``data:`` URL via the resolver.
|
||||
|
||||
Generation tools (image_generate / video_generate) forward model-supplied
|
||||
source images to provider plugins, which historically read local paths off
|
||||
the HOST filesystem regardless of terminal backend. Under a non-local
|
||||
backend that is both broken (the file usually lives in the sandbox, so the
|
||||
host read misses) and inconsistent with the confinement model vision/video
|
||||
analysis enforce (GHSA-gpxw-6wxv-w3qq): the sandbox boundary should govern
|
||||
every model-supplied path.
|
||||
|
||||
This helper is the dispatch-layer chokepoint: URL-shaped sources
|
||||
(http/https/data) pass through untouched; anything path-like resolves
|
||||
through :func:`resolve_image_source` — media-cache host reads, bounded
|
||||
in-sandbox exec-read, lazy env bring-up, credential guard, ingest cap —
|
||||
and comes back as a ``data:`` URL every provider already accepts.
|
||||
|
||||
Callers apply this only under a non-local terminal backend: on the local
|
||||
backend providers keep their existing host-side reads (chosen posture,
|
||||
zero behavior change).
|
||||
Dispatch-layer chokepoint for generation tools: providers historically read
|
||||
model-supplied local paths off the HOST regardless of backend — broken under
|
||||
a sandbox (the file lives there) and inconsistent with the confinement model
|
||||
vision enforces. URL-shaped sources (http/https/data) pass through untouched.
|
||||
Callers apply this only under a non-local backend (local keeps host reads).
|
||||
"""
|
||||
s = (src or "").strip()
|
||||
if not s or s.lower().startswith(("http://", "https://", "data:")):
|
||||
|
||||
+89
-153
@@ -1,28 +1,11 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Video Generation Tool
|
||||
=====================
|
||||
"""``video_generate``: one tool dispatching to a plugin-registered :class:`VideoGenProvider`
|
||||
(``agent/video_gen_provider.py`` ABC, ``agent/video_gen_registry.py``, ``plugins/video_gen/<name>/``).
|
||||
|
||||
Single ``video_generate`` tool that dispatches to a plugin-registered
|
||||
video generation provider. Mirrors the ``image_generate`` design:
|
||||
|
||||
- ``agent/video_gen_provider.py`` defines the :class:`VideoGenProvider` ABC.
|
||||
- ``agent/video_gen_registry.py`` holds the active providers (populated by
|
||||
plugins at import time).
|
||||
- Each provider lives under ``plugins/video_gen/<name>/``.
|
||||
|
||||
The tool is backend-agnostic and ships **no in-tree provider** — enable a
|
||||
plugin (``hermes plugins enable video_gen/<name>``) and select it in
|
||||
``hermes tools`` → Video Generation.
|
||||
|
||||
One tool covers text-to-video, image-to-video and reference-to-video with a
|
||||
compact schema (prompt, image_url, reference_image_urls, duration,
|
||||
aspect_ratio, resolution, negative_prompt, audio, seed, model). Providers
|
||||
ignore parameters they do not support: the tool layer does only lightweight
|
||||
validation (type/required-prompt) and each provider clamps inside
|
||||
:meth:`VideoGenProvider.generate`, so the surface stays stable as providers
|
||||
with different capabilities ship. Video edit/extend are intentionally not
|
||||
exposed here; providers with those workflows expose separate tools.
|
||||
Ships **no in-tree provider**: enable a plugin and select it in ``hermes tools`` → Video
|
||||
Generation. Covers text-, image- and reference-to-video; the tool layer only does lightweight
|
||||
validation and each provider clamps/ignores unsupported params inside ``generate``, so the
|
||||
surface stays stable as providers ship. Video edit/extend are deliberately not exposed here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -45,13 +28,9 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
|
||||
"name": "video_generate",
|
||||
# Placeholder — description AND params are rebuilt dynamically at
|
||||
# get_tool_definitions() time from the active provider's declared
|
||||
# capabilities() and the active model's catalog entry. Optional args
|
||||
# (image_url, reference_image_urls, negative_prompt, audio, seed,
|
||||
# upscale) are advertised ONLY when the active backend/model honors
|
||||
# them; the handler accepts them regardless (replay compat — providers
|
||||
# clamp/ignore). See _build_dynamic_video_schema().
|
||||
# Placeholder: description AND params are rebuilt at get_tool_definitions() time by
|
||||
# _build_dynamic_video_schema() from capabilities() + the model's catalog entry. Optional
|
||||
# args are advertised ONLY when honored; the handler accepts them regardless (replay compat).
|
||||
"description": "(rebuilt at get_definitions() time — see _build_dynamic_video_schema)",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
@@ -89,9 +68,7 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
|
||||
"``video_gen.model``. Unknown models are rejected."
|
||||
),
|
||||
},
|
||||
# image_url / reference_image_urls / negative_prompt / audio / seed /
|
||||
# upscale are added per-capability by _build_dynamic_video_schema.
|
||||
# Do not re-add them statically.
|
||||
# Capability-gated args are added by _build_dynamic_video_schema; never statically.
|
||||
},
|
||||
"required": ["prompt"],
|
||||
},
|
||||
@@ -101,8 +78,6 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
|
||||
# ---------------------------------------------------------------------------
|
||||
# Config readers (mirror image_generation_tool.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _read_video_gen_key(key: str) -> Optional[str]:
|
||||
"""Return the stripped ``video_gen.<key>`` string from config.yaml, or None."""
|
||||
try:
|
||||
@@ -127,23 +102,23 @@ def _read_configured_video_model() -> Optional[str]:
|
||||
return _read_video_gen_key("model")
|
||||
|
||||
|
||||
def _discovered_registry():
|
||||
"""Import the provider registry after (idempotent) plugin discovery so user-installed plugins are visible."""
|
||||
from agent import video_gen_registry
|
||||
from hermes_cli.plugins import _ensure_plugins_discovered
|
||||
|
||||
_ensure_plugins_discovered()
|
||||
return video_gen_registry, _ensure_plugins_discovered
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Availability check + provider resolution
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def check_video_generation_requirements() -> bool:
|
||||
"""True when at least one registered provider reports available.
|
||||
|
||||
Triggers plugin discovery (idempotent) so user-installed plugins are
|
||||
visible to the toolset gate.
|
||||
"""
|
||||
"""True when at least one registered provider reports available."""
|
||||
try:
|
||||
from agent.video_gen_registry import list_providers
|
||||
from hermes_cli.plugins import _ensure_plugins_discovered
|
||||
|
||||
_ensure_plugins_discovered()
|
||||
for provider in list_providers():
|
||||
registry_mod, _ = _discovered_registry()
|
||||
for provider in registry_mod.list_providers():
|
||||
try:
|
||||
if provider.is_available():
|
||||
return True
|
||||
@@ -155,20 +130,14 @@ def check_video_generation_requirements() -> bool:
|
||||
|
||||
|
||||
def _resolve_active_provider():
|
||||
"""Return the active provider object or None.
|
||||
|
||||
Forces a discovery refresh on a miss — handles long-lived sessions that
|
||||
started before a plugin was installed.
|
||||
"""
|
||||
"""Active provider or None; forces a discovery refresh on a miss (long-lived sessions
|
||||
that started before a plugin was installed)."""
|
||||
try:
|
||||
from agent.video_gen_registry import get_active_provider
|
||||
from hermes_cli.plugins import _ensure_plugins_discovered
|
||||
|
||||
_ensure_plugins_discovered()
|
||||
provider = get_active_provider()
|
||||
registry_mod, ensure_discovered = _discovered_registry()
|
||||
provider = registry_mod.get_active_provider()
|
||||
if provider is None:
|
||||
_ensure_plugins_discovered(force=True)
|
||||
provider = get_active_provider()
|
||||
ensure_discovered(force=True)
|
||||
provider = registry_mod.get_active_provider()
|
||||
return provider
|
||||
except Exception as exc:
|
||||
logger.debug("video_gen provider resolution failed: %s", exc)
|
||||
@@ -177,30 +146,28 @@ def _resolve_active_provider():
|
||||
|
||||
def _missing_provider_error(configured: Optional[str]) -> str:
|
||||
if configured:
|
||||
msg = (
|
||||
f"video_gen.provider='{configured}' is set but no plugin "
|
||||
f"registered that name. Run `hermes plugins list` to see "
|
||||
f"installed video gen backends, or `hermes tools` → Video "
|
||||
f"Generation to pick one."
|
||||
)
|
||||
return json.dumps(error_response(
|
||||
error=msg, error_type="provider_not_registered",
|
||||
error=(
|
||||
f"video_gen.provider='{configured}' is set but no plugin "
|
||||
f"registered that name. Run `hermes plugins list` to see "
|
||||
f"installed video gen backends, or `hermes tools` → Video "
|
||||
f"Generation to pick one."
|
||||
),
|
||||
error_type="provider_not_registered",
|
||||
provider=configured,
|
||||
))
|
||||
msg = (
|
||||
"No video generation backend is configured. Run `hermes tools` → "
|
||||
"Video Generation to enable one (xAI, FAL, or Google Veo)."
|
||||
)
|
||||
return json.dumps(error_response(
|
||||
error=msg, error_type="no_provider_configured",
|
||||
error=(
|
||||
"No video generation backend is configured. Run `hermes tools` → "
|
||||
"Video Generation to enable one (xAI, FAL, or Google Veo)."
|
||||
),
|
||||
error_type="no_provider_configured",
|
||||
))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Handler
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _coerce_int(value: Any) -> Optional[int]:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
@@ -211,16 +178,11 @@ def _coerce_int(value: Any) -> Optional[int]:
|
||||
|
||||
|
||||
def _coerce_bool(value: Any) -> Optional[bool]:
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
if isinstance(value, str):
|
||||
v = value.strip().lower()
|
||||
if v in {"true", "1", "yes", "on"}:
|
||||
return True
|
||||
if v in {"false", "0", "no", "off"}:
|
||||
return False
|
||||
return {"true": True, "1": True, "yes": True, "on": True,
|
||||
"false": False, "0": False, "no": False, "off": False}.get(value.strip().lower())
|
||||
return None
|
||||
|
||||
|
||||
@@ -241,8 +203,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
|
||||
reference_image_urls = _normalize_reference_images(args.get("reference_image_urls"))
|
||||
task_id = _kw.get("task_id")
|
||||
|
||||
# Confinement chokepoint (mirrors image_generate): under a non-local
|
||||
# backend, path-like source images reach providers as data: URLs.
|
||||
# Confinement chokepoint (mirrors image_generate): non-local backends hand providers data: URLs.
|
||||
from tools.image_generation_tool import _confine_source_images
|
||||
|
||||
image_url, reference_image_urls, confine_error = _confine_source_images(
|
||||
@@ -258,8 +219,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
|
||||
upscale = _coerce_bool(args.get("upscale"))
|
||||
model_override = (args.get("model") or "").strip() or None
|
||||
|
||||
# Soft validation — providers do their own. The backend may accept
|
||||
# image-only on its image-to-video endpoint, but our surface always needs a prompt.
|
||||
# Soft validation — providers do their own; a backend may accept image-only, our surface never does.
|
||||
if not prompt:
|
||||
return tool_error("prompt is required for video generation")
|
||||
if "operation" in args or "video_url" in args:
|
||||
@@ -291,7 +251,6 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
|
||||
}
|
||||
# Drop None entries so providers see clean defaults.
|
||||
kwargs = {k: v for k, v in kwargs.items() if v is not None}
|
||||
|
||||
pname = getattr(provider, "name", "?")
|
||||
|
||||
def _err(error: str, error_type: str) -> str:
|
||||
@@ -303,8 +262,7 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
|
||||
try:
|
||||
result = provider.generate(prompt=prompt, **kwargs)
|
||||
except TypeError as exc:
|
||||
# A provider that hasn't widened its signature is a plugin bug, not a
|
||||
# caller error — surface a clear contract message.
|
||||
# An un-widened provider signature is a plugin bug, not a caller error.
|
||||
logger.warning(
|
||||
"video_gen provider '%s' rejected kwargs (signature too narrow): %s",
|
||||
pname, exc,
|
||||
@@ -328,12 +286,33 @@ def _handle_video_generate(args: Dict[str, Any], **_kw: Any) -> str:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Dynamic schema — reflect the active backend's actual capabilities
|
||||
# ---------------------------------------------------------------------------
|
||||
# The configured backend determines which modalities, aspect ratios,
|
||||
# resolutions, durations and audio/negative-prompt flags are real; surfacing
|
||||
# the per-model surface in the description means the model usually gets the
|
||||
# call right first try. model_tools.get_tool_definitions() keys its cache on
|
||||
# config.yaml mtime, so the schema rebuilds when provider/model changes.
|
||||
# Surfacing the per-model surface (modalities, enums, durations, audio/negative-prompt)
|
||||
# means the model usually gets the call right first try. model_tools.get_tool_definitions()
|
||||
# keys its cache on config.yaml mtime, so the schema rebuilds on provider/model change.
|
||||
|
||||
# Optional params advertised only when the provider's capabilities() sets the flag
|
||||
# (order = schema property order).
|
||||
_CAPABILITY_PARAMS = (
|
||||
("supports_negative_prompt", "negative_prompt", {
|
||||
"type": "string",
|
||||
"description": "Content to avoid in the output.",
|
||||
}),
|
||||
("supports_audio", "audio", {
|
||||
"type": "boolean",
|
||||
"description": "Enable native audio generation (affects pricing tier).",
|
||||
}),
|
||||
("supports_seed", "seed", {
|
||||
"type": "integer",
|
||||
"description": "Seed for reproducible outputs.",
|
||||
}),
|
||||
("supports_upscale", "upscale", {
|
||||
"type": "boolean",
|
||||
"description": (
|
||||
"High-resolution pass via the backend's video upscaler "
|
||||
"(~2x, extra cost/latency). Omit for native resolution."
|
||||
),
|
||||
}),
|
||||
)
|
||||
|
||||
_GENERIC_DESCRIPTION = (
|
||||
"Generate a video from a text prompt (text-to-video), animate a "
|
||||
@@ -352,14 +331,16 @@ _GENERIC_DESCRIPTION = (
|
||||
)
|
||||
|
||||
|
||||
def _build_dynamic_video_schema() -> Dict[str, Any]:
|
||||
"""Render description AND params from the active backend's declared surface.
|
||||
def _schema(description: str, properties: Dict[str, Any]) -> Dict[str, Any]:
|
||||
return {
|
||||
"description": description,
|
||||
"parameters": {"type": "object", "properties": properties, "required": ["prompt"]},
|
||||
}
|
||||
|
||||
Optional args are advertised only when the resolved provider/model honors
|
||||
them (capabilities() + the model's catalog entry); enums and duration
|
||||
bounds tighten to the active model's sets. The handler still accepts
|
||||
unadvertised args (replay compat): providers clamp or ignore.
|
||||
"""
|
||||
|
||||
def _build_dynamic_video_schema() -> Dict[str, Any]:
|
||||
"""Render description AND params from capabilities() + the model's catalog entry; enums and
|
||||
duration bounds tighten to the active model. Unadvertised args are still accepted (replay compat)."""
|
||||
static_props = VIDEO_GENERATE_SCHEMA["parameters"]["properties"]
|
||||
parts: List[str] = [_GENERIC_DESCRIPTION]
|
||||
|
||||
@@ -371,14 +352,7 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
|
||||
"\nNo video backend is available. Calls will return an error "
|
||||
"until the user picks one via `hermes tools` → Video Generation."
|
||||
)
|
||||
return {
|
||||
"description": "\n".join(parts),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"prompt": static_props["prompt"]},
|
||||
"required": ["prompt"],
|
||||
},
|
||||
}
|
||||
return _schema("\n".join(parts), {"prompt": static_props["prompt"]})
|
||||
|
||||
try:
|
||||
caps = provider.capabilities() or {}
|
||||
@@ -388,17 +362,11 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
|
||||
models = provider.list_models() or []
|
||||
except Exception:
|
||||
models = []
|
||||
|
||||
active_model = configured_model or provider.default_model()
|
||||
model_meta = next(
|
||||
(m for m in models if isinstance(m, dict) and m.get("id") == active_model),
|
||||
{},
|
||||
)
|
||||
model_meta = next((m for m in models if isinstance(m, dict) and m.get("id") == active_model), {})
|
||||
|
||||
# ---- description -------------------------------------------------
|
||||
# Model caveats surface only what differs from the backend's overall
|
||||
# capabilities. FAL's plugin uses the singular ``modality`` key for
|
||||
# single-modality entries.
|
||||
# Model caveats surface only what differs from the backend's overall capabilities.
|
||||
# FAL's plugin uses the singular ``modality`` key for single-modality entries.
|
||||
model_modalities = set(model_meta.get("modalities") or [])
|
||||
modality = model_meta.get("modality")
|
||||
if modality:
|
||||
@@ -435,7 +403,6 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
|
||||
if notice:
|
||||
parts.append(f"- storage: {notice}")
|
||||
|
||||
# ---- params ------------------------------------------------------
|
||||
properties: Dict[str, Any] = {"prompt": static_props["prompt"]}
|
||||
|
||||
if can_i2v:
|
||||
@@ -477,48 +444,17 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
|
||||
param["enum"] = list(caps[caps_key])
|
||||
properties[key] = param
|
||||
|
||||
if caps.get("supports_negative_prompt"):
|
||||
properties["negative_prompt"] = {
|
||||
"type": "string",
|
||||
"description": "Content to avoid in the output.",
|
||||
}
|
||||
if caps.get("supports_audio"):
|
||||
properties["audio"] = {
|
||||
"type": "boolean",
|
||||
"description": (
|
||||
"Enable native audio generation (affects pricing tier)."
|
||||
),
|
||||
}
|
||||
elif caps.get("audio_always_on"):
|
||||
for flag, key, param in _CAPABILITY_PARAMS:
|
||||
if caps.get(flag):
|
||||
properties[key] = param
|
||||
if caps.get("audio_always_on") and not caps.get("supports_audio"):
|
||||
parts.append(
|
||||
"- audio: native stereo audio is generated with every video "
|
||||
"(always on; no toggle) — describe the desired sound in the "
|
||||
"prompt"
|
||||
)
|
||||
if caps.get("supports_seed"):
|
||||
properties["seed"] = {
|
||||
"type": "integer",
|
||||
"description": "Seed for reproducible outputs.",
|
||||
}
|
||||
if caps.get("supports_upscale"):
|
||||
properties["upscale"] = {
|
||||
"type": "boolean",
|
||||
"description": (
|
||||
"High-resolution pass via the backend's video upscaler "
|
||||
"(~2x, extra cost/latency). Omit for native resolution."
|
||||
),
|
||||
}
|
||||
|
||||
properties["model"] = static_props["model"]
|
||||
|
||||
return {
|
||||
"description": "\n".join(parts),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": properties,
|
||||
"required": ["prompt"],
|
||||
},
|
||||
}
|
||||
return _schema("\n".join(parts), properties)
|
||||
|
||||
|
||||
registry.register(
|
||||
|
||||
+602
-1523
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,313 @@
|
||||
"""Image format detection, normalization and region cropping for vision tools.
|
||||
|
||||
Everything here runs BEFORE an image is base64-embedded. A vision tool result
|
||||
is baked into immutable conversation history and re-sent every turn, so an
|
||||
unsupported media type or corrupt bytes would wedge the session with a
|
||||
non-retryable 400 on every resume — normalization must happen up front.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import uuid
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
from hermes_constants import get_hermes_dir
|
||||
|
||||
logger = logging.getLogger("tools.vision_tools")
|
||||
|
||||
_EXTENSION_MIME_TYPES = {
|
||||
".jpg": "image/jpeg",
|
||||
".jpeg": "image/jpeg",
|
||||
".png": "image/png",
|
||||
".gif": "image/gif",
|
||||
".bmp": "image/bmp",
|
||||
".webp": "image/webp",
|
||||
".svg": "image/svg+xml",
|
||||
}
|
||||
|
||||
# Media types the major vision providers (Anthropic in particular) accept
|
||||
# inline. SVG/BMP/TIFF are rejected with a non-retryable 400.
|
||||
_ANTHROPIC_SUPPORTED_MEDIA_TYPES = frozenset(
|
||||
{"image/jpeg", "image/png", "image/gif", "image/webp"}
|
||||
)
|
||||
|
||||
|
||||
def _determine_mime_type(image_path: Path) -> str:
|
||||
"""MIME type from file extension (defaults to image/jpeg)."""
|
||||
return _EXTENSION_MIME_TYPES.get(image_path.suffix.lower(), "image/jpeg")
|
||||
|
||||
|
||||
def _detect_image_mime_type_from_bytes(data: bytes) -> Optional[str]:
|
||||
"""Magic-byte MIME sniff (authoritative; no extension trust).
|
||||
|
||||
Returns ``None`` for anything without a recognized header — including SVG,
|
||||
which has no magic bytes (the resolver sniffs ``<svg`` and passes it
|
||||
through for rasterization).
|
||||
"""
|
||||
header = data[:64]
|
||||
if header.startswith(b"\x89PNG\r\n\x1a\n"):
|
||||
# Magic bytes alone are insufficient: reject corrupt PNGs before they
|
||||
# can be embedded. Pillow is optional — without it fall back to
|
||||
# header-only sniffing; only an actual failed verify() rejects.
|
||||
try:
|
||||
from PIL import Image
|
||||
except ImportError:
|
||||
return "image/png"
|
||||
try:
|
||||
with Image.open(BytesIO(data)) as image:
|
||||
image.verify()
|
||||
except Exception:
|
||||
return None
|
||||
return "image/png"
|
||||
if header.startswith(b"\xff\xd8\xff"):
|
||||
return "image/jpeg"
|
||||
if header.startswith((b"GIF87a", b"GIF89a")):
|
||||
return "image/gif"
|
||||
if header.startswith(b"BM"):
|
||||
return "image/bmp"
|
||||
if len(header) >= 12 and header[:4] == b"RIFF" and header[8:12] == b"WEBP":
|
||||
return "image/webp"
|
||||
return None
|
||||
|
||||
|
||||
def _supported_media_types() -> frozenset:
|
||||
"""Formats the ACTIVE main model's server can decode.
|
||||
|
||||
The managed llama-server decodes with stb_image — no WebP — and an
|
||||
undecodable image part fails SILENTLY (the model confabulates), so the set
|
||||
is narrowed there and normalization converts those formats to PNG.
|
||||
"""
|
||||
try:
|
||||
from agent.auxiliary_client import _runtime_main_value
|
||||
from hermes_cli.local_runtime.capabilities import (
|
||||
ACCEPTED_IMAGE_MIMES,
|
||||
is_managed_provider,
|
||||
)
|
||||
|
||||
if is_managed_provider(
|
||||
str(_runtime_main_value("provider") or ""),
|
||||
str(_runtime_main_value("base_url") or "")):
|
||||
return ACCEPTED_IMAGE_MIMES
|
||||
except Exception: # noqa: BLE001 — best-effort narrowing only
|
||||
pass
|
||||
return _ANTHROPIC_SUPPORTED_MEDIA_TYPES
|
||||
|
||||
|
||||
def _rasterize_svg_to_png(svg_path: Path, out_path: Path) -> bool:
|
||||
"""Best-effort SVG → PNG via cairosvg, svglib+reportlab, rsvg-convert, inkscape (all soft deps)."""
|
||||
def _ok() -> bool:
|
||||
return out_path.exists() and out_path.stat().st_size > 0
|
||||
|
||||
try:
|
||||
import cairosvg # type: ignore
|
||||
cairosvg.svg2png(url=str(svg_path), write_to=str(out_path))
|
||||
return _ok()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from svglib.svglib import svg2rlg # type: ignore
|
||||
from reportlab.graphics import renderPM # type: ignore
|
||||
drawing = svg2rlg(str(svg_path))
|
||||
if drawing is not None:
|
||||
renderPM.drawToFile(drawing, str(out_path), fmt="PNG")
|
||||
return _ok()
|
||||
except Exception:
|
||||
pass
|
||||
import shutil
|
||||
import subprocess
|
||||
for cmd in (
|
||||
["rsvg-convert", "-o", str(out_path), str(svg_path)],
|
||||
["inkscape", str(svg_path), "--export-type=png",
|
||||
f"--export-filename={out_path}"],
|
||||
):
|
||||
if shutil.which(cmd[0]):
|
||||
try:
|
||||
subprocess.run(
|
||||
cmd, check=True, capture_output=True, timeout=30,
|
||||
stdin=subprocess.DEVNULL,
|
||||
)
|
||||
if _ok():
|
||||
return True
|
||||
except Exception:
|
||||
continue
|
||||
return False
|
||||
|
||||
|
||||
def _normalize_to_supported_image(
|
||||
image_path: Path, detected_mime: str
|
||||
) -> tuple[Optional[Path], Optional[str], Optional[str]]:
|
||||
"""Ensure an image is in a provider-supported format.
|
||||
|
||||
Returns ``(path, mime, error)``: the input unchanged when already
|
||||
supported; ``(new_png_path, "image/png", None)`` after conversion — a temp
|
||||
file the CALLER must clean up; ``(None, None, message)`` when impossible.
|
||||
SVG is rasterized; other Pillow-readable rasters (BMP, TIFF) re-encode to PNG.
|
||||
"""
|
||||
if detected_mime in _supported_media_types():
|
||||
return image_path, detected_mime, None
|
||||
|
||||
out_dir = get_hermes_dir("cache/vision", "temp_vision_images")
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
out_path = out_dir / f"converted_{uuid.uuid4()}.png"
|
||||
|
||||
if detected_mime == "image/svg+xml":
|
||||
if _rasterize_svg_to_png(image_path, out_path):
|
||||
return out_path, "image/png", None
|
||||
return (
|
||||
None,
|
||||
None,
|
||||
"This is an SVG, which vision models cannot read directly, and no "
|
||||
"SVG rasterizer is installed (tried cairosvg, svglib, rsvg-convert, "
|
||||
"inkscape). Convert the SVG to PNG first — e.g. open it in a browser "
|
||||
"and screenshot it, or install a rasterizer "
|
||||
"(`pip install cairosvg`) — then re-run vision_analyze on the PNG.",
|
||||
)
|
||||
|
||||
try:
|
||||
from PIL import Image as _PILImage
|
||||
with _PILImage.open(image_path) as _img:
|
||||
if _img.mode not in ("RGB", "RGBA", "L"):
|
||||
_img = _img.convert("RGBA")
|
||||
_img.save(out_path, format="PNG")
|
||||
if out_path.exists() and out_path.stat().st_size > 0:
|
||||
return out_path, "image/png", None
|
||||
except Exception as _exc:
|
||||
logger.warning("Failed to normalize %s image to PNG: %s",
|
||||
detected_mime, _exc)
|
||||
return (
|
||||
None,
|
||||
None,
|
||||
f"Image format {detected_mime!r} is not supported by the vision API "
|
||||
f"and could not be converted to PNG (install Pillow for raster "
|
||||
f"conversion). Convert it to PNG or JPEG and try again.",
|
||||
)
|
||||
|
||||
|
||||
# Full raster validation runs on untrusted images in a shared CPU executor:
|
||||
# bound animated-image work by frame count AND total decoded area so a compact
|
||||
# file cannot monopolize a worker with unbounded frames.
|
||||
_VISION_MAX_VALIDATED_FRAME_COUNT = 100
|
||||
_VISION_MAX_VALIDATED_AGGREGATE_PIXELS = 100_000_000
|
||||
|
||||
|
||||
def _validate_raster_image_decodable(
|
||||
image_path: Path,
|
||||
max_frames: int = _VISION_MAX_VALIDATED_FRAME_COUNT,
|
||||
max_pixels: int = _VISION_MAX_VALIDATED_AGGREGATE_PIXELS,
|
||||
) -> Optional[str]:
|
||||
"""Return an error unless Pillow can fully decode every frame.
|
||||
|
||||
Header sniffing and ``Image.open`` only inspect containers: a timed-out
|
||||
download can look like a valid PNG with a truncated pixel stream. Without
|
||||
Pillow the image passes unvalidated rather than rejecting everything.
|
||||
"""
|
||||
try:
|
||||
from PIL import Image as _PILImage
|
||||
from PIL import ImageSequence as _PILImageSequence
|
||||
except ImportError:
|
||||
return None
|
||||
try:
|
||||
with _PILImage.open(image_path) as image:
|
||||
image.verify()
|
||||
with _PILImage.open(image_path) as image:
|
||||
validated_pixels = 0
|
||||
for frame_number, frame in enumerate(
|
||||
_PILImageSequence.Iterator(image), start=1
|
||||
):
|
||||
if frame_number > max_frames:
|
||||
return (
|
||||
"Image validation rejected animation: "
|
||||
f"frame {frame_number} exceeds the maximum "
|
||||
f"{max_frames} validated frames."
|
||||
)
|
||||
next_validated_pixels = validated_pixels + frame.width * frame.height
|
||||
if next_validated_pixels > max_pixels:
|
||||
return (
|
||||
"Image validation rejected animation: aggregate decoded "
|
||||
f"pixel count would reach {next_validated_pixels} at frame "
|
||||
f"{frame_number}, exceeding the maximum "
|
||||
f"{max_pixels}."
|
||||
)
|
||||
frame.load()
|
||||
validated_pixels = next_validated_pixels
|
||||
except Exception as exc:
|
||||
return f"Image could not be fully decoded: {exc}"
|
||||
return None
|
||||
|
||||
|
||||
def _image_exceeds_dimension(image_path: Path, max_dimension: int) -> bool:
|
||||
"""True if the longest side exceeds ``max_dimension`` px.
|
||||
|
||||
Anthropic enforces an 8000px per-side cap independently of the byte cap.
|
||||
Returns False (no forced resize) without Pillow or on unreadable files —
|
||||
a missing soft dependency must never break the embed path.
|
||||
"""
|
||||
try:
|
||||
from PIL import Image as _PILImage
|
||||
with _PILImage.open(image_path) as _img:
|
||||
return max(_img.size) > max_dimension
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _crop_image_region(
|
||||
image_path: Path,
|
||||
region: Any,
|
||||
offset_out: Optional[dict] = None,
|
||||
) -> tuple[Optional[Path], Optional[str], Optional[str]]:
|
||||
"""Crop to ``region`` = [x1, y1, x2, y2] (original-image pixels).
|
||||
|
||||
Applied BEFORE downscaling so the crop gets the full resolution budget.
|
||||
Coordinates clamp to the image bounds; a zero-area/inverted region is
|
||||
rejected with an error naming the real dimensions. Returns
|
||||
``(cropped_temp_path, mime, None)`` — caller owns cleanup — or
|
||||
``(None, None, error)``. Ported from QwenLM/qwen-code zoom-image.ts (Apache-2.0).
|
||||
"""
|
||||
try:
|
||||
from PIL import Image
|
||||
except ImportError:
|
||||
return None, None, (
|
||||
"region cropping requires Pillow (`pip install Pillow`); "
|
||||
"retry without the region parameter."
|
||||
)
|
||||
|
||||
if (
|
||||
not isinstance(region, (list, tuple))
|
||||
or len(region) != 4
|
||||
or not all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in region)
|
||||
):
|
||||
return None, None, (
|
||||
"Invalid region: expected [x1, y1, x2, y2] as four numbers "
|
||||
"(pixel coordinates in the original image)."
|
||||
)
|
||||
|
||||
try:
|
||||
with Image.open(image_path) as img:
|
||||
width, height = img.size
|
||||
x1, y1, x2, y2 = (int(v) for v in region)
|
||||
cx1 = max(0, min(x1, width))
|
||||
cy1 = max(0, min(y1, height))
|
||||
cx2 = max(0, min(x2, width))
|
||||
cy2 = max(0, min(y2, height))
|
||||
if cx2 <= cx1 or cy2 <= cy1:
|
||||
return None, None, (
|
||||
f"Invalid region [{x1}, {y1}, {x2}, {y2}]: crops to zero "
|
||||
f"area after clamping to the image bounds. The image is "
|
||||
f"{width}x{height} px — pick x1<x2 and y1<y2 inside "
|
||||
f"[0, 0, {width}, {height}]."
|
||||
)
|
||||
cropped = img.crop((cx1, cy1, cx2, cy2))
|
||||
if offset_out is not None:
|
||||
offset_out.update(x=cx1, y=cy1, width=cx2 - cx1, height=cy2 - cy1)
|
||||
out_path = image_path.with_name(
|
||||
f"{image_path.stem}_region_{uuid.uuid4().hex[:8]}.png"
|
||||
)
|
||||
if cropped.mode not in ("RGB", "RGBA", "L", "LA", "P"):
|
||||
cropped = cropped.convert("RGB")
|
||||
cropped.save(out_path, format="PNG")
|
||||
return out_path, "image/png", None
|
||||
except Exception as exc:
|
||||
return None, None, f"Failed to crop region: {exc}"
|
||||
Reference in New Issue
Block a user