#!/usr/bin/env python3 """Vision tools: ``vision_analyze`` (image) and ``video_analyze``. Images resolve through :mod:`tools.image_source`, are normalized to a provider-supported format (:mod:`tools.vision_tools_image_prep`), then either attach natively to a vision-capable main model (multimodal tool-result envelope) or are described by the auxiliary vision LLM router. """ import base64 import asyncio import json from concurrent.futures import ThreadPoolExecutor from io import BytesIO import logging import os import uuid from pathlib import Path from typing import Any, Awaitable, Dict, Optional from urllib.parse import urlparse import httpx # ``agent.auxiliary_client`` costs ~50 ms cold (credential_pool → auth → rich); # only the handlers need it. Both names stay module attributes so tests can # patch ``tools.vision_tools.async_call_llm``; truthy-skip means injected mocks win. async_call_llm: Any = None extract_content_or_reasoning: Any = None def _load_auxiliary_client() -> None: global async_call_llm, extract_content_or_reasoning if async_call_llm is None or extract_content_or_reasoning is None: from agent.auxiliary_client import ( async_call_llm as _acl, extract_content_or_reasoning as _ecr, ) if async_call_llm is None: async_call_llm = _acl if extract_content_or_reasoning is None: extract_content_or_reasoning = _ecr from hermes_constants import get_hermes_dir from tools.debug_helpers import DebugSession from tools.website_policy import check_website_access from tools.vision_tools_image_prep import ( # noqa: F401 — re-exported for tests/image_source _ANTHROPIC_SUPPORTED_MEDIA_TYPES, _VISION_MAX_VALIDATED_AGGREGATE_PIXELS, _VISION_MAX_VALIDATED_FRAME_COUNT, _crop_image_region, _detect_image_mime_type_from_bytes, _determine_mime_type, _image_exceeds_dimension, _normalize_to_supported_image, _rasterize_svg_to_png, _supported_media_types, _validate_raster_image_decodable, ) logger = logging.getLogger(__name__) _debug = DebugSession("vision_tools", env_var="VISION_TOOLS_DEBUG") def _read_vision_setting(env_var: str, key: str, cast, minimum=None): """Env var → config.yaml ``auxiliary.vision.`` → None. Values that fail ``cast`` or fall below ``minimum`` are skipped in favor of the next source (a cap can never be disabled by a bad value). """ def _accept(raw): try: val = cast(raw) except (TypeError, ValueError): return None return val if minimum is None or val >= minimum else None env_val = os.getenv(env_var, "").strip() if env_val: val = _accept(env_val) if val is not None: return val try: from hermes_cli.config import cfg_get, load_config raw = cfg_get(load_config(), "auxiliary", "vision", key) if raw is not None: return _accept(raw) except Exception: pass return None def _resolve_download_timeout() -> float: """HTTP download timeout (separate from ``auxiliary.vision.timeout``, which governs the LLM call).""" val = _read_vision_setting("HERMES_VISION_DOWNLOAD_TIMEOUT", "download_timeout", float) return 30.0 if val is None else val _VISION_DOWNLOAD_TIMEOUT = _resolve_download_timeout() # Hard cap on downloaded media (50 MB): bounds memory/disk against # attacker-hosted multi-gigabyte files. _VISION_MAX_DOWNLOAD_BYTES = 50 * 1024 * 1024 # --------------------------------------------------------------------------- # CPU-burst concurrency cap (vision encode/resize) # --------------------------------------------------------------------------- # A turn can fan out dozens of vision_analyze calls ("analyze every frame"); # each does a CPU-heavy base64 encode + Pillow resize. Sessions share one # process, so unbounded encodes saturate every core and starve the shared event # loop (the dashboard liveness probe flapped UNHEALTHY in prod). We cap ONLY the # CPU burst — the LLM calls stay fully concurrent — on a dedicated executor sized # to the usable core count (the resource actually exhausted; no fixed ceiling). # It must be a threading primitive: each call runs via model_tools._run_async on # a PER-THREAD event loop, so an asyncio semaphore cannot coordinate across them. # The default executor is NOT used: it is shared with the gateway/web server. import threading # noqa: F401 (kept for downstream importers / patch targets) def _detect_host_cpus() -> int: """Usable CPU count (``sched_getaffinity`` honors cpuset pinning), at least 1.""" try: return max(1, len(os.sched_getaffinity(0))) # type: ignore[attr-defined] except (AttributeError, OSError): return max(1, os.cpu_count() or 1) def _resolve_vision_cpu_workers() -> int: """HERMES_VISION_MAX_CONCURRENCY → ``auxiliary.vision.max_concurrency`` → host cores (values < 1 ignored).""" val = _read_vision_setting("HERMES_VISION_MAX_CONCURRENCY", "max_concurrency", int, minimum=1) return _detect_host_cpus() if val is None else val _VISION_CPU_WORKERS = _resolve_vision_cpu_workers() _vision_cpu_executor = ThreadPoolExecutor( max_workers=_VISION_CPU_WORKERS, thread_name_prefix="vision-encode", ) async def _run_encode_on_cpu_executor(fn, *args, **kwargs): """Run a sync encode/resize callable on the bounded vision CPU executor (never the LLM call).""" import functools loop = asyncio.get_running_loop() return await loop.run_in_executor( _vision_cpu_executor, functools.partial(fn, *args, **kwargs) ) def _image_url_shape_ok(url: str) -> bool: """HTTP(S) shape check only (scheme + netloc; no DNS). Extension-less CDN URLs pass.""" if not url or not isinstance(url, str) or not url.startswith(("http://", "https://")): return False return bool(urlparse(url).netloc) async def _validate_image_url_async(url: str) -> bool: """Validate remote image URL (SSRF guard) without blocking the event loop on DNS.""" if not _image_url_shape_ok(url): return False from tools.url_safety import async_is_safe_url return await async_is_safe_url(url) def _is_retryable_download_error(error: Exception) -> bool: """True only for transient download failures worth retrying. Fail-fast: 4xx other than 429 (missing/forbidden), PermissionError (policy or SSRF block), ValueError (too large / blocked redirect — deterministic). Retryable: 429, 5xx, transport errors and anything unclassified. """ if isinstance(error, (PermissionError, ValueError)): return False if isinstance(error, httpx.HTTPStatusError): status = error.response.status_code return not (400 <= status < 500 and status != 429) return True _DOWNLOAD_USER_AGENT = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" ) async def _stream_download_to_file( client, url: str, destination: Path, max_bytes: int, *, headers: dict, media_label: str = "Image", ) -> Path: """Stream a GET to *destination* via a temp file with a running size cap. The body is never fully buffered: chunks go to a temp file and the running count is checked after each one (Content-Length gives an early reject but servers can omit or lie, so the streaming cap is authoritative). The temp file is atomically moved onto *destination* on success, deleted on failure. """ from utils import atomic_replace async with client.stream("GET", url, headers=headers) as response: response.raise_for_status() cl = response.headers.get("content-length") if cl: try: declared_size = int(cl) except ValueError: declared_size = None if declared_size is not None and declared_size > max_bytes: raise ValueError( f"{media_label} too large ({declared_size} bytes, max {max_bytes})" ) blocked = check_website_access(str(response.url)) if blocked: raise PermissionError(blocked["message"]) tmp_destination = destination.with_name( f".{destination.name}.{uuid.uuid4().hex}.tmp" ) bytes_written = 0 try: with tmp_destination.open("wb") as f: async for chunk in response.aiter_bytes(): if not chunk: continue bytes_written += len(chunk) if bytes_written > max_bytes: raise ValueError( f"{media_label} too large ({bytes_written} bytes, max {max_bytes})" ) f.write(chunk) atomic_replace(tmp_destination, destination) except Exception: try: tmp_destination.unlink(missing_ok=True) except OSError: logger.debug( "Could not delete partial download: %s", tmp_destination, exc_info=True ) raise return destination async def _ssrf_redirect_guard(response): """Re-validate each redirect target: a public URL that 302s to http://169.254.169.254/ would otherwise bypass the pre-flight is_safe_url check. Async because httpx.AsyncClient awaits event hooks.""" from tools.url_safety import async_is_safe_url, redirect_target_from_response redirect_url = redirect_target_from_response(response) if redirect_url and not await async_is_safe_url(redirect_url): raise ValueError( f"Blocked redirect to private/internal address: {redirect_url}" ) async def _download_media( url: str, destination: Path, max_retries: int, *, media_label: str, accept: str, max_bytes: int, timeout: float, retry_all: bool, ) -> Path: """Shared SSRF-safe streaming download with exponential backoff (2s/4s/8s). ``retry_all=False`` (images) only retries transient errors per :func:`_is_retryable_download_error` — a 404/403 never succeeds on retry, so burning three backoff rounds just inflates latency. ``retry_all=True`` (video) keeps the legacy retry-everything behavior. """ destination.parent.mkdir(parents=True, exist_ok=True) last_error = None for attempt in range(max_retries): try: blocked = check_website_access(url) if blocked: raise PermissionError(blocked["message"]) from tools.url_safety import create_ssrf_safe_async_client # follow_redirects for CDNs; the client validates DNS at connect # time and the hook re-validates each redirect target. async with create_ssrf_safe_async_client( timeout=timeout, follow_redirects=True, event_hooks={"response": [_ssrf_redirect_guard]}, ) as client: await _stream_download_to_file( client, url, destination, max_bytes, headers={"User-Agent": _DOWNLOAD_USER_AGENT, "Accept": accept}, media_label=media_label, ) return destination except Exception as e: last_error = e final = attempt >= max_retries - 1 if final or (not retry_all and not _is_retryable_download_error(e)): logger.error( "%s download failed after %s attempt(s): %s", media_label, attempt + 1, str(e)[:100], exc_info=True, ) if not retry_all: raise break wait_time = 2 ** (attempt + 1) logger.warning("%s download failed (attempt %s/%s): %s", media_label, attempt + 1, max_retries, str(e)[:50]) if not retry_all: logger.warning("Retrying in %ss...", wait_time) await asyncio.sleep(wait_time) # Reaching here means max_retries was non-positive (or video exhausted retries). if last_error is not None: raise last_error raise RuntimeError( f"_download_{media_label.lower()} exited retry loop without attempting (max_retries={max_retries})" ) async def _download_image(image_url: str, destination: Path, max_retries: int = 3) -> Path: """Download an image with SSRF protection and error-class-aware retry.""" return await _download_media( image_url, destination, max_retries, media_label="Image", accept="image/*,*/*;q=0.8", max_bytes=_VISION_MAX_DOWNLOAD_BYTES, timeout=_VISION_DOWNLOAD_TIMEOUT, retry_all=False, ) def _image_to_base64_data_url(image_path: Path, mime_type: Optional[str] = None) -> str: """``data:;base64,...`` for a file (MIME from extension when not given).""" encoded = base64.b64encode(image_path.read_bytes()).decode("ascii") return f"data:{mime_type or _determine_mime_type(image_path)};base64,{encoded}" # Absolute hard ceiling for vision payloads (20 MB): no major provider accepts more. _MAX_BASE64_BYTES = 20 * 1024 * 1024 # Proactive embed caps for conversation-history reuse. Native vision_analyze # bakes the data URL into the tool result, re-sent every later turn; a 4 MB / # 7900px embed cost ~100-260K billed tokens per image. 256 KB keeps a 1568px # screenshot cheap enough to ride the session; Anthropic's tokenizer downsamples # to a 1568px long edge anyway, so pixels past that cost wire bytes for no # fidelity. The 20 MB / provider 5 MB caps remain as one-shot safety nets. _EMBED_TARGET_BYTES = 256 * 1024 _EMBED_MAX_DIMENSION = 1568 # Target when auto-resizing after a provider size rejection (retry once). _RESIZE_TARGET_BYTES = 5 * 1024 * 1024 _SIZE_ERROR_HINTS = ( "too large", "payload", "413", "content_too_large", "request_too_large", "exceeds", "size limit", ) def _is_image_size_error(error: Exception) -> bool: """Detect if an API error is related to image or payload size.""" err_str = str(error).lower() return any(hint in err_str for hint in _SIZE_ERROR_HINTS + ("image_url", "invalid_request")) def _build_scale_note( scale_info: Optional[dict], crop_offset: Optional[dict], ) -> Optional[str]: """Coordinate-mapping disclosure for downscale (``scale_info``) and/or region crop (``crop_offset``); ``None`` when neither applied — no note, no noise.""" parts = [] if scale_info: ow, oh = scale_info["orig_width"], scale_info["orig_height"] nw, nh = scale_info["new_width"], scale_info["new_height"] fx = ow / nw if nw else 1.0 fy = oh / nh if nh else 1.0 if f"{fx:.2f}" == f"{fy:.2f}": factor_clause = ( f"multiply any coordinates you report by {fx:.2f} " f"to map back to the original image." ) else: factor_clause = ( f"multiply any x coordinates you report by {fx:.2f} and " f"any y coordinates by {fy:.2f} to map back to the " f"original image." ) parts.append( f"Image downscaled from {ow}x{oh} to {nw}x{nh} for vision; " f"{factor_clause}" ) if crop_offset: parts.append( f"Analysis was performed on a cropped region of the original " f"image starting at offset ({crop_offset['x']}, " f"{crop_offset['y']}); coordinates are relative to that crop " f"origin — add the offset to map back to the full image." ) return " ".join(parts) if parts else None def _import_pillow_for_resize(): """Pillow is a lazy-installable soft dependency; return ``PIL.Image`` or None. ``prompt=False``: a blocking input() deadlocks the interactive CLI where prompt_toolkit owns stdin. The install is gated by security.allow_lazy_installs, so reaching it is already opt-in. """ try: from PIL import Image return Image except ImportError: pass try: from tools.lazy_deps import ensure as _ensure_dep _ensure_dep("tool.vision", prompt=False) from PIL import Image return Image except Exception: return None def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None, max_base64_bytes: int = _RESIZE_TARGET_BYTES, max_dimension: Optional[int] = None, scale_out: Optional[dict] = None, force_jpeg: bool = False) -> str: """Base64 data URL, progressively downscaled with Pillow while over budget. Halves dimensions (aspect-preserving, 64px floor) up to 4 times; JPEG also walks a quality ladder (85/70/50) at each step. Without Pillow, or if it still doesn't fit, returns the best attempt (or raw bytes) and lets the caller apply the size check. ``max_dimension``: force a downscale above this long edge even when bytes fit (Anthropic's 8000px cap is independent of bytes). ``force_jpeg``: re-encode PNG input as JPEG when a resize is needed — PNG's only shrink lever is halving dimensions, which destroys text legibility on dense screenshots; history-reuse embeds opt in. Images under both caps return unchanged. """ file_size = image_path.stat().st_size estimated_b64 = (file_size * 4) // 3 + 100 # base64 ~4/3 + data URL header needs_resize_for_bytes = estimated_b64 > max_base64_bytes needs_resize_for_dims = ( max_dimension is not None and _image_exceeds_dimension(image_path, max_dimension) ) data_url = None if not needs_resize_for_bytes and not needs_resize_for_dims: data_url = _image_to_base64_data_url(image_path, mime_type=mime_type) if len(data_url) <= max_base64_bytes: return data_url def _raw() -> str: return data_url or _image_to_base64_data_url(image_path, mime_type=mime_type) Image = _import_pillow_for_resize() if Image is None: logger.info("Pillow not installed — cannot auto-resize oversized image") return _raw() # caller will raise the size error logger.info("Image file is %.1f MB (estimated base64 %.1f MB, limit %.1f MB, max_dimension=%s), auto-resizing...", file_size / (1024 * 1024), estimated_b64 / (1024 * 1024), max_base64_bytes / (1024 * 1024), max_dimension) mime = mime_type or _determine_mime_type(image_path) # JPEG for photos (smaller), PNG for transparency — unless force_jpeg. pil_format = "PNG" if (mime == "image/png" and not force_jpeg) else "JPEG" out_mime = "image/png" if pil_format == "PNG" else "image/jpeg" try: img = Image.open(image_path) except Exception as exc: logger.info("Pillow cannot open image for resizing: %s", exc) return _raw() # JPEG cannot encode alpha/palette modes (force_jpeg routes PNGs here). if pil_format == "JPEG" and img.mode not in {"RGB", "L"}: img = img.convert("RGB") quality_steps = (85, 70, 50) if pil_format == "JPEG" else (None,) orig_dims = prev_dims = (img.width, img.height) candidate = None def _record_scale(w: int, h: int) -> None: if scale_out is not None and (w, h) != orig_dims: scale_out.update(orig_width=orig_dims[0], orig_height=orig_dims[1], new_width=w, new_height=h) for attempt in range(5): if attempt > 0: # Halve, then re-derive the scale from whichever axis hit the 64px # floor so both axes shrink by the same factor. new_w = max(int(img.width * 0.5), 64) new_h = max(int(img.height * 0.5), 64) if new_w == 64 and img.width > 0: new_h = max(int(img.height * (64 / img.width)), 64) elif new_h == 64 and img.height > 0: new_w = max(int(img.width * (64 / img.height)), 64) if (new_w, new_h) == prev_dims: break img = img.resize((new_w, new_h), Image.LANCZOS) prev_dims = (new_w, new_h) logger.info("Resized to %dx%d (attempt %d)", new_w, new_h, attempt) for q in quality_steps: buf = BytesIO() img.save(buf, format=pil_format, **({} if q is None else {"quality": q})) candidate = f"data:{out_mime};base64,{base64.b64encode(buf.getvalue()).decode('ascii')}" dims_ok = max_dimension is None or max(img.width, img.height) <= max_dimension if len(candidate) <= max_base64_bytes and dims_ok: logger.info("Auto-resized image fits: %.1f MB (quality=%s, %dx%d)", len(candidate) / (1024 * 1024), q, img.width, img.height) _record_scale(img.width, img.height) return candidate if candidate is not None: logger.warning("Auto-resize could not fit image under %.1f MB (best: %.1f MB)", max_base64_bytes / (1024 * 1024), len(candidate) / (1024 * 1024)) _record_scale(img.width, img.height) return candidate return _raw() # --------------------------------------------------------------------------- # Native fast path: when the active main model supports vision, skip the aux # LLM and return the image bytes as a multimodal tool-result envelope. The # agent loop unwraps it into an OpenAI-style content list on the `tool` role; # provider adapters translate that per backend, so the main model "sees" the # pixels directly on its next turn. # --------------------------------------------------------------------------- # Providers whose tool results accept image content (spec docs verified # Apr-2026): Anthropic Messages (+ aggregators proxying Claude — assume support, # falling back to text would regress their frontier models), OpenAI Chat # Completions / Responses. Gemini is gated on model: only 3.x supports # multimodal functionResponse. _TOOL_RESULT_MEDIA_PROVIDERS = frozenset({ "openrouter", "nous", "vertex", "bedrock", "anthropic-vertex", "google-vertex", "anthropic", "claude", "anthropic-direct", "openai", "openai-chat", "openai-codex", "azure-openai", }) _GEMINI_PROVIDERS = frozenset({"google", "gemini", "google-gemini", "google-vertex-gemini"}) def _supports_media_in_tool_results(provider: str, model: str) -> bool: """Whether provider+model accepts image content inside a tool-result message. Unknown providers are conservatively False (caller falls back to the aux-LLM text path) unless their ``ProviderProfile`` declares ``supports_vision``. """ if not isinstance(provider, str): return False p = provider.strip().lower() if not p: return False if p in _TOOL_RESULT_MEDIA_PROVIDERS: return True if p in _GEMINI_PROVIDERS: if not isinstance(model, str): return False m = model.strip().lower() return any(tag in m for tag in ("gemini-3", "gemini-pro-3", "gemini-flash-3")) try: from providers import get_provider_profile profile = get_provider_profile(p) if profile is not None and profile.supports_vision: return True except Exception: pass return False def _should_use_native_vision_fast_path() -> bool: """True when image routing resolves to ``native`` AND the provider accepts images in tool results, or the user set the ``model.supports_vision`` override (escape hatch for custom/local providers). Any failure → False.""" try: from agent.auxiliary_client import _read_main_provider, _read_main_model from agent.image_routing import decide_image_input_mode, _lookup_supports_vision from hermes_cli.config import load_config provider = _read_main_provider() model = _read_main_model() cfg = load_config() if decide_image_input_mode(provider, model, cfg) != "native": return False return ( _supports_media_in_tool_results(provider, model) or _lookup_supports_vision(provider, model, cfg) is True ) except Exception as exc: logger.debug("Native vision fast-path check failed: %s", exc) return False def _build_native_vision_tool_result( image_url: str, question: str, image_data_url: str, image_size_bytes: int, scale_note: Optional[str] = None, ) -> Dict[str, Any]: """Multimodal tool-result envelope (``_multimodal`` + ``content`` list). The text part is intentionally minimal — the model already has the question in context; it acknowledges the image is visible. ``text_summary`` is the fallback for providers without multimodal tool results. """ text_part = ( "Image loaded into your context — you can see it natively now. " "Use your built-in vision to answer the user." ) if isinstance(question, str) and question.strip(): text_part += f"\n\nQuestion: {question.strip()}" if scale_note: text_part += f"\n\nNote: {scale_note}" summary = ( f"Image attached natively for the main model " f"({image_size_bytes / 1024:.1f} KB). " "Answer using built-in vision." ) return { "_multimodal": True, "content": [ {"type": "text", "text": text_part}, {"type": "image_url", "image_url": {"url": image_data_url}}, ], "text_summary": summary, "meta": { "image_url": image_url[:200], "size_bytes": image_size_bytes, "native_vision": True, }, } def _unlink_quietly(path: Optional[Path]) -> None: if path is not None: try: if path.exists(): path.unlink() except Exception: pass class _ImagePrepError(ValueError): """Raised by :func:`_prepare_image`; the message is user-facing.""" class _PreparedImage: """Temp image ready to encode; ``path`` is owned by the caller (delete it).""" __slots__ = ("path", "mime", "size_bytes", "crop_offset") def __init__(self, path: Path, mime: Optional[str], size_bytes: int, crop_offset: dict): self.path, self.mime, self.size_bytes, self.crop_offset = path, mime, size_bytes, crop_offset async def _prepare_image( image_url: str, task_id: Optional[str], region: Optional[list], *, validate_decode: bool, ) -> _PreparedImage: """Resolve → materialize → normalize → (validate) → (crop). Raises ``_ImagePrepError``. The single resolver unifies data:/http/file/local/container sources and enforces terminal-backend confinement; bytes land in a temp file so the path-based encode/resize pipeline is reused. Unsupported formats (SVG, BMP) are converted to PNG BEFORE encoding — an unsupported media_type baked into immutable history would 400 on every resume. The crop runs BEFORE any downscale so the region keeps the full resolution budget. Blocking rasterizer/Pillow work is offloaded. Intermediate temp files are deleted; on error nothing is left behind. """ from tools.image_source import ImageResolutionError, ResolveContext, resolve_image_source try: resolved = await resolve_image_source(image_url, ResolveContext(task_id=task_id)) except ImageResolutionError as exc: raise _ImagePrepError(str(exc)) from exc temp_dir = get_hermes_dir("cache/vision", "temp_vision_images") temp_dir.mkdir(parents=True, exist_ok=True) path = temp_dir / f"temp_image_{uuid.uuid4()}.img" await asyncio.to_thread(path.write_bytes, resolved.data) mime = resolved.mime size_bytes = len(resolved.data) crop_offset: dict = {} try: normalized_path, mime, norm_err = await asyncio.to_thread( _normalize_to_supported_image, path, mime, ) if norm_err or normalized_path is None: raise _ImagePrepError(norm_err or "Image normalization failed.") if normalized_path != path: _unlink_quietly(path) path = normalized_path size_bytes = path.stat().st_size if validate_decode: decode_error = await _run_encode_on_cpu_executor( _validate_raster_image_decodable, path, _VISION_MAX_VALIDATED_FRAME_COUNT, _VISION_MAX_VALIDATED_AGGREGATE_PIXELS, ) if decode_error: raise _ImagePrepError(decode_error) if region is not None: cropped_path, cropped_mime, crop_err = await asyncio.to_thread( _crop_image_region, path, region, offset_out=crop_offset, ) if crop_err or cropped_path is None: raise _ImagePrepError(crop_err or "Region crop failed.") _unlink_quietly(path) path = cropped_path mime = cropped_mime size_bytes = path.stat().st_size except BaseException: _unlink_quietly(path) raise return _PreparedImage(path, mime, size_bytes, crop_offset) def _too_large_message(image_data_url: str) -> str: return ( f"Image too large for vision API: base64 payload is " f"{len(image_data_url) / (1024 * 1024):.1f} MB " f"(limit {_MAX_BASE64_BYTES / (1024 * 1024):.0f} MB) " f"even after resizing. Install Pillow " f"(`pip install Pillow`) for better auto-resize, " f"or compress the image manually." ) async def _vision_analyze_native( image_url: str, question: str, task_id: Optional[str] = None, region: Optional[list] = None, ) -> Any: """Fast path for vision-capable main models. Returns a ``_multimodal`` envelope dict on success, or a JSON error string (the normal tool-result contract) on failure. """ if not isinstance(image_url, str) or not image_url.strip(): return tool_error("image_url is required", success=False) prepared: Optional[_PreparedImage] = None try: from tools.interrupt import is_interrupted if is_interrupted(): return tool_error("Interrupted", success=False) try: prepared = await _prepare_image(image_url, task_id, region, validate_decode=True) except _ImagePrepError as exc: return tool_error(str(exc), success=False) image_data_url = await _run_encode_on_cpu_executor( _image_to_base64_data_url, prepared.path, mime_type=prepared.mime, ) # Proactive embed cap: this image is re-sent on every later turn, so # resize DOWN to the history-reuse target whenever the byte or long-edge # cap is exceeded, not just at the 20 MB hard ceiling. _scale_info: dict = {} _over_dims = await _run_encode_on_cpu_executor( _image_exceeds_dimension, prepared.path, _EMBED_MAX_DIMENSION, ) if len(image_data_url) > _EMBED_TARGET_BYTES or _over_dims: image_data_url = await _run_encode_on_cpu_executor( _resize_image_for_vision, prepared.path, mime_type=prepared.mime, max_base64_bytes=_EMBED_TARGET_BYTES, max_dimension=_EMBED_MAX_DIMENSION, scale_out=_scale_info, force_jpeg=True, ) # Reject rather than embed a session-wedging payload. if len(image_data_url) > _MAX_BASE64_BYTES: return tool_error(_too_large_message(image_data_url), success=False) return _build_native_vision_tool_result( image_url=image_url, question=question, image_data_url=image_data_url, image_size_bytes=prepared.size_bytes, scale_note=_build_scale_note( _scale_info or None, prepared.crop_offset or None, ), ) except Exception as exc: logger.warning("Native vision fast path failed: %s", exc) return tool_error(f"Native vision failed: {exc}", success=False) finally: # Only delete temp files we created — never user-provided paths. if prepared is not None: _unlink_quietly(prepared.path) def _read_vision_call_settings(default_timeout: float, *, min_timeout: Optional[float] = None): """``auxiliary.vision.timeout`` / ``.temperature`` from config.yaml (defaults 120s-ish / 0.1). Local vision models (llama.cpp, ollama) can take well over 30s, hence the generous defaults; ``min_timeout`` lets video enforce a floor. """ timeout, temperature = default_timeout, 0.1 try: from hermes_cli.config import cfg_get, load_config _vision_cfg = cfg_get(load_config(), "auxiliary", "vision", default={}) _vt = _vision_cfg.get("timeout") if _vt is not None: timeout = float(_vt) if min_timeout is None else max(float(_vt), min_timeout) _vtemp = _vision_cfg.get("temperature") if _vtemp is not None: temperature = float(_vtemp) except Exception: pass return timeout, temperature # Error-message classification for the aux-LLM paths: first matching hint set # wins (order matters — billing before capability before size/format). _BILLING_HINTS = ("402", "insufficient", "payment required", "credits", "billing") _IMAGE_ERROR_RULES = ( (_BILLING_HINTS, "Insufficient credits or payment required. Please top up your " "API provider account and try again. Error: {e}"), (("does not support", "not support image", "content_policy", "multimodal", "unrecognized request argument", "image input"), "{model} does not support vision or our request was not " "accepted by the server. Error: {e}"), (("invalid_request", "image_url"), "The vision API rejected the image. This can happen when the " "image is in an unsupported format, corrupted, or still too " "large after auto-resize. Try a smaller JPEG/PNG and retry. " "Error: {e}"), ) _VIDEO_ERROR_RULES = ( (_BILLING_HINTS, _IMAGE_ERROR_RULES[0][1]), (("does not support", "not support video", "content_policy", "multimodal", "unrecognized request argument", "video input", "video_url"), "The model does not support video analysis or the request was " "rejected. Ensure you're using a video-capable model " "(e.g. google/gemini-2.5-flash). Error: {e}"), (_SIZE_ERROR_HINTS, "The video is too large for the API. Try compressing or trimming " "the video (max ~50 MB). Error: {e}"), ) def _classify_analysis_error(e: Exception, rules, fallback: str, **fmt) -> str: err_str = str(e).lower() for hints, template in rules: if any(hint in err_str for hint in hints): return template.format(e=e, **fmt) return fallback.format(e=e) def _debug_call_data(kind: str, source: str, user_prompt: str, model) -> dict: return { "parameters": { f"{kind}_url": source, "user_prompt": user_prompt[:200] + "..." if len(user_prompt) > 200 else user_prompt, "model": model, }, "error": None, "success": False, "analysis_length": 0, "model_used": model, f"{kind}_size_bytes": 0, } def _finish_analysis(tool_name: str, debug_call_data: dict, result: dict) -> str: _debug.log_call(tool_name, debug_call_data) _debug.save() return json.dumps(result, indent=2, ensure_ascii=False) def _cleanup_temp_media(path: Optional[Path], label: str) -> None: if path and path.exists(): try: path.unlink() logger.debug("Cleaned up temporary %s file", label) except Exception as cleanup_error: logger.warning( "Could not delete temporary file: %s", cleanup_error, exc_info=True ) async def _call_vision_llm(call_kwargs: dict, empty_log: str): """Call the aux vision LLM, retrying once on empty content (reasoning-only response).""" _load_auxiliary_client() response = await async_call_llm(**call_kwargs) analysis = extract_content_or_reasoning(response) if not analysis: logger.warning(empty_log) response = await async_call_llm(**call_kwargs) analysis = extract_content_or_reasoning(response) return analysis async def vision_analyze_tool( image_url: str, user_prompt: str, model: str = None, task_id: Optional[str] = None, region: Optional[list] = None, ) -> str: """Describe an image (URL, local path, data: URL) with the auxiliary vision LLM. ``user_prompt`` is pre-formatted by the caller. Returns JSON ``{"success": bool, "analysis": str}`` (``analysis`` carries the error explanation on failure). Temp images live under $HERMES_HOME/cache/vision/. """ if not isinstance(user_prompt, str): user_prompt = str(user_prompt) if user_prompt is not None else "" debug_call_data = _debug_call_data("image", image_url, user_prompt, model) prepared: Optional[_PreparedImage] = None try: from tools.interrupt import is_interrupted if is_interrupted(): return tool_error("Interrupted", success=False) logger.info("Analyzing image: %s", image_url[:60]) logger.info("User prompt: %s", user_prompt[:100]) prepared = await _prepare_image(image_url, task_id, region, validate_decode=False) logger.info("Image ready (%.1f KB)", prepared.size_bytes / 1024) # Send at full resolution first; on a size rejection, downscale and retry. logger.info("Converting image to base64...") image_data_url = await _run_encode_on_cpu_executor( _image_to_base64_data_url, prepared.path, mime_type=prepared.mime) logger.info("Image converted to base64 (%.1f KB)", len(image_data_url) / 1024) _scale_info: dict = {} if len(image_data_url) > _MAX_BASE64_BYTES: image_data_url = await _run_encode_on_cpu_executor( _resize_image_for_vision, prepared.path, mime_type=prepared.mime, scale_out=_scale_info) if len(image_data_url) > _MAX_BASE64_BYTES: raise ValueError(_too_large_message(image_data_url)) debug_call_data["image_size_bytes"] = prepared.size_bytes messages = [{ "role": "user", "content": [ {"type": "text", "text": user_prompt}, {"type": "image_url", "image_url": {"url": image_data_url}}, ], }] logger.info("Processing image with vision model...") vision_timeout, vision_temperature = _read_vision_call_settings(120.0) call_kwargs = { "task": "vision", "messages": messages, "temperature": vision_temperature, "timeout": vision_timeout, } if model: call_kwargs["model"] = model _load_auxiliary_client() try: response = await async_call_llm(**call_kwargs) except Exception as _api_err: if (_is_image_size_error(_api_err) and len(image_data_url) > _RESIZE_TARGET_BYTES): logger.info( "API rejected image (%.1f MB, likely too large); " "auto-resizing to ~%.0f MB and retrying...", len(image_data_url) / (1024 * 1024), _RESIZE_TARGET_BYTES / (1024 * 1024), ) image_data_url = await _run_encode_on_cpu_executor( _resize_image_for_vision, prepared.path, mime_type=prepared.mime, scale_out=_scale_info) messages[0]["content"][1]["image_url"]["url"] = image_data_url response = await async_call_llm(**call_kwargs) else: raise analysis = extract_content_or_reasoning(response) if not analysis: logger.warning("Vision LLM returned empty content, retrying once") response = await async_call_llm(**call_kwargs) analysis = extract_content_or_reasoning(response) analysis_length = len(analysis) logger.info("Image analysis completed (%s characters)", analysis_length) analysis = analysis or "There was a problem with the request and the image could not be analyzed." scale_note = _build_scale_note(_scale_info or None, prepared.crop_offset or None) result = { "success": True, "analysis": f"[{scale_note}] {analysis}" if scale_note else analysis, } if scale_note: result["scale_note"] = scale_note debug_call_data["success"] = True debug_call_data["analysis_length"] = analysis_length return _finish_analysis("vision_analyze_tool", debug_call_data, result) except Exception as e: error_msg = f"Error analyzing image: {str(e)}" logger.error("%s", error_msg, exc_info=True) analysis = _classify_analysis_error( e, _IMAGE_ERROR_RULES, "There was a problem with the request and the image could not " "be analyzed. Error: {e}", model=model, ) debug_call_data["error"] = error_msg return _finish_analysis("vision_analyze_tool", debug_call_data, { "success": False, "error": error_msg, "analysis": analysis, }) finally: if prepared is not None: _cleanup_temp_media(prepared.path, "image") def check_vision_requirements() -> bool: """True when ``call_llm(task="vision")`` could resolve a client. Mirrors its runtime fallback chain: explicit ``auxiliary.vision.provider``, then the auto chain (main provider → openrouter → nous) — without the auto step the tool would vanish whenever the explicit name was unresolvable. Probe mode skips real SDK client construction (openai import + SSL setup) on the tool-gating path; resolution policy is identical. """ try: from agent.auxiliary_client import aux_probe_mode, resolve_vision_provider_client except ImportError: return False try: with aux_probe_mode(): _provider, client, _model = resolve_vision_provider_client() if client is not None: return True _provider, client, _model = resolve_vision_provider_client(provider="auto") return client is not None except Exception: return False # --------------------------------------------------------------------------- # Registry # --------------------------------------------------------------------------- from tools.registry import registry, tool_error VISION_ANALYZE_SCHEMA = { "name": "vision_analyze", # Routing mechanics are deliberately absent (the route is automatic and the # native result says so itself); region keeps its pre-effect guidance — a # model that doesn't know crops keep full resolution never zooms. "description": ( "Load an image into the conversation so you can see it. Call it " "any time the user references an image — then answer from what " "you see." ), "parameters": { "type": "object", "properties": { "image_url": { "type": "string", "description": "Image URL (http/https), local file path, or data: URL to load." }, "question": { "type": "string", "description": "Your question or request about the image." }, "region": { "type": "array", "items": {"type": "integer"}, "minItems": 4, "maxItems": 4, "description": ( "Optional [x1, y1, x2, y2] crop in ORIGINAL-image pixel " "coordinates, applied before any downscaling — the crop " "keeps full resolution. Load the full image first, then " "re-call with a region to zoom into small text or fine " "detail." ) } }, "required": ["image_url", "question"] } } def _configured_aux_model(sections: tuple, env_vars: tuple) -> Optional[str]: """First non-empty ``auxiliary.
.model`` from config.yaml, else the first non-empty env var (legacy override), else None.""" try: from hermes_cli.config import cfg_get, load_config _cfg = load_config() for section in sections: _vmodel = cfg_get(_cfg, "auxiliary", section, "model") if _vmodel: model = str(_vmodel).strip() or None if model: return model break except Exception: pass for env_var in env_vars: val = os.getenv(env_var, "").strip() if val: return val return None async def _handle_vision_analyze(args: Dict[str, Any], **kw: Any) -> str: image_url = args.get("image_url", "") question = args.get("question", "") region = args.get("region") task_id = kw.get("task_id") # No concurrency gate around the whole analysis — the CPU burst is bounded # inside the encode/resize step, so multi-image fan-out keeps full request # concurrency. Native fast path: main model sees the pixels directly, no # aux call, no information loss. if _should_use_native_vision_fast_path(): logger.info("vision_analyze: native fast path") return await _vision_analyze_native(image_url, question, task_id=task_id, region=region) # Legacy path: aux LLM describes the image and we return its text. full_prompt = ( "Fully describe and explain everything about this image, then answer the " f"following question:\n\n{question}" ) model = _configured_aux_model(("vision",), ("AUXILIARY_VISION_MODEL",)) return await vision_analyze_tool(image_url, full_prompt, model, task_id=task_id, region=region) registry.register( name="vision_analyze", toolset="vision", schema=VISION_ANALYZE_SCHEMA, handler=_handle_vision_analyze, check_fn=check_vision_requirements, is_async=True, emoji="👁️", ) # --------------------------------------------------------------------------- # Video Analysis Tool # --------------------------------------------------------------------------- # Extension → MIME. avi/mkv fall back to mp4. _VIDEO_MIME_TYPES = { ".mp4": "video/mp4", ".webm": "video/webm", ".mov": "video/mov", ".avi": "video/mp4", ".mkv": "video/mp4", ".mpeg": "video/mpeg", ".mpg": "video/mpeg", } _MAX_VIDEO_BASE64_BYTES = 50 * 1024 * 1024 # 50 MB hard cap _VIDEO_SIZE_WARN_BYTES = 20 * 1024 * 1024 def _detect_video_mime_type(video_path: Path) -> Optional[str]: """Video MIME type from extension, or None if unsupported.""" return _VIDEO_MIME_TYPES.get(video_path.suffix.lower()) def _unsupported_video_format(suffix: str) -> str: return ( f"Unsupported video format: '{suffix}'. " f"Supported: {', '.join(sorted(_VIDEO_MIME_TYPES.keys()))}" ) def _video_to_base64_data_url(video_path: Path, mime_type: Optional[str] = None) -> str: encoded = base64.b64encode(video_path.read_bytes()).decode("ascii") mime = mime_type or _VIDEO_MIME_TYPES.get(video_path.suffix.lower(), "video/mp4") return f"data:{mime};base64,{encoded}" def _is_path_like_video_source(value: str) -> bool: lowered = (value or "").strip().lower() return bool(lowered) and not lowered.startswith(("http://", "https://", "data:")) async def _materialize_video_from_terminal_backend(video_source: str, task_id: Optional[str]) -> Path: """Read a path via the shared media resolver into a local temp video file. ``permitted=("video",)`` gives terminal-backend video reads the exact pipeline vision_analyze uses: media-cache host reads (gateway downloads live on the host, not in the sandbox), bounded in-sandbox exec-read, lazy env bring-up, the credential-read guard, and the 50MB ingest cap. """ from tools.image_source import ImageResolutionError, ResolveContext, resolve_image_source source = video_source if source.startswith("file://"): source = source[len("file://"):] suffix = Path(source).suffix.lower() if suffix not in _VIDEO_MIME_TYPES: raise ValueError(_unsupported_video_format(suffix)) try: resolved = await resolve_image_source( video_source, ResolveContext(task_id=task_id), permitted=("video",) ) except ImageResolutionError as exc: raise ValueError(f"Could not read video from terminal backend: {exc}") from exc temp_dir = get_hermes_dir("cache/video", "temp_video_files") temp_dir.mkdir(parents=True, exist_ok=True) temp_path = temp_dir / f"terminal_video_{uuid.uuid4()}{suffix}" temp_path.write_bytes(resolved.data) return temp_path async def _download_video(video_url: str, destination: Path, max_retries: int = 3) -> Path: """Download video with SSRF protection; every failure class is retried.""" return await _download_media( video_url, destination, max_retries, media_label="Video", accept="video/*,*/*;q=0.8", max_bytes=_MAX_VIDEO_BASE64_BYTES, timeout=60.0, retry_all=True, ) async def video_analyze_tool( video_url: str, user_prompt: str, model: str = None, task_id: Optional[str] = None, ) -> str: """Analyze a video via multimodal LLM. Returns JSON {success, analysis}.""" if not isinstance(user_prompt, str): user_prompt = str(user_prompt) if user_prompt is not None else "" debug_call_data = _debug_call_data("video", video_url, user_prompt, model) temp_video_path = None should_cleanup = True try: from tools.interrupt import is_interrupted if is_interrupted(): return tool_error("Interrupted", success=False) logger.info("Analyzing video: %s", video_url[:60]) logger.info("User prompt: %s", user_prompt[:100]) resolved_url = video_url if resolved_url.startswith("file://"): resolved_url = resolved_url[len("file://"):] local_path = Path(os.path.expanduser(resolved_url)) from tools.image_source import _is_local_terminal_backend if not _is_local_terminal_backend() and _is_path_like_video_source(video_url): logger.info("Reading video source via terminal backend: %s", video_url) temp_video_path = await _materialize_video_from_terminal_backend(video_url, task_id) elif local_path.is_file(): from agent.file_safety import raise_if_read_blocked raise_if_read_blocked(str(local_path)) logger.info("Using local video file: %s", video_url) temp_video_path = local_path should_cleanup = False elif await _validate_image_url_async(video_url): blocked = check_website_access(video_url) if blocked: raise PermissionError(blocked["message"]) temp_dir = get_hermes_dir("cache/video", "temp_video_files") temp_video_path = temp_dir / f"temp_video_{uuid.uuid4()}.mp4" await _download_video(video_url, temp_video_path) else: raise ValueError( "Invalid video source. Provide an HTTP/HTTPS URL or a valid local file path." ) video_size_bytes = temp_video_path.stat().st_size video_size_mb = video_size_bytes / (1024 * 1024) logger.info("Video ready (%.1f MB)", video_size_mb) detected_mime = _detect_video_mime_type(temp_video_path) if not detected_mime: raise ValueError(_unsupported_video_format(temp_video_path.suffix)) if video_size_bytes > _VIDEO_SIZE_WARN_BYTES: logger.warning("Video is %.1f MB — may be slow or rejected", video_size_mb) video_data_url = _video_to_base64_data_url(temp_video_path, mime_type=detected_mime) if len(video_data_url) > _MAX_VIDEO_BASE64_BYTES: raise ValueError( f"Video too large for API: base64 payload is {len(video_data_url) / (1024 * 1024):.1f} MB " f"(limit {_MAX_VIDEO_BASE64_BYTES / (1024 * 1024):.0f} MB). " f"Compress or trim the video and retry." ) debug_call_data["video_size_bytes"] = video_size_bytes messages = [{ "role": "user", "content": [ {"type": "text", "text": user_prompt}, {"type": "video_url", "video_url": {"url": video_data_url}}, ], }] vision_timeout, vision_temperature = _read_vision_call_settings(180.0, min_timeout=180.0) call_kwargs = { "task": "vision", "messages": messages, "temperature": vision_temperature, "timeout": vision_timeout, } if model: call_kwargs["model"] = model analysis = await _call_vision_llm(call_kwargs, "Empty video response, retrying once") analysis_length = len(analysis) if analysis else 0 logger.info("Video analysis completed (%s characters)", analysis_length) result = { "success": True, "analysis": analysis or "There was a problem with the request and the video could not be analyzed.", } debug_call_data["success"] = True debug_call_data["analysis_length"] = analysis_length return _finish_analysis("video_analyze_tool", debug_call_data, result) except Exception as e: error_msg = f"Error analyzing video: {str(e)}" logger.error("%s", error_msg, exc_info=True) analysis = _classify_analysis_error( e, _VIDEO_ERROR_RULES, "There was a problem with the request and the video could not " "be analyzed. Error: {e}", ) debug_call_data["error"] = error_msg return _finish_analysis("video_analyze_tool", debug_call_data, { "success": False, "error": error_msg, "analysis": analysis, }) finally: if should_cleanup: _cleanup_temp_media(temp_video_path, "video") VIDEO_ANALYZE_SCHEMA = { "name": "video_analyze", "description": ( "Analyze a video from a URL or local file path using a multimodal AI model. " "Sends the video to a video-capable model (e.g. Gemini) for understanding. " "Use this for video files — for images, use vision_analyze instead. " "Supports mp4, webm, mov, avi, mkv, mpeg formats. " "Note: large videos (>20 MB) may be slow; max ~50 MB." ), "parameters": { "type": "object", "properties": { "video_url": { "type": "string", "description": "Video URL (http/https) or local file path to analyze.", }, "question": { "type": "string", "description": "Your specific question about the video. The AI will describe what happens in the video and answer your question.", }, }, "required": ["video_url", "question"], }, } def _handle_video_analyze(args: Dict[str, Any], **kw: Any) -> Awaitable[str]: video_url = args.get("video_url", "") question = args.get("question", "") full_prompt = ( "Fully describe and explain everything happening in this video, " "including visual content, motion, audio cues, text overlays, and scene " f"transitions. Then answer the following question:\n\n{question}" ) model = _configured_aux_model( ("video", "vision"), ("AUXILIARY_VIDEO_MODEL", "AUXILIARY_VISION_MODEL"), ) return video_analyze_tool(video_url, full_prompt, model, task_id=kw.get("task_id")) registry.register( name="video_analyze", toolset="video", schema=VIDEO_ANALYZE_SCHEMA, handler=_handle_video_analyze, check_fn=check_vision_requirements, is_async=True, emoji="🎬", )