1421 lines
55 KiB
Python
1421 lines
55 KiB
Python
#!/usr/bin/env python3
|
|
"""Vision tools: ``vision_analyze`` (image) and ``video_analyze``.
|
|
|
|
Images resolve through :mod:`tools.image_source`, are normalized to a
|
|
provider-supported format (:mod:`tools.vision_tools_image_prep`), then either
|
|
attach natively to a vision-capable main model (multimodal tool-result
|
|
envelope) or are described by the auxiliary vision LLM router.
|
|
"""
|
|
|
|
import base64
|
|
import asyncio
|
|
import json
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
from io import BytesIO
|
|
import logging
|
|
import os
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Any, Awaitable, Dict, Optional
|
|
from urllib.parse import urlparse
|
|
import httpx
|
|
|
|
# ``agent.auxiliary_client`` costs ~50 ms cold (credential_pool → auth → rich);
|
|
# only the handlers need it. Both names stay module attributes so tests can
|
|
# patch ``tools.vision_tools.async_call_llm``; truthy-skip means injected mocks win.
|
|
async_call_llm: Any = None
|
|
extract_content_or_reasoning: Any = None
|
|
|
|
|
|
def _load_auxiliary_client() -> None:
|
|
global async_call_llm, extract_content_or_reasoning
|
|
if async_call_llm is None or extract_content_or_reasoning is None:
|
|
from agent.auxiliary_client import (
|
|
async_call_llm as _acl,
|
|
extract_content_or_reasoning as _ecr,
|
|
)
|
|
if async_call_llm is None:
|
|
async_call_llm = _acl
|
|
if extract_content_or_reasoning is None:
|
|
extract_content_or_reasoning = _ecr
|
|
|
|
|
|
from hermes_constants import get_hermes_dir
|
|
from tools.debug_helpers import DebugSession
|
|
from tools.website_policy import check_website_access
|
|
from tools.vision_tools_image_prep import ( # noqa: F401 — re-exported for tests/image_source
|
|
_ANTHROPIC_SUPPORTED_MEDIA_TYPES,
|
|
_VISION_MAX_VALIDATED_AGGREGATE_PIXELS,
|
|
_VISION_MAX_VALIDATED_FRAME_COUNT,
|
|
_crop_image_region,
|
|
_detect_image_mime_type_from_bytes,
|
|
_determine_mime_type,
|
|
_image_exceeds_dimension,
|
|
_normalize_to_supported_image,
|
|
_rasterize_svg_to_png,
|
|
_supported_media_types,
|
|
_validate_raster_image_decodable,
|
|
)
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_debug = DebugSession("vision_tools", env_var="VISION_TOOLS_DEBUG")
|
|
|
|
|
|
def _read_vision_setting(env_var: str, key: str, cast, minimum=None):
|
|
"""Env var → config.yaml ``auxiliary.vision.<key>`` → None.
|
|
|
|
Values that fail ``cast`` or fall below ``minimum`` are skipped in favor of
|
|
the next source (a cap can never be disabled by a bad value).
|
|
"""
|
|
def _accept(raw):
|
|
try:
|
|
val = cast(raw)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
return val if minimum is None or val >= minimum else None
|
|
|
|
env_val = os.getenv(env_var, "").strip()
|
|
if env_val:
|
|
val = _accept(env_val)
|
|
if val is not None:
|
|
return val
|
|
try:
|
|
from hermes_cli.config import cfg_get, load_config
|
|
raw = cfg_get(load_config(), "auxiliary", "vision", key)
|
|
if raw is not None:
|
|
return _accept(raw)
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def _resolve_download_timeout() -> float:
|
|
"""HTTP download timeout (separate from ``auxiliary.vision.timeout``, which governs the LLM call)."""
|
|
val = _read_vision_setting("HERMES_VISION_DOWNLOAD_TIMEOUT", "download_timeout", float)
|
|
return 30.0 if val is None else val
|
|
|
|
|
|
_VISION_DOWNLOAD_TIMEOUT = _resolve_download_timeout()
|
|
|
|
# Hard cap on downloaded media (50 MB): bounds memory/disk against
|
|
# attacker-hosted multi-gigabyte files.
|
|
_VISION_MAX_DOWNLOAD_BYTES = 50 * 1024 * 1024
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# CPU-burst concurrency cap (vision encode/resize)
|
|
# ---------------------------------------------------------------------------
|
|
# A turn can fan out dozens of vision_analyze calls ("analyze every frame");
|
|
# each does a CPU-heavy base64 encode + Pillow resize. Sessions share one
|
|
# process, so unbounded encodes saturate every core and starve the shared event
|
|
# loop (the dashboard liveness probe flapped UNHEALTHY in prod). We cap ONLY the
|
|
# CPU burst — the LLM calls stay fully concurrent — on a dedicated executor sized
|
|
# to the usable core count (the resource actually exhausted; no fixed ceiling).
|
|
# It must be a threading primitive: each call runs via model_tools._run_async on
|
|
# a PER-THREAD event loop, so an asyncio semaphore cannot coordinate across them.
|
|
# The default executor is NOT used: it is shared with the gateway/web server.
|
|
import threading # noqa: F401 (kept for downstream importers / patch targets)
|
|
|
|
|
|
def _detect_host_cpus() -> int:
|
|
"""Usable CPU count (``sched_getaffinity`` honors cpuset pinning), at least 1."""
|
|
try:
|
|
return max(1, len(os.sched_getaffinity(0))) # type: ignore[attr-defined]
|
|
except (AttributeError, OSError):
|
|
return max(1, os.cpu_count() or 1)
|
|
|
|
|
|
def _resolve_vision_cpu_workers() -> int:
|
|
"""HERMES_VISION_MAX_CONCURRENCY → ``auxiliary.vision.max_concurrency`` → host cores (values < 1 ignored)."""
|
|
val = _read_vision_setting("HERMES_VISION_MAX_CONCURRENCY", "max_concurrency", int, minimum=1)
|
|
return _detect_host_cpus() if val is None else val
|
|
|
|
|
|
_VISION_CPU_WORKERS = _resolve_vision_cpu_workers()
|
|
|
|
_vision_cpu_executor = ThreadPoolExecutor(
|
|
max_workers=_VISION_CPU_WORKERS,
|
|
thread_name_prefix="vision-encode",
|
|
)
|
|
|
|
|
|
async def _run_encode_on_cpu_executor(fn, *args, **kwargs):
|
|
"""Run a sync encode/resize callable on the bounded vision CPU executor (never the LLM call)."""
|
|
import functools
|
|
loop = asyncio.get_running_loop()
|
|
return await loop.run_in_executor(
|
|
_vision_cpu_executor, functools.partial(fn, *args, **kwargs)
|
|
)
|
|
|
|
|
|
def _image_url_shape_ok(url: str) -> bool:
|
|
"""HTTP(S) shape check only (scheme + netloc; no DNS). Extension-less CDN URLs pass."""
|
|
if not url or not isinstance(url, str) or not url.startswith(("http://", "https://")):
|
|
return False
|
|
return bool(urlparse(url).netloc)
|
|
|
|
|
|
async def _validate_image_url_async(url: str) -> bool:
|
|
"""Validate remote image URL (SSRF guard) without blocking the event loop on DNS."""
|
|
if not _image_url_shape_ok(url):
|
|
return False
|
|
from tools.url_safety import async_is_safe_url
|
|
return await async_is_safe_url(url)
|
|
|
|
|
|
def _is_retryable_download_error(error: Exception) -> bool:
|
|
"""True only for transient download failures worth retrying.
|
|
|
|
Fail-fast: 4xx other than 429 (missing/forbidden), PermissionError (policy
|
|
or SSRF block), ValueError (too large / blocked redirect — deterministic).
|
|
Retryable: 429, 5xx, transport errors and anything unclassified.
|
|
"""
|
|
if isinstance(error, (PermissionError, ValueError)):
|
|
return False
|
|
if isinstance(error, httpx.HTTPStatusError):
|
|
status = error.response.status_code
|
|
return not (400 <= status < 500 and status != 429)
|
|
return True
|
|
|
|
|
|
_DOWNLOAD_USER_AGENT = (
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
|
)
|
|
|
|
|
|
async def _stream_download_to_file(
|
|
client,
|
|
url: str,
|
|
destination: Path,
|
|
max_bytes: int,
|
|
*,
|
|
headers: dict,
|
|
media_label: str = "Image",
|
|
) -> Path:
|
|
"""Stream a GET to *destination* via a temp file with a running size cap.
|
|
|
|
The body is never fully buffered: chunks go to a temp file and the running
|
|
count is checked after each one (Content-Length gives an early reject but
|
|
servers can omit or lie, so the streaming cap is authoritative). The temp
|
|
file is atomically moved onto *destination* on success, deleted on failure.
|
|
"""
|
|
from utils import atomic_replace
|
|
|
|
async with client.stream("GET", url, headers=headers) as response:
|
|
response.raise_for_status()
|
|
|
|
cl = response.headers.get("content-length")
|
|
if cl:
|
|
try:
|
|
declared_size = int(cl)
|
|
except ValueError:
|
|
declared_size = None
|
|
if declared_size is not None and declared_size > max_bytes:
|
|
raise ValueError(
|
|
f"{media_label} too large ({declared_size} bytes, max {max_bytes})"
|
|
)
|
|
|
|
blocked = check_website_access(str(response.url))
|
|
if blocked:
|
|
raise PermissionError(blocked["message"])
|
|
|
|
tmp_destination = destination.with_name(
|
|
f".{destination.name}.{uuid.uuid4().hex}.tmp"
|
|
)
|
|
bytes_written = 0
|
|
try:
|
|
with tmp_destination.open("wb") as f:
|
|
async for chunk in response.aiter_bytes():
|
|
if not chunk:
|
|
continue
|
|
bytes_written += len(chunk)
|
|
if bytes_written > max_bytes:
|
|
raise ValueError(
|
|
f"{media_label} too large ({bytes_written} bytes, max {max_bytes})"
|
|
)
|
|
f.write(chunk)
|
|
atomic_replace(tmp_destination, destination)
|
|
except Exception:
|
|
try:
|
|
tmp_destination.unlink(missing_ok=True)
|
|
except OSError:
|
|
logger.debug(
|
|
"Could not delete partial download: %s", tmp_destination, exc_info=True
|
|
)
|
|
raise
|
|
|
|
return destination
|
|
|
|
|
|
async def _ssrf_redirect_guard(response):
|
|
"""Re-validate each redirect target: a public URL that 302s to
|
|
http://169.254.169.254/ would otherwise bypass the pre-flight is_safe_url check.
|
|
Async because httpx.AsyncClient awaits event hooks."""
|
|
from tools.url_safety import async_is_safe_url, redirect_target_from_response
|
|
redirect_url = redirect_target_from_response(response)
|
|
if redirect_url and not await async_is_safe_url(redirect_url):
|
|
raise ValueError(
|
|
f"Blocked redirect to private/internal address: {redirect_url}"
|
|
)
|
|
|
|
|
|
async def _download_media(
|
|
url: str,
|
|
destination: Path,
|
|
max_retries: int,
|
|
*,
|
|
media_label: str,
|
|
accept: str,
|
|
max_bytes: int,
|
|
timeout: float,
|
|
retry_all: bool,
|
|
) -> Path:
|
|
"""Shared SSRF-safe streaming download with exponential backoff (2s/4s/8s).
|
|
|
|
``retry_all=False`` (images) only retries transient errors per
|
|
:func:`_is_retryable_download_error` — a 404/403 never succeeds on retry, so
|
|
burning three backoff rounds just inflates latency. ``retry_all=True``
|
|
(video) keeps the legacy retry-everything behavior.
|
|
"""
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
last_error = None
|
|
for attempt in range(max_retries):
|
|
try:
|
|
blocked = check_website_access(url)
|
|
if blocked:
|
|
raise PermissionError(blocked["message"])
|
|
|
|
from tools.url_safety import create_ssrf_safe_async_client
|
|
|
|
# follow_redirects for CDNs; the client validates DNS at connect
|
|
# time and the hook re-validates each redirect target.
|
|
async with create_ssrf_safe_async_client(
|
|
timeout=timeout,
|
|
follow_redirects=True,
|
|
event_hooks={"response": [_ssrf_redirect_guard]},
|
|
) as client:
|
|
await _stream_download_to_file(
|
|
client, url, destination, max_bytes,
|
|
headers={"User-Agent": _DOWNLOAD_USER_AGENT, "Accept": accept},
|
|
media_label=media_label,
|
|
)
|
|
return destination
|
|
except Exception as e:
|
|
last_error = e
|
|
final = attempt >= max_retries - 1
|
|
if final or (not retry_all and not _is_retryable_download_error(e)):
|
|
logger.error(
|
|
"%s download failed after %s attempt(s): %s",
|
|
media_label, attempt + 1, str(e)[:100], exc_info=True,
|
|
)
|
|
if not retry_all:
|
|
raise
|
|
break
|
|
wait_time = 2 ** (attempt + 1)
|
|
logger.warning("%s download failed (attempt %s/%s): %s",
|
|
media_label, attempt + 1, max_retries, str(e)[:50])
|
|
if not retry_all:
|
|
logger.warning("Retrying in %ss...", wait_time)
|
|
await asyncio.sleep(wait_time)
|
|
|
|
# Reaching here means max_retries was non-positive (or video exhausted retries).
|
|
if last_error is not None:
|
|
raise last_error
|
|
raise RuntimeError(
|
|
f"_download_{media_label.lower()} exited retry loop without attempting (max_retries={max_retries})"
|
|
)
|
|
|
|
|
|
async def _download_image(image_url: str, destination: Path, max_retries: int = 3) -> Path:
|
|
"""Download an image with SSRF protection and error-class-aware retry."""
|
|
return await _download_media(
|
|
image_url, destination, max_retries,
|
|
media_label="Image", accept="image/*,*/*;q=0.8",
|
|
max_bytes=_VISION_MAX_DOWNLOAD_BYTES, timeout=_VISION_DOWNLOAD_TIMEOUT,
|
|
retry_all=False,
|
|
)
|
|
|
|
|
|
def _image_to_base64_data_url(image_path: Path, mime_type: Optional[str] = None) -> str:
|
|
"""``data:<mime>;base64,...`` for a file (MIME from extension when not given)."""
|
|
encoded = base64.b64encode(image_path.read_bytes()).decode("ascii")
|
|
return f"data:{mime_type or _determine_mime_type(image_path)};base64,{encoded}"
|
|
|
|
|
|
# Absolute hard ceiling for vision payloads (20 MB): no major provider accepts more.
|
|
_MAX_BASE64_BYTES = 20 * 1024 * 1024
|
|
|
|
# Proactive embed caps for conversation-history reuse. Native vision_analyze
|
|
# bakes the data URL into the tool result, re-sent every later turn; a 4 MB /
|
|
# 7900px embed cost ~100-260K billed tokens per image. 256 KB keeps a 1568px
|
|
# screenshot cheap enough to ride the session; Anthropic's tokenizer downsamples
|
|
# to a 1568px long edge anyway, so pixels past that cost wire bytes for no
|
|
# fidelity. The 20 MB / provider 5 MB caps remain as one-shot safety nets.
|
|
_EMBED_TARGET_BYTES = 256 * 1024
|
|
_EMBED_MAX_DIMENSION = 1568
|
|
|
|
# Target when auto-resizing after a provider size rejection (retry once).
|
|
_RESIZE_TARGET_BYTES = 5 * 1024 * 1024
|
|
|
|
_SIZE_ERROR_HINTS = (
|
|
"too large", "payload", "413", "content_too_large",
|
|
"request_too_large", "exceeds", "size limit",
|
|
)
|
|
|
|
|
|
def _is_image_size_error(error: Exception) -> bool:
|
|
"""Detect if an API error is related to image or payload size."""
|
|
err_str = str(error).lower()
|
|
return any(hint in err_str for hint in _SIZE_ERROR_HINTS + ("image_url", "invalid_request"))
|
|
|
|
|
|
def _build_scale_note(
|
|
scale_info: Optional[dict],
|
|
crop_offset: Optional[dict],
|
|
) -> Optional[str]:
|
|
"""Coordinate-mapping disclosure for downscale (``scale_info``) and/or region
|
|
crop (``crop_offset``); ``None`` when neither applied — no note, no noise."""
|
|
parts = []
|
|
if scale_info:
|
|
ow, oh = scale_info["orig_width"], scale_info["orig_height"]
|
|
nw, nh = scale_info["new_width"], scale_info["new_height"]
|
|
fx = ow / nw if nw else 1.0
|
|
fy = oh / nh if nh else 1.0
|
|
if f"{fx:.2f}" == f"{fy:.2f}":
|
|
factor_clause = (
|
|
f"multiply any coordinates you report by {fx:.2f} "
|
|
f"to map back to the original image."
|
|
)
|
|
else:
|
|
factor_clause = (
|
|
f"multiply any x coordinates you report by {fx:.2f} and "
|
|
f"any y coordinates by {fy:.2f} to map back to the "
|
|
f"original image."
|
|
)
|
|
parts.append(
|
|
f"Image downscaled from {ow}x{oh} to {nw}x{nh} for vision; "
|
|
f"{factor_clause}"
|
|
)
|
|
if crop_offset:
|
|
parts.append(
|
|
f"Analysis was performed on a cropped region of the original "
|
|
f"image starting at offset ({crop_offset['x']}, "
|
|
f"{crop_offset['y']}); coordinates are relative to that crop "
|
|
f"origin — add the offset to map back to the full image."
|
|
)
|
|
return " ".join(parts) if parts else None
|
|
|
|
|
|
def _import_pillow_for_resize():
|
|
"""Pillow is a lazy-installable soft dependency; return ``PIL.Image`` or None.
|
|
|
|
``prompt=False``: a blocking input() deadlocks the interactive CLI where
|
|
prompt_toolkit owns stdin. The install is gated by
|
|
security.allow_lazy_installs, so reaching it is already opt-in.
|
|
"""
|
|
try:
|
|
from PIL import Image
|
|
return Image
|
|
except ImportError:
|
|
pass
|
|
try:
|
|
from tools.lazy_deps import ensure as _ensure_dep
|
|
_ensure_dep("tool.vision", prompt=False)
|
|
from PIL import Image
|
|
return Image
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None,
|
|
max_base64_bytes: int = _RESIZE_TARGET_BYTES,
|
|
max_dimension: Optional[int] = None,
|
|
scale_out: Optional[dict] = None,
|
|
force_jpeg: bool = False) -> str:
|
|
"""Base64 data URL, progressively downscaled with Pillow while over budget.
|
|
|
|
Halves dimensions (aspect-preserving, 64px floor) up to 4 times; JPEG also
|
|
walks a quality ladder (85/70/50) at each step. Without Pillow, or if it
|
|
still doesn't fit, returns the best attempt (or raw bytes) and lets the
|
|
caller apply the size check.
|
|
|
|
``max_dimension``: force a downscale above this long edge even when bytes
|
|
fit (Anthropic's 8000px cap is independent of bytes). ``force_jpeg``:
|
|
re-encode PNG input as JPEG when a resize is needed — PNG's only shrink
|
|
lever is halving dimensions, which destroys text legibility on dense
|
|
screenshots; history-reuse embeds opt in. Images under both caps return unchanged.
|
|
"""
|
|
file_size = image_path.stat().st_size
|
|
estimated_b64 = (file_size * 4) // 3 + 100 # base64 ~4/3 + data URL header
|
|
needs_resize_for_bytes = estimated_b64 > max_base64_bytes
|
|
needs_resize_for_dims = (
|
|
max_dimension is not None and _image_exceeds_dimension(image_path, max_dimension)
|
|
)
|
|
|
|
data_url = None
|
|
if not needs_resize_for_bytes and not needs_resize_for_dims:
|
|
data_url = _image_to_base64_data_url(image_path, mime_type=mime_type)
|
|
if len(data_url) <= max_base64_bytes:
|
|
return data_url
|
|
|
|
def _raw() -> str:
|
|
return data_url or _image_to_base64_data_url(image_path, mime_type=mime_type)
|
|
|
|
Image = _import_pillow_for_resize()
|
|
if Image is None:
|
|
logger.info("Pillow not installed — cannot auto-resize oversized image")
|
|
return _raw() # caller will raise the size error
|
|
|
|
logger.info("Image file is %.1f MB (estimated base64 %.1f MB, limit %.1f MB, max_dimension=%s), auto-resizing...",
|
|
file_size / (1024 * 1024), estimated_b64 / (1024 * 1024),
|
|
max_base64_bytes / (1024 * 1024), max_dimension)
|
|
|
|
mime = mime_type or _determine_mime_type(image_path)
|
|
# JPEG for photos (smaller), PNG for transparency — unless force_jpeg.
|
|
pil_format = "PNG" if (mime == "image/png" and not force_jpeg) else "JPEG"
|
|
out_mime = "image/png" if pil_format == "PNG" else "image/jpeg"
|
|
|
|
try:
|
|
img = Image.open(image_path)
|
|
except Exception as exc:
|
|
logger.info("Pillow cannot open image for resizing: %s", exc)
|
|
return _raw()
|
|
# JPEG cannot encode alpha/palette modes (force_jpeg routes PNGs here).
|
|
if pil_format == "JPEG" and img.mode not in {"RGB", "L"}:
|
|
img = img.convert("RGB")
|
|
|
|
quality_steps = (85, 70, 50) if pil_format == "JPEG" else (None,)
|
|
orig_dims = prev_dims = (img.width, img.height)
|
|
candidate = None
|
|
|
|
def _record_scale(w: int, h: int) -> None:
|
|
if scale_out is not None and (w, h) != orig_dims:
|
|
scale_out.update(orig_width=orig_dims[0], orig_height=orig_dims[1],
|
|
new_width=w, new_height=h)
|
|
|
|
for attempt in range(5):
|
|
if attempt > 0:
|
|
# Halve, then re-derive the scale from whichever axis hit the 64px
|
|
# floor so both axes shrink by the same factor.
|
|
new_w = max(int(img.width * 0.5), 64)
|
|
new_h = max(int(img.height * 0.5), 64)
|
|
if new_w == 64 and img.width > 0:
|
|
new_h = max(int(img.height * (64 / img.width)), 64)
|
|
elif new_h == 64 and img.height > 0:
|
|
new_w = max(int(img.width * (64 / img.height)), 64)
|
|
if (new_w, new_h) == prev_dims:
|
|
break
|
|
img = img.resize((new_w, new_h), Image.LANCZOS)
|
|
prev_dims = (new_w, new_h)
|
|
logger.info("Resized to %dx%d (attempt %d)", new_w, new_h, attempt)
|
|
|
|
for q in quality_steps:
|
|
buf = BytesIO()
|
|
img.save(buf, format=pil_format, **({} if q is None else {"quality": q}))
|
|
candidate = f"data:{out_mime};base64,{base64.b64encode(buf.getvalue()).decode('ascii')}"
|
|
dims_ok = max_dimension is None or max(img.width, img.height) <= max_dimension
|
|
if len(candidate) <= max_base64_bytes and dims_ok:
|
|
logger.info("Auto-resized image fits: %.1f MB (quality=%s, %dx%d)",
|
|
len(candidate) / (1024 * 1024), q,
|
|
img.width, img.height)
|
|
_record_scale(img.width, img.height)
|
|
return candidate
|
|
|
|
if candidate is not None:
|
|
logger.warning("Auto-resize could not fit image under %.1f MB (best: %.1f MB)",
|
|
max_base64_bytes / (1024 * 1024), len(candidate) / (1024 * 1024))
|
|
_record_scale(img.width, img.height)
|
|
return candidate
|
|
return _raw()
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Native fast path: when the active main model supports vision, skip the aux
|
|
# LLM and return the image bytes as a multimodal tool-result envelope. The
|
|
# agent loop unwraps it into an OpenAI-style content list on the `tool` role;
|
|
# provider adapters translate that per backend, so the main model "sees" the
|
|
# pixels directly on its next turn.
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# Providers whose tool results accept image content (spec docs verified
|
|
# Apr-2026): Anthropic Messages (+ aggregators proxying Claude — assume support,
|
|
# falling back to text would regress their frontier models), OpenAI Chat
|
|
# Completions / Responses. Gemini is gated on model: only 3.x supports
|
|
# multimodal functionResponse.
|
|
_TOOL_RESULT_MEDIA_PROVIDERS = frozenset({
|
|
"openrouter", "nous", "vertex", "bedrock", "anthropic-vertex", "google-vertex",
|
|
"anthropic", "claude", "anthropic-direct",
|
|
"openai", "openai-chat", "openai-codex", "azure-openai",
|
|
})
|
|
_GEMINI_PROVIDERS = frozenset({"google", "gemini", "google-gemini", "google-vertex-gemini"})
|
|
|
|
|
|
def _supports_media_in_tool_results(provider: str, model: str) -> bool:
|
|
"""Whether provider+model accepts image content inside a tool-result message.
|
|
|
|
Unknown providers are conservatively False (caller falls back to the aux-LLM
|
|
text path) unless their ``ProviderProfile`` declares ``supports_vision``.
|
|
"""
|
|
if not isinstance(provider, str):
|
|
return False
|
|
p = provider.strip().lower()
|
|
if not p:
|
|
return False
|
|
if p in _TOOL_RESULT_MEDIA_PROVIDERS:
|
|
return True
|
|
if p in _GEMINI_PROVIDERS:
|
|
if not isinstance(model, str):
|
|
return False
|
|
m = model.strip().lower()
|
|
return any(tag in m for tag in ("gemini-3", "gemini-pro-3", "gemini-flash-3"))
|
|
try:
|
|
from providers import get_provider_profile
|
|
profile = get_provider_profile(p)
|
|
if profile is not None and profile.supports_vision:
|
|
return True
|
|
except Exception:
|
|
pass
|
|
return False
|
|
|
|
|
|
def _should_use_native_vision_fast_path() -> bool:
|
|
"""True when image routing resolves to ``native`` AND the provider accepts
|
|
images in tool results, or the user set the ``model.supports_vision``
|
|
override (escape hatch for custom/local providers). Any failure → False."""
|
|
try:
|
|
from agent.auxiliary_client import _read_main_provider, _read_main_model
|
|
from agent.image_routing import decide_image_input_mode, _lookup_supports_vision
|
|
from hermes_cli.config import load_config
|
|
|
|
provider = _read_main_provider()
|
|
model = _read_main_model()
|
|
cfg = load_config()
|
|
if decide_image_input_mode(provider, model, cfg) != "native":
|
|
return False
|
|
return (
|
|
_supports_media_in_tool_results(provider, model)
|
|
or _lookup_supports_vision(provider, model, cfg) is True
|
|
)
|
|
except Exception as exc:
|
|
logger.debug("Native vision fast-path check failed: %s", exc)
|
|
return False
|
|
|
|
|
|
def _build_native_vision_tool_result(
|
|
image_url: str,
|
|
question: str,
|
|
image_data_url: str,
|
|
image_size_bytes: int,
|
|
scale_note: Optional[str] = None,
|
|
) -> Dict[str, Any]:
|
|
"""Multimodal tool-result envelope (``_multimodal`` + ``content`` list).
|
|
|
|
The text part is intentionally minimal — the model already has the question
|
|
in context; it acknowledges the image is visible. ``text_summary`` is the
|
|
fallback for providers without multimodal tool results.
|
|
"""
|
|
text_part = (
|
|
"Image loaded into your context — you can see it natively now. "
|
|
"Use your built-in vision to answer the user."
|
|
)
|
|
if isinstance(question, str) and question.strip():
|
|
text_part += f"\n\nQuestion: {question.strip()}"
|
|
if scale_note:
|
|
text_part += f"\n\nNote: {scale_note}"
|
|
|
|
summary = (
|
|
f"Image attached natively for the main model "
|
|
f"({image_size_bytes / 1024:.1f} KB). "
|
|
"Answer using built-in vision."
|
|
)
|
|
|
|
return {
|
|
"_multimodal": True,
|
|
"content": [
|
|
{"type": "text", "text": text_part},
|
|
{"type": "image_url", "image_url": {"url": image_data_url}},
|
|
],
|
|
"text_summary": summary,
|
|
"meta": {
|
|
"image_url": image_url[:200],
|
|
"size_bytes": image_size_bytes,
|
|
"native_vision": True,
|
|
},
|
|
}
|
|
|
|
|
|
def _unlink_quietly(path: Optional[Path]) -> None:
|
|
if path is not None:
|
|
try:
|
|
if path.exists():
|
|
path.unlink()
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
class _ImagePrepError(ValueError):
|
|
"""Raised by :func:`_prepare_image`; the message is user-facing."""
|
|
|
|
|
|
class _PreparedImage:
|
|
"""Temp image ready to encode; ``path`` is owned by the caller (delete it)."""
|
|
__slots__ = ("path", "mime", "size_bytes", "crop_offset")
|
|
|
|
def __init__(self, path: Path, mime: Optional[str], size_bytes: int, crop_offset: dict):
|
|
self.path, self.mime, self.size_bytes, self.crop_offset = path, mime, size_bytes, crop_offset
|
|
|
|
|
|
async def _prepare_image(
|
|
image_url: str, task_id: Optional[str], region: Optional[list], *, validate_decode: bool,
|
|
) -> _PreparedImage:
|
|
"""Resolve → materialize → normalize → (validate) → (crop). Raises ``_ImagePrepError``.
|
|
|
|
The single resolver unifies data:/http/file/local/container sources and
|
|
enforces terminal-backend confinement; bytes land in a temp file so the
|
|
path-based encode/resize pipeline is reused. Unsupported formats (SVG, BMP)
|
|
are converted to PNG BEFORE encoding — an unsupported media_type baked into
|
|
immutable history would 400 on every resume. The crop runs BEFORE any
|
|
downscale so the region keeps the full resolution budget. Blocking
|
|
rasterizer/Pillow work is offloaded. Intermediate temp files are deleted;
|
|
on error nothing is left behind.
|
|
"""
|
|
from tools.image_source import ImageResolutionError, ResolveContext, resolve_image_source
|
|
|
|
try:
|
|
resolved = await resolve_image_source(image_url, ResolveContext(task_id=task_id))
|
|
except ImageResolutionError as exc:
|
|
raise _ImagePrepError(str(exc)) from exc
|
|
|
|
temp_dir = get_hermes_dir("cache/vision", "temp_vision_images")
|
|
temp_dir.mkdir(parents=True, exist_ok=True)
|
|
path = temp_dir / f"temp_image_{uuid.uuid4()}.img"
|
|
await asyncio.to_thread(path.write_bytes, resolved.data)
|
|
mime = resolved.mime
|
|
size_bytes = len(resolved.data)
|
|
crop_offset: dict = {}
|
|
try:
|
|
normalized_path, mime, norm_err = await asyncio.to_thread(
|
|
_normalize_to_supported_image, path, mime,
|
|
)
|
|
if norm_err or normalized_path is None:
|
|
raise _ImagePrepError(norm_err or "Image normalization failed.")
|
|
if normalized_path != path:
|
|
_unlink_quietly(path)
|
|
path = normalized_path
|
|
size_bytes = path.stat().st_size
|
|
|
|
if validate_decode:
|
|
decode_error = await _run_encode_on_cpu_executor(
|
|
_validate_raster_image_decodable, path,
|
|
_VISION_MAX_VALIDATED_FRAME_COUNT, _VISION_MAX_VALIDATED_AGGREGATE_PIXELS,
|
|
)
|
|
if decode_error:
|
|
raise _ImagePrepError(decode_error)
|
|
|
|
if region is not None:
|
|
cropped_path, cropped_mime, crop_err = await asyncio.to_thread(
|
|
_crop_image_region, path, region, offset_out=crop_offset,
|
|
)
|
|
if crop_err or cropped_path is None:
|
|
raise _ImagePrepError(crop_err or "Region crop failed.")
|
|
_unlink_quietly(path)
|
|
path = cropped_path
|
|
mime = cropped_mime
|
|
size_bytes = path.stat().st_size
|
|
except BaseException:
|
|
_unlink_quietly(path)
|
|
raise
|
|
return _PreparedImage(path, mime, size_bytes, crop_offset)
|
|
|
|
|
|
def _too_large_message(image_data_url: str) -> str:
|
|
return (
|
|
f"Image too large for vision API: base64 payload is "
|
|
f"{len(image_data_url) / (1024 * 1024):.1f} MB "
|
|
f"(limit {_MAX_BASE64_BYTES / (1024 * 1024):.0f} MB) "
|
|
f"even after resizing. Install Pillow "
|
|
f"(`pip install Pillow`) for better auto-resize, "
|
|
f"or compress the image manually."
|
|
)
|
|
|
|
|
|
async def _vision_analyze_native(
|
|
image_url: str,
|
|
question: str,
|
|
task_id: Optional[str] = None,
|
|
region: Optional[list] = None,
|
|
) -> Any:
|
|
"""Fast path for vision-capable main models.
|
|
|
|
Returns a ``_multimodal`` envelope dict on success, or a JSON error string
|
|
(the normal tool-result contract) on failure.
|
|
"""
|
|
if not isinstance(image_url, str) or not image_url.strip():
|
|
return tool_error("image_url is required", success=False)
|
|
|
|
prepared: Optional[_PreparedImage] = None
|
|
try:
|
|
from tools.interrupt import is_interrupted
|
|
if is_interrupted():
|
|
return tool_error("Interrupted", success=False)
|
|
|
|
try:
|
|
prepared = await _prepare_image(image_url, task_id, region, validate_decode=True)
|
|
except _ImagePrepError as exc:
|
|
return tool_error(str(exc), success=False)
|
|
|
|
image_data_url = await _run_encode_on_cpu_executor(
|
|
_image_to_base64_data_url, prepared.path, mime_type=prepared.mime,
|
|
)
|
|
|
|
# Proactive embed cap: this image is re-sent on every later turn, so
|
|
# resize DOWN to the history-reuse target whenever the byte or long-edge
|
|
# cap is exceeded, not just at the 20 MB hard ceiling.
|
|
_scale_info: dict = {}
|
|
_over_dims = await _run_encode_on_cpu_executor(
|
|
_image_exceeds_dimension, prepared.path, _EMBED_MAX_DIMENSION,
|
|
)
|
|
if len(image_data_url) > _EMBED_TARGET_BYTES or _over_dims:
|
|
image_data_url = await _run_encode_on_cpu_executor(
|
|
_resize_image_for_vision,
|
|
prepared.path, mime_type=prepared.mime,
|
|
max_base64_bytes=_EMBED_TARGET_BYTES,
|
|
max_dimension=_EMBED_MAX_DIMENSION,
|
|
scale_out=_scale_info,
|
|
force_jpeg=True,
|
|
)
|
|
# Reject rather than embed a session-wedging payload.
|
|
if len(image_data_url) > _MAX_BASE64_BYTES:
|
|
return tool_error(_too_large_message(image_data_url), success=False)
|
|
|
|
return _build_native_vision_tool_result(
|
|
image_url=image_url,
|
|
question=question,
|
|
image_data_url=image_data_url,
|
|
image_size_bytes=prepared.size_bytes,
|
|
scale_note=_build_scale_note(
|
|
_scale_info or None, prepared.crop_offset or None,
|
|
),
|
|
)
|
|
|
|
except Exception as exc:
|
|
logger.warning("Native vision fast path failed: %s", exc)
|
|
return tool_error(f"Native vision failed: {exc}", success=False)
|
|
finally:
|
|
# Only delete temp files we created — never user-provided paths.
|
|
if prepared is not None:
|
|
_unlink_quietly(prepared.path)
|
|
|
|
|
|
def _read_vision_call_settings(default_timeout: float, *, min_timeout: Optional[float] = None):
|
|
"""``auxiliary.vision.timeout`` / ``.temperature`` from config.yaml (defaults 120s-ish / 0.1).
|
|
|
|
Local vision models (llama.cpp, ollama) can take well over 30s, hence the
|
|
generous defaults; ``min_timeout`` lets video enforce a floor.
|
|
"""
|
|
timeout, temperature = default_timeout, 0.1
|
|
try:
|
|
from hermes_cli.config import cfg_get, load_config
|
|
_vision_cfg = cfg_get(load_config(), "auxiliary", "vision", default={})
|
|
_vt = _vision_cfg.get("timeout")
|
|
if _vt is not None:
|
|
timeout = float(_vt) if min_timeout is None else max(float(_vt), min_timeout)
|
|
_vtemp = _vision_cfg.get("temperature")
|
|
if _vtemp is not None:
|
|
temperature = float(_vtemp)
|
|
except Exception:
|
|
pass
|
|
return timeout, temperature
|
|
|
|
|
|
# Error-message classification for the aux-LLM paths: first matching hint set
|
|
# wins (order matters — billing before capability before size/format).
|
|
_BILLING_HINTS = ("402", "insufficient", "payment required", "credits", "billing")
|
|
_IMAGE_ERROR_RULES = (
|
|
(_BILLING_HINTS,
|
|
"Insufficient credits or payment required. Please top up your "
|
|
"API provider account and try again. Error: {e}"),
|
|
(("does not support", "not support image", "content_policy", "multimodal",
|
|
"unrecognized request argument", "image input"),
|
|
"{model} does not support vision or our request was not "
|
|
"accepted by the server. Error: {e}"),
|
|
(("invalid_request", "image_url"),
|
|
"The vision API rejected the image. This can happen when the "
|
|
"image is in an unsupported format, corrupted, or still too "
|
|
"large after auto-resize. Try a smaller JPEG/PNG and retry. "
|
|
"Error: {e}"),
|
|
)
|
|
_VIDEO_ERROR_RULES = (
|
|
(_BILLING_HINTS, _IMAGE_ERROR_RULES[0][1]),
|
|
(("does not support", "not support video", "content_policy", "multimodal",
|
|
"unrecognized request argument", "video input", "video_url"),
|
|
"The model does not support video analysis or the request was "
|
|
"rejected. Ensure you're using a video-capable model "
|
|
"(e.g. google/gemini-2.5-flash). Error: {e}"),
|
|
(_SIZE_ERROR_HINTS,
|
|
"The video is too large for the API. Try compressing or trimming "
|
|
"the video (max ~50 MB). Error: {e}"),
|
|
)
|
|
|
|
|
|
def _classify_analysis_error(e: Exception, rules, fallback: str, **fmt) -> str:
|
|
err_str = str(e).lower()
|
|
for hints, template in rules:
|
|
if any(hint in err_str for hint in hints):
|
|
return template.format(e=e, **fmt)
|
|
return fallback.format(e=e)
|
|
|
|
|
|
def _debug_call_data(kind: str, source: str, user_prompt: str, model) -> dict:
|
|
return {
|
|
"parameters": {
|
|
f"{kind}_url": source,
|
|
"user_prompt": user_prompt[:200] + "..." if len(user_prompt) > 200 else user_prompt,
|
|
"model": model,
|
|
},
|
|
"error": None,
|
|
"success": False,
|
|
"analysis_length": 0,
|
|
"model_used": model,
|
|
f"{kind}_size_bytes": 0,
|
|
}
|
|
|
|
|
|
def _finish_analysis(tool_name: str, debug_call_data: dict, result: dict) -> str:
|
|
_debug.log_call(tool_name, debug_call_data)
|
|
_debug.save()
|
|
return json.dumps(result, indent=2, ensure_ascii=False)
|
|
|
|
|
|
def _cleanup_temp_media(path: Optional[Path], label: str) -> None:
|
|
if path and path.exists():
|
|
try:
|
|
path.unlink()
|
|
logger.debug("Cleaned up temporary %s file", label)
|
|
except Exception as cleanup_error:
|
|
logger.warning(
|
|
"Could not delete temporary file: %s", cleanup_error, exc_info=True
|
|
)
|
|
|
|
|
|
async def _call_vision_llm(call_kwargs: dict, empty_log: str):
|
|
"""Call the aux vision LLM, retrying once on empty content (reasoning-only response)."""
|
|
_load_auxiliary_client()
|
|
response = await async_call_llm(**call_kwargs)
|
|
analysis = extract_content_or_reasoning(response)
|
|
if not analysis:
|
|
logger.warning(empty_log)
|
|
response = await async_call_llm(**call_kwargs)
|
|
analysis = extract_content_or_reasoning(response)
|
|
return analysis
|
|
|
|
|
|
async def vision_analyze_tool(
|
|
image_url: str,
|
|
user_prompt: str,
|
|
model: str = None,
|
|
task_id: Optional[str] = None,
|
|
region: Optional[list] = None,
|
|
) -> str:
|
|
"""Describe an image (URL, local path, data: URL) with the auxiliary vision LLM.
|
|
|
|
``user_prompt`` is pre-formatted by the caller. Returns JSON
|
|
``{"success": bool, "analysis": str}`` (``analysis`` carries the error
|
|
explanation on failure). Temp images live under $HERMES_HOME/cache/vision/.
|
|
"""
|
|
if not isinstance(user_prompt, str):
|
|
user_prompt = str(user_prompt) if user_prompt is not None else ""
|
|
debug_call_data = _debug_call_data("image", image_url, user_prompt, model)
|
|
|
|
prepared: Optional[_PreparedImage] = None
|
|
try:
|
|
from tools.interrupt import is_interrupted
|
|
if is_interrupted():
|
|
return tool_error("Interrupted", success=False)
|
|
|
|
logger.info("Analyzing image: %s", image_url[:60])
|
|
logger.info("User prompt: %s", user_prompt[:100])
|
|
|
|
prepared = await _prepare_image(image_url, task_id, region, validate_decode=False)
|
|
logger.info("Image ready (%.1f KB)", prepared.size_bytes / 1024)
|
|
|
|
# Send at full resolution first; on a size rejection, downscale and retry.
|
|
logger.info("Converting image to base64...")
|
|
image_data_url = await _run_encode_on_cpu_executor(
|
|
_image_to_base64_data_url, prepared.path, mime_type=prepared.mime)
|
|
logger.info("Image converted to base64 (%.1f KB)", len(image_data_url) / 1024)
|
|
|
|
_scale_info: dict = {}
|
|
if len(image_data_url) > _MAX_BASE64_BYTES:
|
|
image_data_url = await _run_encode_on_cpu_executor(
|
|
_resize_image_for_vision,
|
|
prepared.path, mime_type=prepared.mime,
|
|
scale_out=_scale_info)
|
|
if len(image_data_url) > _MAX_BASE64_BYTES:
|
|
raise ValueError(_too_large_message(image_data_url))
|
|
|
|
debug_call_data["image_size_bytes"] = prepared.size_bytes
|
|
|
|
messages = [{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": user_prompt},
|
|
{"type": "image_url", "image_url": {"url": image_data_url}},
|
|
],
|
|
}]
|
|
|
|
logger.info("Processing image with vision model...")
|
|
vision_timeout, vision_temperature = _read_vision_call_settings(120.0)
|
|
call_kwargs = {
|
|
"task": "vision",
|
|
"messages": messages,
|
|
"temperature": vision_temperature,
|
|
"timeout": vision_timeout,
|
|
}
|
|
if model:
|
|
call_kwargs["model"] = model
|
|
_load_auxiliary_client()
|
|
try:
|
|
response = await async_call_llm(**call_kwargs)
|
|
except Exception as _api_err:
|
|
if (_is_image_size_error(_api_err)
|
|
and len(image_data_url) > _RESIZE_TARGET_BYTES):
|
|
logger.info(
|
|
"API rejected image (%.1f MB, likely too large); "
|
|
"auto-resizing to ~%.0f MB and retrying...",
|
|
len(image_data_url) / (1024 * 1024),
|
|
_RESIZE_TARGET_BYTES / (1024 * 1024),
|
|
)
|
|
image_data_url = await _run_encode_on_cpu_executor(
|
|
_resize_image_for_vision,
|
|
prepared.path, mime_type=prepared.mime,
|
|
scale_out=_scale_info)
|
|
messages[0]["content"][1]["image_url"]["url"] = image_data_url
|
|
response = await async_call_llm(**call_kwargs)
|
|
else:
|
|
raise
|
|
|
|
analysis = extract_content_or_reasoning(response)
|
|
if not analysis:
|
|
logger.warning("Vision LLM returned empty content, retrying once")
|
|
response = await async_call_llm(**call_kwargs)
|
|
analysis = extract_content_or_reasoning(response)
|
|
|
|
analysis_length = len(analysis)
|
|
logger.info("Image analysis completed (%s characters)", analysis_length)
|
|
|
|
analysis = analysis or "There was a problem with the request and the image could not be analyzed."
|
|
scale_note = _build_scale_note(_scale_info or None, prepared.crop_offset or None)
|
|
result = {
|
|
"success": True,
|
|
"analysis": f"[{scale_note}] {analysis}" if scale_note else analysis,
|
|
}
|
|
if scale_note:
|
|
result["scale_note"] = scale_note
|
|
|
|
debug_call_data["success"] = True
|
|
debug_call_data["analysis_length"] = analysis_length
|
|
return _finish_analysis("vision_analyze_tool", debug_call_data, result)
|
|
|
|
except Exception as e:
|
|
error_msg = f"Error analyzing image: {str(e)}"
|
|
logger.error("%s", error_msg, exc_info=True)
|
|
analysis = _classify_analysis_error(
|
|
e, _IMAGE_ERROR_RULES,
|
|
"There was a problem with the request and the image could not "
|
|
"be analyzed. Error: {e}",
|
|
model=model,
|
|
)
|
|
debug_call_data["error"] = error_msg
|
|
return _finish_analysis("vision_analyze_tool", debug_call_data, {
|
|
"success": False,
|
|
"error": error_msg,
|
|
"analysis": analysis,
|
|
})
|
|
|
|
finally:
|
|
if prepared is not None:
|
|
_cleanup_temp_media(prepared.path, "image")
|
|
|
|
|
|
def check_vision_requirements() -> bool:
|
|
"""True when ``call_llm(task="vision")`` could resolve a client.
|
|
|
|
Mirrors its runtime fallback chain: explicit ``auxiliary.vision.provider``,
|
|
then the auto chain (main provider → openrouter → nous) — without the auto
|
|
step the tool would vanish whenever the explicit name was unresolvable.
|
|
Probe mode skips real SDK client construction (openai import + SSL setup)
|
|
on the tool-gating path; resolution policy is identical.
|
|
"""
|
|
try:
|
|
from agent.auxiliary_client import aux_probe_mode, resolve_vision_provider_client
|
|
except ImportError:
|
|
return False
|
|
try:
|
|
with aux_probe_mode():
|
|
_provider, client, _model = resolve_vision_provider_client()
|
|
if client is not None:
|
|
return True
|
|
_provider, client, _model = resolve_vision_provider_client(provider="auto")
|
|
return client is not None
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Registry
|
|
# ---------------------------------------------------------------------------
|
|
from tools.registry import registry, tool_error
|
|
|
|
VISION_ANALYZE_SCHEMA = {
|
|
"name": "vision_analyze",
|
|
# Routing mechanics are deliberately absent (the route is automatic and the
|
|
# native result says so itself); region keeps its pre-effect guidance — a
|
|
# model that doesn't know crops keep full resolution never zooms.
|
|
"description": (
|
|
"Load an image into the conversation so you can see it. Call it "
|
|
"any time the user references an image — then answer from what "
|
|
"you see."
|
|
),
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"image_url": {
|
|
"type": "string",
|
|
"description": "Image URL (http/https), local file path, or data: URL to load."
|
|
},
|
|
"question": {
|
|
"type": "string",
|
|
"description": "Your question or request about the image."
|
|
},
|
|
"region": {
|
|
"type": "array",
|
|
"items": {"type": "integer"},
|
|
"minItems": 4,
|
|
"maxItems": 4,
|
|
"description": (
|
|
"Optional [x1, y1, x2, y2] crop in ORIGINAL-image pixel "
|
|
"coordinates, applied before any downscaling — the crop "
|
|
"keeps full resolution. Load the full image first, then "
|
|
"re-call with a region to zoom into small text or fine "
|
|
"detail."
|
|
)
|
|
}
|
|
},
|
|
"required": ["image_url", "question"]
|
|
}
|
|
}
|
|
|
|
|
|
def _configured_aux_model(sections: tuple, env_vars: tuple) -> Optional[str]:
|
|
"""First non-empty ``auxiliary.<section>.model`` from config.yaml, else the
|
|
first non-empty env var (legacy override), else None."""
|
|
try:
|
|
from hermes_cli.config import cfg_get, load_config
|
|
_cfg = load_config()
|
|
for section in sections:
|
|
_vmodel = cfg_get(_cfg, "auxiliary", section, "model")
|
|
if _vmodel:
|
|
model = str(_vmodel).strip() or None
|
|
if model:
|
|
return model
|
|
break
|
|
except Exception:
|
|
pass
|
|
for env_var in env_vars:
|
|
val = os.getenv(env_var, "").strip()
|
|
if val:
|
|
return val
|
|
return None
|
|
|
|
|
|
async def _handle_vision_analyze(args: Dict[str, Any], **kw: Any) -> str:
|
|
image_url = args.get("image_url", "")
|
|
question = args.get("question", "")
|
|
region = args.get("region")
|
|
task_id = kw.get("task_id")
|
|
|
|
# No concurrency gate around the whole analysis — the CPU burst is bounded
|
|
# inside the encode/resize step, so multi-image fan-out keeps full request
|
|
# concurrency. Native fast path: main model sees the pixels directly, no
|
|
# aux call, no information loss.
|
|
if _should_use_native_vision_fast_path():
|
|
logger.info("vision_analyze: native fast path")
|
|
return await _vision_analyze_native(image_url, question, task_id=task_id, region=region)
|
|
|
|
# Legacy path: aux LLM describes the image and we return its text.
|
|
full_prompt = (
|
|
"Fully describe and explain everything about this image, then answer the "
|
|
f"following question:\n\n{question}"
|
|
)
|
|
model = _configured_aux_model(("vision",), ("AUXILIARY_VISION_MODEL",))
|
|
return await vision_analyze_tool(image_url, full_prompt, model, task_id=task_id, region=region)
|
|
|
|
|
|
registry.register(
|
|
name="vision_analyze",
|
|
toolset="vision",
|
|
schema=VISION_ANALYZE_SCHEMA,
|
|
handler=_handle_vision_analyze,
|
|
check_fn=check_vision_requirements,
|
|
is_async=True,
|
|
emoji="👁️",
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Video Analysis Tool
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# Extension → MIME. avi/mkv fall back to mp4.
|
|
_VIDEO_MIME_TYPES = {
|
|
".mp4": "video/mp4",
|
|
".webm": "video/webm",
|
|
".mov": "video/mov",
|
|
".avi": "video/mp4",
|
|
".mkv": "video/mp4",
|
|
".mpeg": "video/mpeg",
|
|
".mpg": "video/mpeg",
|
|
}
|
|
|
|
_MAX_VIDEO_BASE64_BYTES = 50 * 1024 * 1024 # 50 MB hard cap
|
|
_VIDEO_SIZE_WARN_BYTES = 20 * 1024 * 1024
|
|
|
|
|
|
def _detect_video_mime_type(video_path: Path) -> Optional[str]:
|
|
"""Video MIME type from extension, or None if unsupported."""
|
|
return _VIDEO_MIME_TYPES.get(video_path.suffix.lower())
|
|
|
|
|
|
def _unsupported_video_format(suffix: str) -> str:
|
|
return (
|
|
f"Unsupported video format: '{suffix}'. "
|
|
f"Supported: {', '.join(sorted(_VIDEO_MIME_TYPES.keys()))}"
|
|
)
|
|
|
|
|
|
def _video_to_base64_data_url(video_path: Path, mime_type: Optional[str] = None) -> str:
|
|
encoded = base64.b64encode(video_path.read_bytes()).decode("ascii")
|
|
mime = mime_type or _VIDEO_MIME_TYPES.get(video_path.suffix.lower(), "video/mp4")
|
|
return f"data:{mime};base64,{encoded}"
|
|
|
|
|
|
def _is_path_like_video_source(value: str) -> bool:
|
|
lowered = (value or "").strip().lower()
|
|
return bool(lowered) and not lowered.startswith(("http://", "https://", "data:"))
|
|
|
|
|
|
async def _materialize_video_from_terminal_backend(video_source: str, task_id: Optional[str]) -> Path:
|
|
"""Read a path via the shared media resolver into a local temp video file.
|
|
|
|
``permitted=("video",)`` gives terminal-backend video reads the exact
|
|
pipeline vision_analyze uses: media-cache host reads (gateway downloads live
|
|
on the host, not in the sandbox), bounded in-sandbox exec-read, lazy env
|
|
bring-up, the credential-read guard, and the 50MB ingest cap.
|
|
"""
|
|
from tools.image_source import ImageResolutionError, ResolveContext, resolve_image_source
|
|
|
|
source = video_source
|
|
if source.startswith("file://"):
|
|
source = source[len("file://"):]
|
|
suffix = Path(source).suffix.lower()
|
|
if suffix not in _VIDEO_MIME_TYPES:
|
|
raise ValueError(_unsupported_video_format(suffix))
|
|
|
|
try:
|
|
resolved = await resolve_image_source(
|
|
video_source, ResolveContext(task_id=task_id), permitted=("video",)
|
|
)
|
|
except ImageResolutionError as exc:
|
|
raise ValueError(f"Could not read video from terminal backend: {exc}") from exc
|
|
|
|
temp_dir = get_hermes_dir("cache/video", "temp_video_files")
|
|
temp_dir.mkdir(parents=True, exist_ok=True)
|
|
temp_path = temp_dir / f"terminal_video_{uuid.uuid4()}{suffix}"
|
|
temp_path.write_bytes(resolved.data)
|
|
return temp_path
|
|
|
|
|
|
async def _download_video(video_url: str, destination: Path, max_retries: int = 3) -> Path:
|
|
"""Download video with SSRF protection; every failure class is retried."""
|
|
return await _download_media(
|
|
video_url, destination, max_retries,
|
|
media_label="Video", accept="video/*,*/*;q=0.8",
|
|
max_bytes=_MAX_VIDEO_BASE64_BYTES, timeout=60.0, retry_all=True,
|
|
)
|
|
|
|
|
|
async def video_analyze_tool(
|
|
video_url: str,
|
|
user_prompt: str,
|
|
model: str = None,
|
|
task_id: Optional[str] = None,
|
|
) -> str:
|
|
"""Analyze a video via multimodal LLM. Returns JSON {success, analysis}."""
|
|
if not isinstance(user_prompt, str):
|
|
user_prompt = str(user_prompt) if user_prompt is not None else ""
|
|
debug_call_data = _debug_call_data("video", video_url, user_prompt, model)
|
|
|
|
temp_video_path = None
|
|
should_cleanup = True
|
|
|
|
try:
|
|
from tools.interrupt import is_interrupted
|
|
if is_interrupted():
|
|
return tool_error("Interrupted", success=False)
|
|
|
|
logger.info("Analyzing video: %s", video_url[:60])
|
|
logger.info("User prompt: %s", user_prompt[:100])
|
|
|
|
resolved_url = video_url
|
|
if resolved_url.startswith("file://"):
|
|
resolved_url = resolved_url[len("file://"):]
|
|
local_path = Path(os.path.expanduser(resolved_url))
|
|
|
|
from tools.image_source import _is_local_terminal_backend
|
|
if not _is_local_terminal_backend() and _is_path_like_video_source(video_url):
|
|
logger.info("Reading video source via terminal backend: %s", video_url)
|
|
temp_video_path = await _materialize_video_from_terminal_backend(video_url, task_id)
|
|
elif local_path.is_file():
|
|
from agent.file_safety import raise_if_read_blocked
|
|
raise_if_read_blocked(str(local_path))
|
|
logger.info("Using local video file: %s", video_url)
|
|
temp_video_path = local_path
|
|
should_cleanup = False
|
|
elif await _validate_image_url_async(video_url):
|
|
blocked = check_website_access(video_url)
|
|
if blocked:
|
|
raise PermissionError(blocked["message"])
|
|
temp_dir = get_hermes_dir("cache/video", "temp_video_files")
|
|
temp_video_path = temp_dir / f"temp_video_{uuid.uuid4()}.mp4"
|
|
await _download_video(video_url, temp_video_path)
|
|
else:
|
|
raise ValueError(
|
|
"Invalid video source. Provide an HTTP/HTTPS URL or a valid local file path."
|
|
)
|
|
|
|
video_size_bytes = temp_video_path.stat().st_size
|
|
video_size_mb = video_size_bytes / (1024 * 1024)
|
|
logger.info("Video ready (%.1f MB)", video_size_mb)
|
|
|
|
detected_mime = _detect_video_mime_type(temp_video_path)
|
|
if not detected_mime:
|
|
raise ValueError(_unsupported_video_format(temp_video_path.suffix))
|
|
|
|
if video_size_bytes > _VIDEO_SIZE_WARN_BYTES:
|
|
logger.warning("Video is %.1f MB — may be slow or rejected", video_size_mb)
|
|
|
|
video_data_url = _video_to_base64_data_url(temp_video_path, mime_type=detected_mime)
|
|
if len(video_data_url) > _MAX_VIDEO_BASE64_BYTES:
|
|
raise ValueError(
|
|
f"Video too large for API: base64 payload is {len(video_data_url) / (1024 * 1024):.1f} MB "
|
|
f"(limit {_MAX_VIDEO_BASE64_BYTES / (1024 * 1024):.0f} MB). "
|
|
f"Compress or trim the video and retry."
|
|
)
|
|
|
|
debug_call_data["video_size_bytes"] = video_size_bytes
|
|
|
|
messages = [{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": user_prompt},
|
|
{"type": "video_url", "video_url": {"url": video_data_url}},
|
|
],
|
|
}]
|
|
|
|
vision_timeout, vision_temperature = _read_vision_call_settings(180.0, min_timeout=180.0)
|
|
call_kwargs = {
|
|
"task": "vision",
|
|
"messages": messages,
|
|
"temperature": vision_temperature,
|
|
"timeout": vision_timeout,
|
|
}
|
|
if model:
|
|
call_kwargs["model"] = model
|
|
|
|
analysis = await _call_vision_llm(call_kwargs, "Empty video response, retrying once")
|
|
|
|
analysis_length = len(analysis) if analysis else 0
|
|
logger.info("Video analysis completed (%s characters)", analysis_length)
|
|
|
|
result = {
|
|
"success": True,
|
|
"analysis": analysis or "There was a problem with the request and the video could not be analyzed.",
|
|
}
|
|
debug_call_data["success"] = True
|
|
debug_call_data["analysis_length"] = analysis_length
|
|
return _finish_analysis("video_analyze_tool", debug_call_data, result)
|
|
|
|
except Exception as e:
|
|
error_msg = f"Error analyzing video: {str(e)}"
|
|
logger.error("%s", error_msg, exc_info=True)
|
|
analysis = _classify_analysis_error(
|
|
e, _VIDEO_ERROR_RULES,
|
|
"There was a problem with the request and the video could not "
|
|
"be analyzed. Error: {e}",
|
|
)
|
|
debug_call_data["error"] = error_msg
|
|
return _finish_analysis("video_analyze_tool", debug_call_data, {
|
|
"success": False,
|
|
"error": error_msg,
|
|
"analysis": analysis,
|
|
})
|
|
|
|
finally:
|
|
if should_cleanup:
|
|
_cleanup_temp_media(temp_video_path, "video")
|
|
|
|
|
|
VIDEO_ANALYZE_SCHEMA = {
|
|
"name": "video_analyze",
|
|
"description": (
|
|
"Analyze a video from a URL or local file path using a multimodal AI model. "
|
|
"Sends the video to a video-capable model (e.g. Gemini) for understanding. "
|
|
"Use this for video files — for images, use vision_analyze instead. "
|
|
"Supports mp4, webm, mov, avi, mkv, mpeg formats. "
|
|
"Note: large videos (>20 MB) may be slow; max ~50 MB."
|
|
),
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"video_url": {
|
|
"type": "string",
|
|
"description": "Video URL (http/https) or local file path to analyze.",
|
|
},
|
|
"question": {
|
|
"type": "string",
|
|
"description": "Your specific question about the video. The AI will describe what happens in the video and answer your question.",
|
|
},
|
|
},
|
|
"required": ["video_url", "question"],
|
|
},
|
|
}
|
|
|
|
|
|
def _handle_video_analyze(args: Dict[str, Any], **kw: Any) -> Awaitable[str]:
|
|
video_url = args.get("video_url", "")
|
|
question = args.get("question", "")
|
|
full_prompt = (
|
|
"Fully describe and explain everything happening in this video, "
|
|
"including visual content, motion, audio cues, text overlays, and scene "
|
|
f"transitions. Then answer the following question:\n\n{question}"
|
|
)
|
|
model = _configured_aux_model(
|
|
("video", "vision"), ("AUXILIARY_VIDEO_MODEL", "AUXILIARY_VISION_MODEL"),
|
|
)
|
|
return video_analyze_tool(video_url, full_prompt, model, task_id=kw.get("task_id"))
|
|
|
|
|
|
registry.register(
|
|
name="video_analyze",
|
|
toolset="video",
|
|
schema=VIDEO_ANALYZE_SCHEMA,
|
|
handler=_handle_video_analyze,
|
|
check_fn=check_vision_requirements,
|
|
is_async=True,
|
|
emoji="🎬",
|
|
)
|