diff --git a/agent/agent_init.py b/agent/agent_init.py index 67fc38bf3f..938e13f612 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -529,6 +529,7 @@ def init_agent( read_preview_callback: callable = None, read_window_below_callback: callable = None, setup_mcp_callback: callable = None, + tour_callback: callable = None, step_callback: callable = None, stream_delta_callback: callable = None, interim_assistant_callback: callable = None, @@ -824,6 +825,7 @@ def init_agent( agent.read_preview_callback = read_preview_callback agent.read_window_below_callback = read_window_below_callback agent.setup_mcp_callback = setup_mcp_callback + agent.tour_callback = tour_callback agent.step_callback = step_callback agent.stream_delta_callback = stream_delta_callback agent.interim_assistant_callback = interim_assistant_callback diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 5d87894a65..79cfe58135 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -99,7 +99,7 @@ def _ra(): AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset( - {"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "read_window_below", "setup_mcp", "delegate_task"} + {"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} ) @@ -3236,6 +3236,23 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i ), next_args, ) + elif function_name == "tour": + def _execute(next_args: dict) -> Any: + from tools.tour_tool import tour_tool as _tour_tool + return _finish_agent_tool( + _tour_tool( + action=next_args.get("action", ""), + surface=next_args.get("surface"), + selector=next_args.get("selector"), + title=next_args.get("title"), + text=next_args.get("text"), + side=next_args.get("side"), + steps=next_args.get("steps"), + step_index=next_args.get("step_index"), + callback=getattr(agent, "tour_callback", None), + ), + next_args, + ) elif function_name == "setup_mcp": def _execute(next_args: dict) -> Any: from tools.setup_mcp_tool import setup_mcp_tool as _setup_mcp_tool diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 4fedb6dc7d..90226b02e1 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -2197,6 +2197,33 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): agent._vprint(f" {_get_cute_tool_message_impl('read_window_below', function_args, tool_duration, result=function_result)}") + elif function_name == "tour": + def _execute(next_args: dict) -> Any: + from tools.tour_tool import tour_tool as _tour_tool + return _tour_tool( + action=next_args.get("action", ""), + surface=next_args.get("surface"), + selector=next_args.get("selector"), + title=next_args.get("title"), + text=next_args.get("text"), + side=next_args.get("side"), + steps=next_args.get("steps"), + step_index=next_args.get("step_index"), + callback=getattr(agent, "tour_callback", None), + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('tour', function_args, tool_duration, result=function_result)}") elif function_name == "setup_mcp": def _execute(next_args: dict) -> Any: from tools.setup_mcp_tool import setup_mcp_tool as _setup_mcp_tool diff --git a/run_agent.py b/run_agent.py index b09262eae4..b4308f27a6 100644 --- a/run_agent.py +++ b/run_agent.py @@ -472,6 +472,7 @@ class AIAgent: read_preview_callback: callable = None, read_window_below_callback: callable = None, setup_mcp_callback: callable = None, + tour_callback: callable = None, step_callback: callable = None, stream_delta_callback: callable = None, interim_assistant_callback: callable = None, @@ -560,6 +561,7 @@ class AIAgent: read_preview_callback=read_preview_callback, read_window_below_callback=read_window_below_callback, setup_mcp_callback=setup_mcp_callback, + tour_callback=tour_callback, step_callback=step_callback, stream_delta_callback=stream_delta_callback, interim_assistant_callback=interim_assistant_callback, diff --git a/tools/tour_tool.py b/tools/tour_tool.py new file mode 100644 index 0000000000..9893020ba5 --- /dev/null +++ b/tools/tour_tool.py @@ -0,0 +1,202 @@ +#!/usr/bin/env python3 +"""Run a guided tour (highlight + narrate UI elements) in the Hermes desktop GUI. + +One generic tool, no baked-in tour definitions: the agent discovers what is on +screen (``action="targets"``), then highlights any element by CSS selector with +its own title/text — either one step at a time (``show``, agent-paced) or as a +full step list the user pages through with Next/Prev (``start``). + +Two surfaces share the same engine (driver.js in the renderer): + +- ``surface="app"`` — the Hermes desktop app's own DOM (tours of Hermes itself). +- ``surface="preview"`` — the page loaded in the in-app browser/preview pane + (tours of ANY web app, e.g. a project open via open_preview). + +Round-trips through the gateway's blocking-prompt bridge like ``read_preview``: +tui_gateway emits ``tour.request``, the renderer drives driver.js (injecting it +into the preview's webview when needed) and answers ``tour.respond`` with the +outcome, so the agent knows whether the selector matched. This module is just +schema + a thin dispatcher over the platform-injected callback. + +Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for +desktop-sourced sessions. +""" + +import json +from typing import Callable, Optional + +from tools.registry import registry, tool_error + +ACTIONS = ("targets", "show", "start", "next", "prev", "stop") +SURFACES = ("app", "preview") +SIDES = ("top", "right", "bottom", "left") + + +def tour_tool( + action: str = "", + surface: Optional[str] = None, + selector: Optional[str] = None, + title: Optional[str] = None, + text: Optional[str] = None, + side: Optional[str] = None, + steps: Optional[list] = None, + step_index: Optional[int] = None, + callback: Optional[Callable] = None, +) -> str: + """Dispatch one tour action to the desktop renderer and return its outcome.""" + if callback is None: + return tool_error("tour is only available in the Hermes desktop app.") + + verb = (action or "").strip().lower() + if verb not in ACTIONS: + return tool_error(f"action must be one of: {', '.join(ACTIONS)}.") + + where = (surface or "app").strip().lower() + if where not in SURFACES: + return tool_error(f"surface must be one of: {', '.join(SURFACES)}.") + + if side is not None and side not in SIDES: + return tool_error(f"side must be one of: {', '.join(SIDES)}.") + + # Every highlighted moment needs something to point at or something to say. + def _empty(step: dict) -> bool: + return not (step.get("selector") or step.get("title") or step.get("text")) + + if verb == "show" and _empty({"selector": selector, "title": title, "text": text}): + return tool_error("show needs a selector (and/or title/text for the popover).") + + if verb == "start": + if not isinstance(steps, list) or not steps: + return tool_error("start needs a non-empty steps array.") + for i, step in enumerate(steps): + if not isinstance(step, dict): + return tool_error(f"steps[{i}] must be an object.") + if _empty(step): + return tool_error(f"steps[{i}] needs a selector and/or title/text.") + + payload = { + key: val + for key, val in ( + ("action", verb), + ("surface", where), + ("selector", selector), + ("title", title), + ("text", text), + ("side", side), + ("steps", steps), + ("step_index", step_index), + ) + if val is not None + } + + try: + raw = callback(payload) + except Exception as exc: + return tool_error(f"Tour action failed: {exc}") + + if not raw: + return tool_error( + "The tour request timed out, or no GUI window answered. " + "For surface='preview' open a page in the preview pane first." + ) + + # The renderer answers with a JSON object; pass it through, else wrap it. + try: + return json.dumps(json.loads(raw), ensure_ascii=False) + except (TypeError, ValueError): + return json.dumps({"text": str(raw)}, ensure_ascii=False) + + +_STEP_SCHEMA = { + "type": "object", + "properties": { + "selector": { + "type": "string", + "description": "CSS selector of the element this step highlights. Omit for a centered narration-only step.", + }, + "title": {"type": "string", "description": "Popover title."}, + "text": {"type": "string", "description": "Popover body text."}, + "side": { + "type": "string", + "enum": list(SIDES), + "description": "Preferred popover side. Omit to auto-place.", + }, + }, +} + +TOUR_SCHEMA = { + "name": "tour", + "description": ( + "Give a live guided tour in the Hermes desktop GUI: dim the screen, " + "highlight an element, and attach a popover with your own title/text. " + "Works on two surfaces — 'app' (the Hermes app itself) and 'preview' " + "(whatever page is open in the in-app browser, so any web app can be " + "toured). ALWAYS call action='targets' first to discover what is on " + "screen instead of guessing selectors; each target reports " + "`stable: true` when its selector keys off identity (data-tour, id, " + "data-testid, aria-label) and survives a re-render — prefer those, and " + "re-scan if a selector stops matching. Then either narrate at your own " + "pace with action='show' (one highlight per call — replaces the " + "previous one; pair each with a chat message describing it), or hand " + "control to the user with action='start' + a steps array (driver.js " + "renders Next/Prev buttons; 'next'/'prev' also page it " + "programmatically). action='stop' clears the tour. Use when the user " + "asks how something works, where something is, or for a walkthrough of " + "an app or workflow." + ), + "parameters": { + "type": "object", + "properties": { + "action": { + "type": "string", + "enum": list(ACTIONS), + "description": "targets: list tourable elements. show: highlight one element. start: begin a multi-step user-paced tour. next/prev: page a started tour. stop: end the tour.", + }, + "surface": { + "type": "string", + "enum": list(SURFACES), + "description": "Where the tour runs: 'app' (Hermes desktop UI, default) or 'preview' (the page in the in-app browser pane).", + }, + "selector": { + "type": "string", + "description": "For show: CSS selector of the element to highlight (from action='targets', preferring a stable one). Omit for a centered narration popover.", + }, + "title": {"type": "string", "description": "For show: popover title."}, + "text": {"type": "string", "description": "For show: popover body text."}, + "side": { + "type": "string", + "enum": list(SIDES), + "description": "For show: preferred popover side. Omit to auto-place.", + }, + "steps": { + "type": "array", + "items": _STEP_SCHEMA, + "description": "For start: the ordered tour steps.", + }, + "step_index": { + "type": "integer", + "description": "For start: 0-indexed step to begin at (default 0).", + }, + }, + "required": ["action"], + }, +} + + +registry.register( + name="tour", + toolset="desktop_ui", + schema=TOUR_SCHEMA, + handler=lambda args, **kw: tour_tool( + action=args.get("action", ""), + surface=args.get("surface"), + selector=args.get("selector"), + title=args.get("title"), + text=args.get("text"), + side=args.get("side"), + steps=args.get("steps"), + step_index=args.get("step_index"), + callback=kw.get("callback"), + ), + emoji="🧭", +) diff --git a/toolsets.py b/toolsets.py index 8bb86024f5..18a6bf4227 100644 --- a/toolsets.py +++ b/toolsets.py @@ -279,7 +279,7 @@ TOOLSETS = { "open_preview", "read_preview", "read_window_below", "focus_pane", "react_to_message", - "setup_mcp", + "setup_mcp", "tour", ], "includes": [] }, diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index bb8f329db3..115fd14047 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -1432,6 +1432,16 @@ def _(rid, params: dict) -> dict: return _respond(rid, params, "text", allow_expired=True) +@method("tour.respond") +def _(rid, params: dict) -> dict: + # `text` is a JSON string with the tour action's outcome (tour tool) — + # matched targets, the active step, or an error naming the bad selector. + # allow_expired=True for the same reason as terminal.read: a preview tour + # injecting driver.js into a slow page can lose the race with the tool's + # bounded wait. + return _respond(rid, params, "text", allow_expired=True) + + @method("mcp.setup.respond") def _(rid, params: dict) -> dict: # `result` is a JSON string of the setup card's outcome ({status, server, diff --git a/tui_gateway/server.py b/tui_gateway/server.py index d024c7e411..e7b09ab8c2 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3551,6 +3551,7 @@ def _block( "preview.read.request", "window.read.request", "mcp.setup.request", + "tour.request", }: _emit( f"{event.removesuffix('.request')}.expire", @@ -6307,6 +6308,17 @@ def _agent_cbs(sid: str) -> dict: {"server": server, "action": action, "reason": reason}, timeout=600, ), + # tour tool (desktop GUI): the renderer drives driver.js — highlighting + # elements in the app's own DOM or injecting the engine into the + # preview pane's webview — and answers tour.respond with the outcome + # (did the selector match, which step is active). Generous timeout: a + # preview tour's first action loads the engine into a live page. + "tour_callback": lambda payload: _block( + "tour.request", + sid, + dict(payload), + timeout=45, + ), } # Interim assistant commentary (text alongside tool calls, or the attempted