From e16ad33a9d2447c823ab3bc29cacea9c1987cc92 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 08:26:24 -0700 Subject: [PATCH] =?UTF-8?q?feat(tool-search):=20core-tool=20deferral=20?= =?UTF-8?q?=E2=80=94=20curated=2019-tool=20set=20behind=20the=20bridge=20b?= =?UTF-8?q?y=20default;=20renames=20todo=5Flist/cronjob=5Fmanage/process?= =?UTF-8?q?=5Fmanage/gui=5Ftour/show=5Ftip=20with=20legacy=20aliases=20(13?= =?UTF-8?q?.4K=20->=206.9K=20desktop=20schemas,=20-49%)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- agent/agent_runtime_helpers.py | 6 +-- agent/coding_context.py | 2 +- agent/context_compressor.py | 6 +-- agent/conversation_loop.py | 2 +- agent/display.py | 16 +++--- agent/tool_executor.py | 16 ++++-- agent/tool_guardrails.py | 8 +-- agent/turn_summary.py | 2 +- hermes_cli/tools_config.py | 2 +- model_tools.py | 24 +++++++-- tests/tools/test_tool_search.py | 42 ++++++++++++--- tools/cronjob_tools.py | 4 +- tools/delegate_tool.py | 2 +- tools/process_registry.py | 4 +- tools/tip_tool.py | 4 +- tools/todo_tool.py | 4 +- tools/tool_search.py | 91 ++++++++++++++++++++++++++------- tools/tour_tool.py | 4 +- toolsets.py | 30 +++++------ tui_gateway/server.py | 2 +- 20 files changed, 189 insertions(+), 82 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index dd30d99ed5..2127e6b165 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -108,7 +108,7 @@ def _ra(): AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset( - {"todo", "session_search", "memory", "clarify", "read_terminal", "desktop_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} + {"todo_list", "session_search", "memory", "clarify", "read_terminal", "desktop_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} ) @@ -3401,7 +3401,7 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i pass return result - if function_name == "todo": + if function_name == "todo_list": def _execute(next_args: dict) -> Any: from tools.todo_tool import todo_tool as _todo_tool return _finish_agent_tool( @@ -3543,7 +3543,7 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i ), next_args, ) - elif function_name == "tour": + elif function_name == "gui_tour": def _execute(next_args: dict) -> Any: from tools.tour_tool import tour_tool as _tour_tool return _finish_agent_tool( diff --git a/agent/coding_context.py b/agent/coding_context.py index 333de26b05..f4928af245 100644 --- a/agent/coding_context.py +++ b/agent/coding_context.py @@ -547,7 +547,7 @@ class RuntimeMode: trailing: list[str] = [] if self.profile.guidance: brief = self.profile.guidance - if valid_tool_names is not None and "todo" not in valid_tool_names: + if valid_tool_names is not None and "todo_list" not in valid_tool_names: brief = brief.replace( "- Track multi-step work with `todo`. Reference code as " "`path:line` instead of pasting whole files.", diff --git a/agent/context_compressor.py b/agent/context_compressor.py index c8eb489496..955201de80 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -2172,7 +2172,7 @@ def _summarize_tool_result_unguarded(tool_name: str, tool_args: str, tool_conten target = args.get("target", "?") return f"[memory] {action} on {target}" - if tool_name == "todo": + if tool_name == "todo_list": return "[todo] updated task list" if tool_name == "clarify": @@ -2225,11 +2225,11 @@ def _summarize_tool_result_unguarded(tool_name: str, tool_args: str, tool_conten if tool_name == "text_to_speech": return f"[text_to_speech] generated audio ({content_len:,} chars)" - if tool_name == "cronjob": + if tool_name == "cronjob_manage": action = args.get("action", "?") return f"[cronjob] {action}" - if tool_name == "process": + if tool_name == "process_manage": action = args.get("action", "?") sid = args.get("session_id", "?") return f"[process] {action} session={sid}" diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index ceb7ce16aa..47f349bf88 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -7353,7 +7353,7 @@ def run_conversation( # This classification is needed regardless of whether the turn has visible content, # because a substantive tool-only turn must invalidate any older housekeeping fallback. _HOUSEKEEPING_TOOLS = frozenset({ - "memory", "todo", "skill_manage", "session_search", + "memory", "todo_list", "skill_manage", "session_search", }) _all_housekeeping = all( tc.function.name in _HOUSEKEEPING_TOOLS diff --git a/agent/display.py b/agent/display.py index 2880cecccb..7a5dfc47a4 100644 --- a/agent/display.py +++ b/agent/display.py @@ -462,7 +462,7 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - "image_generate": "prompt", "text_to_speech": "text", "vision_analyze": "question", "skill_view": "name", "skills_list": "category", - "cronjob": "action", + "cronjob_manage": "action", "execute_code": "code", "browser_exec": "code", "delegate_task": "goal", "clarify": "question", "skill_manage": "name", } @@ -496,7 +496,7 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - preview = _oneline(str(goal)) return _truncate_preview(preview, max_len) if preview else None - if tool_name == "process": + if tool_name == "process_manage": action = args.get("action", "") sid = args.get("session_id", "") data = args.get("data", "") @@ -511,7 +511,7 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - parts = [p for p in parts if p] return " ".join(parts) if parts else None - if tool_name == "todo": + if tool_name == "todo_list": todos_arg = args.get("todos") merge = args.get("merge", False) if todos_arg is None: @@ -657,10 +657,10 @@ _TOOL_VERBS: dict[str, str] = { "skills_list": "Listing skills", "skill_manage": "Updating skill", "delegate_task": "Delegating", - "cronjob": "Scheduling", + "cronjob_manage": "Scheduling", "clarify": "Asking", "memory": "Updating memory", - "todo": "Updating tasks", + "todo_list": "Updating tasks", } # Verbs that read better without the raw argument preview appended. @@ -1433,7 +1433,7 @@ def _get_cute_tool_message( return _wrap(f"┊ 📄 fetch pages {dur}") if tool_name == "terminal": return _wrap(f"┊ 💻 $ {_trunc(build_tool_preview(tool_name, args) or args.get('command', ''), 42)} {dur}") - if tool_name == "process": + if tool_name == "process_manage": action = args.get("action", "?") sid = args.get("session_id", "")[:12] labels = {"list": "ls processes", "poll": f"poll {sid}", "log": f"log {sid}", @@ -1473,7 +1473,7 @@ def _get_cute_tool_message( return _wrap(f"┊ 🖼️ images extracting {dur}") if tool_name == "browser_vision": return _wrap(f"┊ 👁️ vision analyzing page {dur}") - if tool_name == "todo": + if tool_name == "todo_list": todos_arg = args.get("todos") merge = args.get("merge", False) # Parse result for completion progress @@ -1532,7 +1532,7 @@ def _get_cute_tool_message( return _wrap(f"┊ 👁️ vision {_trunc(args.get('question', ''), 30)} {dur}") if tool_name == "send_message": return _wrap(f"┊ 📨 send {args.get('target', '?')}: \"{_trunc(args.get('message', ''), 25)}\" {dur}") - if tool_name == "cronjob": + if tool_name == "cronjob_manage": action = args.get("action", "?") if action == "create": skills = args.get("skills") or ([] if not args.get("skill") else [args.get("skill")]) diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 5ee9da8444..de0df8c068 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -1151,6 +1151,10 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe parsed_calls = [] for tool_call in tool_calls: function_name = tool_call.function.name + # Legacy tool-name aliases (2026-08 renames) — map BEFORE the + # agent-loop branches (todo_list etc. dispatch above the registry). + from model_tools import _LEGACY_TOOL_ALIASES as _lta + function_name = _lta.get(function_name, function_name) function_args, malformed_args_result = _parse_tool_arguments( tool_call.function.arguments @@ -2007,6 +2011,10 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe break function_name = tool_call.function.name + # Legacy tool-name aliases (2026-08 renames) — map BEFORE the + # agent-loop branches (todo_list etc. dispatch above the registry). + from model_tools import _LEGACY_TOOL_ALIASES as _lta + function_name = _lta.get(function_name, function_name) function_args, malformed_args_result = _parse_tool_arguments( tool_call.function.arguments @@ -2081,7 +2089,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_start_time = time.time() - if function_name == "todo": + if function_name == "todo_list": def _execute(next_args: dict) -> Any: from tools.todo_tool import todo_tool as _todo_tool return _todo_tool( @@ -2101,7 +2109,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe )) tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): - agent._vprint(f" {_get_cute_tool_message_impl('todo', function_args, tool_duration, result=function_result)}") + agent._vprint(f" {_get_cute_tool_message_impl('todo_list', function_args, tool_duration, result=function_result)}") elif function_name == "message_agent": # Bot Mode teammate DM (tools/bot_mode_dm.py) — injected, not # registered: only a canonical Bot Chat session carries the @@ -2336,7 +2344,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): agent._vprint(f" {_get_cute_tool_message_impl('read_window_below', function_args, tool_duration, result=function_result)}") - elif function_name == "tour": + elif function_name == "gui_tour": def _execute(next_args: dict) -> Any: from tools.tour_tool import tour_tool as _tour_tool return _tour_tool( @@ -2362,7 +2370,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe )) tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): - agent._vprint(f" {_get_cute_tool_message_impl('tour', function_args, tool_duration, result=function_result)}") + agent._vprint(f" {_get_cute_tool_message_impl('gui_tour', function_args, tool_duration, result=function_result)}") elif function_name == "setup_mcp": def _execute(next_args: dict) -> Any: from tools.setup_mcp_tool import setup_mcp_tool as _setup_mcp_tool diff --git a/agent/tool_guardrails.py b/agent/tool_guardrails.py index ff30ee516c..de08b427c6 100644 --- a/agent/tool_guardrails.py +++ b/agent/tool_guardrails.py @@ -44,7 +44,7 @@ MUTATING_TOOL_NAMES = frozenset( "execute_code", "write_file", "patch", - "todo", + "todo_list", "memory", "skill_manage", "browser_click", @@ -53,9 +53,9 @@ MUTATING_TOOL_NAMES = frozenset( "browser_scroll", "browser_navigate", "send_message", - "cronjob", + "cronjob_manage", "delegate_task", - "process", + "process_manage", } ) @@ -67,7 +67,7 @@ MUTATING_TOOL_NAMES = frozenset( # unannotated. STALL_GUARD_REPEATABLE_TOOLS = frozenset( { - "process", + "process_manage", } ) diff --git a/agent/turn_summary.py b/agent/turn_summary.py index f4440afb50..5953629eb9 100644 --- a/agent/turn_summary.py +++ b/agent/turn_summary.py @@ -69,7 +69,7 @@ _VERB_GROUPS: dict[str, tuple[str, str, str]] = { "skill_view": ("read", "skill", "skills"), "skill_manage": ("updated", "skill", "skills"), "skills_list": ("listed skills", "time", "times"), - "todo": ("updated", "task list", "task lists"), + "todo_list": ("updated", "task list", "task lists"), "delegate_task": ("delegated", "task", "tasks"), "memory": ("updated", "memory", "memories"), } diff --git a/hermes_cli/tools_config.py b/hermes_cli/tools_config.py index 018178d916..aaae786433 100644 --- a/hermes_cli/tools_config.py +++ b/hermes_cli/tools_config.py @@ -107,7 +107,7 @@ CONFIGURABLE_TOOLSETS = [ ("tts", "🔊 Text-to-Speech", "text_to_speech"), ("stt", "🎙️ Speech-to-Text", "voice transcription (gateway voice messages + voice mode)"), ("skills", "📚 Skills", "list, view, manage"), - ("todo", "📋 Task Planning", "todo"), + ("todo", "📋 Task Planning", "todo_list"), ("memory", "💾 Memory", "persistent memory across sessions"), ("context_engine", "🧩 Context Engine", "runtime tools from the active context engine"), ("session_search", "🔎 Session Search", "search past conversations"), diff --git a/model_tools.py b/model_tools.py index 20ce327a2a..0ebd572624 100644 --- a/model_tools.py +++ b/model_tools.py @@ -279,7 +279,7 @@ _LEGACY_TOOLSET_MAP = { "browser_press", "browser_get_images", "browser_vision", "browser_console" ], - "cronjob_tools": ["cronjob"], + "cronjob_tools": ["cronjob_manage"], "file_tools": ["read_file", "write_file", "patch", "search_files"], "tts_tools": ["text_to_speech"], } @@ -602,7 +602,7 @@ def _compute_tool_definitions( # Same session-level seam as the browser_exec gate above. if "delegate_task" in available_tool_names: blocked_present = [ - t for t in ("clarify", "memory", "cronjob") if t in available_tool_names + t for t in ("clarify", "memory", "cronjob_manage") if t in available_tool_names ] if len(blocked_present) < 3: full_offvariant = "delegate_task, clarify, memory, or cronjob" @@ -788,7 +788,18 @@ def _resolve_active_context_length() -> int: # because they need agent-level state (TodoStore, MemoryStore, etc.). # The registry still holds their schemas; dispatch just returns a stub error # so if something slips through, the LLM sees a sensible message. -_AGENT_LOOP_TOOLS = {"todo", "memory", "session_search", "delegate_task"} +_AGENT_LOOP_TOOLS = {"todo_list", "memory", "session_search", "delegate_task"} + +# Legacy tool-name aliases (2026-08 renames): accepted at every dispatch seam +# (handle_function_call + both executors) so old sessions and saved prompts +# keep working; schemas only advertise the new names. +_LEGACY_TOOL_ALIASES = { + "todo": "todo_list", + "cronjob": "cronjob_manage", + "process": "process_manage", + "tour": "gui_tour", + "tip": "show_tip", +} _READ_SEARCH_TOOLS = {"read_file", "search_files"} @@ -1284,6 +1295,13 @@ def handle_function_call( function_args = {} _tool_middleware_trace = list(tool_request_middleware_trace or []) + # ── Legacy tool-name aliases (2026-08 renames) ──────────────────── + # Old sessions resuming mid-conversation (and users' muscle memory in + # saved skills/cron prompts) still emit the pre-rename names. Alias at + # the dispatch seam so every replay keeps working; new schemas only + # advertise the new names, so fresh sessions never see the old ones. + function_name = _LEGACY_TOOL_ALIASES.get(function_name, function_name) + # ── Tool Search bridge dispatch ────────────────────────────────── # tool_search and tool_describe are pure catalog reads — handle them # inline. tool_call is unwrapped to the underlying tool so that every diff --git a/tests/tools/test_tool_search.py b/tests/tools/test_tool_search.py index 3e2c80e8ad..902902291d 100644 --- a/tests/tools/test_tool_search.py +++ b/tests/tools/test_tool_search.py @@ -96,7 +96,30 @@ class TestClassification: assert not is_deferrable_tool_name(name), name assert name not in _HERMES_CORE_TOOLS - def test_gui_surface_alone_does_not_activate_the_bridge(self): + def test_gui_surface_defers_by_default(self): + """2026-08 core-deferral reversal: the curated defer set (GUI surface + included) hides behind the bridge BY DEFAULT. project tools not in + the defer set stay direct.""" + from tools.registry import discover_builtin_tools + from tools.tool_search import ToolSearchConfig, assemble_tool_defs + + discover_builtin_tools() + assembled = assemble_tool_defs( + [_td(name, f"GUI {name}") for name in + {"read_window_below", "apply_layout", "project_list"}], + context_length=200_000, + config=ToolSearchConfig.from_raw({"enabled": "on"}), + ) + assert assembled.activated + names = {td["function"]["name"] for td in assembled.tool_defs} + assert "read_window_below" not in names + assert "apply_layout" not in names + # project_list is NOT in the curated defer set → stays direct. + assert "project_list" in names + + def test_defer_override_restores_legacy_direct_gui(self): + """tools.tool_search.defer: [] restores the everything-eager legacy: + GUI tools alone no longer activate the bridge.""" from tools.registry import discover_builtin_tools from tools.tool_search import ToolSearchConfig, assemble_tool_defs @@ -105,14 +128,15 @@ class TestClassification: assembled = assemble_tool_defs( [_td(name, f"GUI {name}") for name in names], context_length=200_000, - config=ToolSearchConfig.from_raw({"enabled": "on"}), + config=ToolSearchConfig.from_raw({"enabled": "on", "defer": []}), ) assert not assembled.activated assert {td["function"]["name"] for td in assembled.tool_defs} == names - def test_gui_surface_stays_direct_when_mcp_activates_the_bridge(self): - """MCP/plugin tools turn Tool Search on; the session's GUI tools stay - in the model-facing array so HUD can still name read_window_below.""" + def test_core_working_set_never_defers_even_with_mcp_active(self): + """The bridge activates for MCP, but working-set core tools (terminal, + files, memory...) stay direct — the deferral set is the CURATED list, + not all of core.""" from tools.registry import discover_builtin_tools, registry from tools.tool_search import ( BRIDGE_TOOL_NAMES, @@ -131,8 +155,8 @@ class TestClassification: assembled = assemble_tool_defs( [ - _td("read_window_below", "Identify the window below"), - _td("apply_layout", "Apply a layout preset"), + _td("terminal", "Run a command"), + _td("memory", "Persistent memory"), _td("computer_use", "Drive the OS"), _td(mcp_name, "Deferred MCP capability"), ], @@ -144,7 +168,9 @@ class TestClassification: assert assembled.activated assert mcp_name not in names assert BRIDGE_TOOL_NAMES <= names - assert {"read_window_below", "apply_layout", "computer_use"} <= names + assert {"terminal", "memory"} <= names + # computer_use IS in the curated defer set → behind the bridge. + assert "computer_use" not in names def test_unknown_tool_not_deferrable(self): """Defensive: a tool name we cannot resolve to a registry entry must diff --git a/tools/cronjob_tools.py b/tools/cronjob_tools.py index 5fd0619458..a551b7cb2e 100644 --- a/tools/cronjob_tools.py +++ b/tools/cronjob_tools.py @@ -1949,7 +1949,7 @@ def cronjob( CRONJOB_SCHEMA = { - "name": "cronjob", + "name": "cronjob_manage", "description": """Manage scheduled cron jobs: action='create' schedules a job from a prompt and/or skills; 'list' inspects jobs; 'update'/'pause'/'resume'/'remove' manage one by job_id (always list first — never guess job IDs); 'run' fires a job immediately in the BACKGROUND (returns a handle at once, outcome re-enters the conversation when done — do not wait or poll; optional 'prompt' adds transient context for that fire only). Jobs run in a fresh session with no current-chat context, so prompts must be self-contained, and the agent's FINAL RESPONSE is what gets delivered — cron runs are autonomous and cannot ask questions. Prefer updating an existing job over creating near-duplicates.""", @@ -2098,7 +2098,7 @@ def _cronjob_handler(args, **kw): registry.register( - name="cronjob", + name="cronjob_manage", toolset="cronjob", schema=CRONJOB_SCHEMA, handler=_cronjob_handler, diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index d3c2a2dda8..3aace8bbac 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -53,7 +53,7 @@ DELEGATE_BLOCKED_TOOLS = frozenset( "clarify", # no user interaction "memory", # no writes to shared MEMORY.md "send_message", # no cross-platform side effects - "cronjob", # no scheduling more work in the parent's name + "cronjob_manage", # no scheduling more work in the parent's name ] ) diff --git a/tools/process_registry.py b/tools/process_registry.py index bca6f91a05..409175bf99 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -3242,7 +3242,7 @@ def format_process_notification(evt: dict) -> "str | None": from tools.registry import registry, tool_error PROCESS_SCHEMA = { - "name": "process", + "name": "process_manage", # Dieted (#95681): the action enum names the verbs; the description # keeps only non-obvious semantics. write-vs-submit is the tool's one # real trap (a lone \n on a Windows PTY is not a line terminator) — @@ -3363,7 +3363,7 @@ def _handle_process(args, **kw): registry.register( - name="process", + name="process_manage", toolset="terminal", schema=PROCESS_SCHEMA, handler=_handle_process, diff --git a/tools/tip_tool.py b/tools/tip_tool.py index ab80092f17..35d4908ba1 100644 --- a/tools/tip_tool.py +++ b/tools/tip_tool.py @@ -60,7 +60,7 @@ def tip_tool(text: str, selector: str, title: str = "", side: str = "") -> str: TIP_SCHEMA = { - "name": "tip", + "name": "show_tip", "description": ( "Point at one thing in the desktop UI with a small arrow bubble (no " "dimming, no tour chrome) — for when a sentence is clearer with a " @@ -96,7 +96,7 @@ TIP_SCHEMA = { registry.register( - name="tip", + name="show_tip", toolset="desktop_ui", schema=TIP_SCHEMA, handler=lambda args, **kw: tip_tool( diff --git a/tools/todo_tool.py b/tools/todo_tool.py index 213f65b28d..c90d14fcd5 100644 --- a/tools/todo_tool.py +++ b/tools/todo_tool.py @@ -352,7 +352,7 @@ def check_todo_requirements() -> bool: # static tool schema (cached, never changes mid-conversation). TODO_SCHEMA = { - "name": "todo", + "name": "todo_list", # Dieted (#95681): the item shape and merge semantics live ONLY in the # parameter schema below — the description teaches behavior, not # structure the params already define. @@ -414,7 +414,7 @@ TODO_SCHEMA = { from tools.registry import registry, tool_error registry.register( - name="todo", + name="todo_list", toolset="todo", schema=TODO_SCHEMA, handler=lambda args, **kw: todo_tool( diff --git a/tools/tool_search.py b/tools/tool_search.py index 66f1b198c5..92b4afef6b 100644 --- a/tools/tool_search.py +++ b/tools/tool_search.py @@ -110,6 +110,14 @@ class ToolSearchConfig: # Absolute cap on the embedded listing, regardless of context size. # Effective budget = min(listing_max_tokens, threshold_pct% of context). listing_max_tokens: int = 4000 + # Core/GUI tool names deferred behind the bridge. None = use the curated + # default (_DEFAULT_DEFERRED_TOOLS); an explicit list from config + # replaces the default wholesale ([] = defer no core tools — legacy). + defer_tools: Optional[frozenset] = None + + @property + def effective_defer_tools(self) -> frozenset: + return _DEFAULT_DEFERRED_TOOLS if self.defer_tools is None else self.defer_tools @classmethod def from_raw(cls, raw: Any) -> "ToolSearchConfig": @@ -159,6 +167,14 @@ class ToolSearchConfig: listing = "auto" listing_max_tokens = max(200, min(60000, _safe_int(raw.get("listing_max_tokens"), 4000))) + defer_raw = raw.get("defer") + if isinstance(defer_raw, (list, tuple, set)): + defer_tools = frozenset( + str(n).strip() for n in defer_raw if str(n).strip() + ) + else: + defer_tools = None # curated default + return cls( enabled=enabled, threshold_pct=threshold_pct, @@ -166,6 +182,7 @@ class ToolSearchConfig: max_search_limit=max_search_limit, listing=listing, listing_max_tokens=listing_max_tokens, + defer_tools=defer_tools, ) @@ -230,21 +247,46 @@ def _core_tool_names() -> frozenset[str]: # Session-gated GUI toolsets. Off ``_HERMES_CORE_TOOLS`` so non-GUI clients -# never pay their schema; once a session enables them they stay direct. +# never pay their schema; once a session enables them they stay direct +# UNLESS the deferral list (below) names them. _DIRECT_SURFACE_TOOLSETS = frozenset({"desktop_ui", "project"}) +# Core-tool deferral (2026-08, maintainer-directed): the curated set of +# event-triggered tools that hide behind the bridge BY DEFAULT. These are +# tools a session reaches for when something specific happens (user asks +# for a tour / a cron job / a screenshot / a clarification), not tools in +# the every-turn working set — so a catalog stub is enough to find them. +# Config override: ``tools.tool_search.defer`` (list of tool names); +# ``[]`` restores the legacy everything-eager behavior, any other list +# replaces this default wholesale. Names here are POST-rename. +_DEFAULT_DEFERRED_TOOLS = frozenset({ + "computer_use", "session_search", "clarify", "image_generate", + "todo_list", "process_manage", "cronjob_manage", + # Desktop GUI surface (desktop_ui + project toolsets) + "drive_preview", "gui_tour", "desktop_preview", "annotate_preview", + "show_tip", "setup_mcp", "desktop_project", "close_terminal", + "apply_layout", "read_terminal", "read_window_below", "focus_pane", +}) -def is_deferrable_tool_name(name: str) -> bool: + +def is_deferrable_tool_name(name: str, defer_tools: Optional[frozenset] = None) -> bool: """Return True if a tool with this name is *eligible* for deferral. - A tool is deferrable iff it is registered with an MCP toolset prefix - OR it is neither in ``_HERMES_CORE_TOOLS`` nor a session-gated GUI - surface toolset. Core and direct surface tools are never deferred even - when their toolset is technically plugin-provided (this protects - against accidental shadowing). + A tool is deferrable iff: + * it is named in ``defer_tools`` (the maintainer-curated core-deferral + set, or the user's ``tools.tool_search.defer`` override) — this is + the 2026-08 revision of the old "core never defers" rule: core tools + in the WORKING set (terminal, files, memory, ...) still never defer, + but the curated event-triggered set (computer_use, clarify, the GUI + surface, ...) hides behind the bridge by default; OR + * it is registered with an MCP toolset prefix; OR + * it is neither in ``_HERMES_CORE_TOOLS`` nor a session-gated GUI + surface toolset (plugin tools). """ if name in BRIDGE_TOOL_NAMES: return False + if defer_tools is not None and name in defer_tools: + return True if name in _core_tool_names(): return False # Check registry toolset for MCP prefix. @@ -265,6 +307,7 @@ def is_deferrable_tool_name(name: str) -> bool: def _describe_classification( name: str, + defer_tools: Optional[frozenset] = None, ) -> Literal["available", "not_found", "not_deferrable"]: """Classify a describe name without treating unknown names as errors.""" try: @@ -274,6 +317,8 @@ def _describe_classification( return "not_found" if entry is None: return "not_found" + if defer_tools is not None and name in defer_tools: + return "available" if ( name in BRIDGE_TOOL_NAMES or name in _core_tool_names() @@ -283,12 +328,15 @@ def _describe_classification( return "available" -def classify_tools(tool_defs: List[Dict[str, Any]]) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: +def classify_tools( + tool_defs: List[Dict[str, Any]], + defer_tools: Optional[frozenset] = None, +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: """Split a tool-defs list into (visible, deferrable). - ``visible`` retains every tool that must stay in the model-facing array: - every core tool, every session-gated GUI surface tool, plus any tool we - can't classify. ``deferrable`` is the candidate set for catalog entry. + ``visible`` retains every tool that must stay in the model-facing array. + ``deferrable`` is the candidate set for catalog entry — MCP/plugin tools + plus any core/GUI tool named in ``defer_tools``. """ visible: List[Dict[str, Any]] = [] deferrable: List[Dict[str, Any]] = [] @@ -299,7 +347,7 @@ def classify_tools(tool_defs: List[Dict[str, Any]]) -> Tuple[List[Dict[str, Any] # Should never happen — bridge tools are added after classification — # but be defensive. continue - if is_deferrable_tool_name(name): + if is_deferrable_tool_name(name, defer_tools): deferrable.append(td) else: visible.append(td) @@ -927,7 +975,7 @@ def assemble_tool_defs( incoming = [td for td in tool_defs if (td.get("function") or {}).get("name") not in BRIDGE_TOOL_NAMES] - visible, deferrable = classify_tools(incoming) + visible, deferrable = classify_tools(incoming, config.effective_defer_tools) if not deferrable: return AssemblyResult(tool_defs=incoming, activated=False) @@ -1078,7 +1126,9 @@ def dispatch_tool_search(args: Dict[str, Any], else: limit = max(1, min(config.max_search_limit, _safe_int(raw_limit, config.search_default_limit))) - _, deferrable = classify_tools(current_tool_defs) + _, deferrable = classify_tools( + current_tool_defs, load_config_readonly().effective_defer_tools + ) catalog = build_catalog(deferrable) results: List[Dict[str, Any]] = [] @@ -1151,7 +1201,9 @@ def dispatch_tool_describe(args: Dict[str, Any], "Retry with fewer names per call." ) - _, deferrable = classify_tools(current_tool_defs) + _, deferrable = classify_tools( + current_tool_defs, load_config_readonly().effective_defer_tools + ) by_name: Dict[str, Dict[str, Any]] = {} for td in deferrable: fn = td.get("function") or {} @@ -1168,7 +1220,9 @@ def dispatch_tool_describe(args: Dict[str, Any], "description": fn.get("description", ""), "parameters": fn.get("parameters", {}), } - elif _describe_classification(name) == "not_deferrable": + elif _describe_classification( + name, load_config_readonly().effective_defer_tools + ) == "not_deferrable": errors[name] = ( f"'{name}' is not a deferrable tool. If you see it in the tools list " "already, call it directly; otherwise check the spelling against tool_search." @@ -1198,9 +1252,10 @@ def scoped_deferrable_names(tool_defs: List[Dict[str, Any]]) -> frozenset[str]: an out-of-scope tool via the bridge. """ names: set[str] = set() + defer_tools = load_config_readonly().effective_defer_tools for td in tool_defs: name = (td.get("function") or {}).get("name", "") - if name and is_deferrable_tool_name(name): + if name and is_deferrable_tool_name(name, defer_tools): names.add(name) return frozenset(names) @@ -1283,7 +1338,7 @@ def resolve_underlying_call(args: Dict[str, Any]) -> Tuple[Optional[str], Dict[s return None, {}, f"tool_call 'arguments' is not valid JSON: {e}" if not isinstance(raw_args, dict): return None, {}, "tool_call 'arguments' must be an object" - if not is_deferrable_tool_name(name): + if not is_deferrable_tool_name(name, load_config_readonly().effective_defer_tools): return None, {}, ( f"'{name}' is not a deferrable tool. If it appears in the model-facing tools " "list already, call it directly instead of via tool_call." diff --git a/tools/tour_tool.py b/tools/tour_tool.py index 2ccff61002..9a1632b1b4 100644 --- a/tools/tour_tool.py +++ b/tools/tour_tool.py @@ -125,7 +125,7 @@ _STEP_SCHEMA = { } TOUR_SCHEMA = { - "name": "tour", + "name": "gui_tour", # Dieted (#95681): targets-first flow + stable-selector preference kept # (pre-effect: skipping them means guessed selectors on re-rendering UI). "description": ( @@ -180,7 +180,7 @@ TOUR_SCHEMA = { registry.register( - name="tour", + name="gui_tour", toolset="desktop_ui", schema=TOUR_SCHEMA, handler=lambda args, **kw: tour_tool( diff --git a/toolsets.py b/toolsets.py index 2ddd6461c1..c8b76dafce 100644 --- a/toolsets.py +++ b/toolsets.py @@ -32,7 +32,7 @@ _HERMES_CORE_TOOLS = [ # Web "web_search", "web_extract", # Terminal + process management - "terminal", "process", + "terminal", "process_manage", # NOTE: the desktop GUI affordances (read_terminal, open_preview, …) are # deliberately NOT here, for the same reason as the `project` tools below: # they only work where a GUI renderer can answer them. They live in the @@ -56,7 +56,7 @@ _HERMES_CORE_TOOLS = [ # Text-to-speech "text_to_speech", # Planning & memory - "todo", "memory", + "todo_list", "memory", # NOTE: the desktop Project tools (project_list/create/switch) are # deliberately NOT here. They only make sense where a GUI can follow the # move, so they live in the `project` toolset and are enabled solely by the @@ -69,7 +69,7 @@ _HERMES_CORE_TOOLS = [ # Code execution + delegation "execute_code", "delegate_task", # Cronjob management - "cronjob", + "cronjob_manage", # Home Assistant smart home control (gated on HASS_TOKEN via check_fn) "ha_list_entities", "ha_get_state", "ha_list_services", "ha_call_service", # Kanban multi-agent coordination — only in schema when the agent is @@ -169,7 +169,7 @@ TOOLSETS = { "terminal": { "description": "Terminal/command execution and process management tools", - "tools": ["terminal", "process"], + "tools": ["terminal", "process_manage"], "includes": [] }, @@ -193,7 +193,7 @@ TOOLSETS = { "cronjob": { "description": "Cronjob management tool - create, list, update, pause, resume, remove, and trigger scheduled tasks", - "tools": ["cronjob"], + "tools": ["cronjob_manage"], "includes": [] }, @@ -212,7 +212,7 @@ TOOLSETS = { "todo": { "description": "Task planning and tracking for multi-step work", - "tools": ["todo"], + "tools": ["todo_list"], "includes": [] }, @@ -256,7 +256,7 @@ TOOLSETS = { "desktop_preview", "drive_preview", "annotate_preview", "read_window_below", "focus_pane", "react_to_message", - "setup_mcp", "tour", "tip", + "setup_mcp", "gui_tour", "show_tip", ], "includes": [] }, @@ -363,7 +363,7 @@ TOOLSETS = { "debugging": { "description": "Debugging and troubleshooting toolkit", - "tools": ["terminal", "process"], + "tools": ["terminal", "process_manage"], "includes": ["web", "file"] # For searching error messages and solutions, and file operations }, @@ -386,7 +386,7 @@ TOOLSETS = { "description": "Coding-focused toolset: files, terminal, search, web docs, skills, todo, delegate, vision, browser", "tools": [ "web_search", "web_extract", - "terminal", "process", + "terminal", "process_manage", "read_file", "write_file", "patch", "search_files", "vision_analyze", "skills_list", "skill_view", "skill_manage", @@ -395,7 +395,7 @@ TOOLSETS = { "browser_press", "browser_get_images", "browser_vision", "browser_console", "browser_cdp", "browser_dialog", "browser_exec", - "todo", "memory", + "todo_list", "memory", "session_search", "clarify", "execute_code", "delegate_task", ], @@ -419,7 +419,7 @@ TOOLSETS = { "description": "Editor integration (VS Code, Zed, JetBrains) — coding-focused tools without messaging, audio, or clarify UI", "tools": [ "web_search", "web_extract", - "terminal", "process", + "terminal", "process_manage", "read_file", "write_file", "patch", "search_files", "vision_analyze", "skills_list", "skill_view", "skill_manage", @@ -428,7 +428,7 @@ TOOLSETS = { "browser_press", "browser_get_images", "browser_vision", "browser_console", "browser_cdp", "browser_dialog", "browser_exec", - "todo", "memory", + "todo_list", "memory", "session_search", "execute_code", "delegate_task", ], @@ -441,7 +441,7 @@ TOOLSETS = { # Web "web_search", "web_extract", # Terminal + process management - "terminal", "process", + "terminal", "process_manage", # File manipulation "read_file", "write_file", "patch", "search_files", # Vision + image generation @@ -455,13 +455,13 @@ TOOLSETS = { "browser_vision", "browser_console", "browser_cdp", "browser_dialog", "browser_exec", # Planning & memory - "todo", "memory", + "todo_list", "memory", # Session history search "session_search", # Code execution + delegation "execute_code", "delegate_task", # Cronjob management - "cronjob", + "cronjob_manage", # Home Assistant smart home control (gated on HASS_TOKEN via check_fn) "ha_list_entities", "ha_get_state", "ha_list_services", "ha_call_service", diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 6f0f6ee2ef..3d6889173f 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -7659,7 +7659,7 @@ def _on_tool_complete(sid: str, tool_call_id: str, name: str, args: dict, result result_text = _tool_result_text(result) if result_text: payload["result_text"] = result_text - if name == "todo": + if name == "todo_list": try: data = json.loads(result) if isinstance(data, dict) and isinstance(data.get("todos"), list):