diff --git a/apps/shared/src/gateway-contract.generated.ts b/apps/shared/src/gateway-contract.generated.ts
index 8c1f85edd5..8262e5c128 100644
--- a/apps/shared/src/gateway-contract.generated.ts
+++ b/apps/shared/src/gateway-contract.generated.ts
@@ -3533,7 +3533,7 @@ export interface McpServerRuntimeRow {
disabled: boolean
status: McpRuntimeStatus
}
-export type McpRuntimeStatus = 'connected' | 'disabled' | 'connecting' | 'failed' | 'configured'
+export type McpRuntimeStatus = 'connected' | 'disabled' | 'connecting' | 'failed' | 'lazy' | 'configured'
/** ``preset`` (catalog id) and/or ``config`` (url/command/args/env/headers/auth/tools); a ``bearer_token`` is written to the profile's .env, only the header template persists. */
export interface McpServersAddParams {
profile?: string | null
diff --git a/apps/shared/src/gateway-contract.openrpc.json b/apps/shared/src/gateway-contract.openrpc.json
index a05d5905fe..c4480ec5fa 100644
--- a/apps/shared/src/gateway-contract.openrpc.json
+++ b/apps/shared/src/gateway-contract.openrpc.json
@@ -14154,6 +14154,7 @@
"disabled",
"connecting",
"failed",
+ "lazy",
"configured"
],
"title": "McpRuntimeStatus",
diff --git a/cli-config.yaml.example b/cli-config.yaml.example
index 1ab64bce30..91e2a2c47d 100644
--- a/cli-config.yaml.example
+++ b/cli-config.yaml.example
@@ -1470,6 +1470,12 @@ platform_toolsets:
# Lower it below the server's session TTL for servers that expire idle
# sessions quickly (e.g. Unreal Engine editor MCP, ~15s), otherwise idle
# tool calls hit an expired session and pay a slow reconnect. Floored at 5s.
+# lazy: register the server's tools from the on-disk schema cache at startup
+# and only spawn/connect it on the first tool call (default: false). Saves
+# one child process per server in processes that rarely call tools. Needs
+# one prior live connect to populate the cache; a missing or stale cache
+# entry falls back to the normal eager connect. The banner and the TUI show
+# such a server as "lazy" with its cached tool count until it is first used.
#
# mcp_servers:
# time:
diff --git a/tests/hermes_cli/test_mcp_startup.py b/tests/hermes_cli/test_mcp_startup.py
index ebfe99701f..df03dade2d 100644
--- a/tests/hermes_cli/test_mcp_startup.py
+++ b/tests/hermes_cli/test_mcp_startup.py
@@ -309,7 +309,7 @@ def _retry_logger():
)
-def _install_retry_stubs(monkeypatch, *, connected: bool, calls: dict):
+def _install_retry_stubs(monkeypatch, *, connected: bool, calls: dict, status: str = "configured"):
monkeypatch.setitem(
sys.modules,
"hermes_cli.config",
@@ -327,7 +327,7 @@ def _install_retry_stubs(monkeypatch, *, connected: bool, calls: dict):
"tools.mcp_tool_discovery",
types.SimpleNamespace(
discover_mcp_tools=lambda: calls.__setitem__("mcp", calls["mcp"] + 1),
- get_mcp_status=lambda: [{"connected": connected}],
+ get_mcp_status=lambda: [{"name": "demo", "connected": connected, "status": status}],
),
)
@@ -440,3 +440,29 @@ def test_prepare_agent_startup_installs_server_filter(monkeypatch, _reset_mcp_se
monkeypatch.setattr(main_mod, "_command_has_dedicated_mcp_startup", lambda args: True)
main_mod._prepare_agent_startup(_agent_args(toolsets="terminal,code-mcp"))
assert mcp_startup.get_mcp_server_filter() == ["terminal", "code-mcp"]
+
+
+@pytest.mark.parametrize(("status", "retried"), [("lazy", False), ("configured", True)])
+def test_lazy_only_discovery_counts_as_usable_at_both_startup_sites(monkeypatch, status, retried):
+ """A finished run whose servers are all ``lazy`` (registered from the schema cache, spawned
+ on first use) left them usable: no zero-connected warning, and the re-entry check must not
+ re-spawn discovery (#111717). Control: a run that left them merely ``configured`` still warns
+ and is still retried (#66981)."""
+ calls = {"mcp": 0}
+ _install_retry_stubs(monkeypatch, connected=False, calls=calls, status=status)
+ warnings: list = []
+ logger = types.SimpleNamespace(debug=lambda *_a, **_k: None,
+ warning=lambda msg, *a, **_k: warnings.append(msg % a if a else msg))
+
+ mcp_startup.start_background_mcp_discovery(logger=logger, thread_name="t") # first run
+ thread = mcp_startup._current_home_thread()
+ if thread is not None:
+ thread.join(timeout=5.0)
+ mcp_startup.start_background_mcp_discovery(logger=logger, thread_name="t") # re-entry after it finished
+ thread = mcp_startup._current_home_thread()
+ if thread is not None:
+ thread.join(timeout=5.0)
+
+ assert calls["mcp"] == (2 if retried else 1)
+ assert any("zero connected" in w for w in warnings) is retried
+ assert any("retrying discovery thread" in w for w in warnings) is retried
diff --git a/tests/tools/test_mcp_lazy_start.py b/tests/tools/test_mcp_lazy_start.py
index b1ecc21cda..adfe451d50 100644
--- a/tests/tools/test_mcp_lazy_start.py
+++ b/tests/tools/test_mcp_lazy_start.py
@@ -334,3 +334,48 @@ class TestResolveServerLazy:
def test_explicit_false(self):
assert _mcp_discovery._resolve_server_lazy("s", {"command": "npx", "lazy": False}) is False
+
+
+class TestLazyMcpStatus:
+ def test_lazy_registration_reports_lazy_with_cached_tools_not_failed(self, caplog):
+ """A ``lazy: true`` server registered from the schema cache is a working server: status
+ ``lazy`` with its cached tool count (``connected`` False), and the discovery summary counts
+ it as a lazy server, never as failed (#111717). Controls: an unregistered eager server stays
+ ``configured``; a live one stays ``connected``."""
+ import logging
+
+ from tools import mcp_tool_config as _mcp_config
+ from tools import mcp_tool_loop as _mcp_loop
+ from tools.mcp_tool_scope import _server_key
+
+ servers = _lazy_config()
+ controls = {"eager": {"command": "/nonexistent/eager"}, "live": {"command": "/nonexistent/live"}}
+ live = SimpleNamespace(session=object(), _registered_tool_names=["l1", "l2"], _sampling=None, _tools=[])
+ cached = ["mcp_playwright_browser_navigate", "mcp_playwright_browser_click"]
+
+ def _fake_register(name, cfg, entry):
+ key = _server_key(name)
+ mcp._lazy_server_configs[key] = dict(cfg)
+ mcp._lazy_server_tool_names[key] = list(cached)
+ return list(cached)
+
+ with patch("tools.mcp_tool._MCP_AVAILABLE", True), \
+ patch.object(_mcp_config, "_load_mcp_config", return_value=dict(servers)), \
+ patch.object(_mcp_loop, "_try_acquire_mcp_discovery_lock", return_value=mcp._LOCK_UNAVAILABLE), \
+ patch("tools.mcp_schema_cache.config_fingerprint", return_value="abc"), \
+ patch("tools.mcp_schema_cache.get_cached_entry", return_value=_fake_cache_entry()), \
+ patch("tools.mcp_tool_registration._register_from_cache_sync", side_effect=_fake_register), \
+ patch("tools.mcp_tool_discovery._discover_and_register_server", new_callable=AsyncMock), \
+ patch("tools.mcp_tool_loop._ensure_mcp_loop"), patch("tools.mcp_tool_loop._run_on_mcp_loop"), \
+ caplog.at_level(logging.INFO, logger="tools.mcp_tool"):
+ _mcp_discovery.discover_mcp_tools()
+ mcp._servers[_server_key("live")] = live
+ status = {e["name"]: e for e in _mcp_discovery.get_mcp_status({**servers, **controls})}
+
+ summaries = [r.getMessage() for r in caplog.records if "tool(s) from" in r.getMessage()]
+ assert summaries and all("failed" not in m for m in summaries), summaries
+ assert any("1 lazy, not spawned yet" in m for m in summaries), summaries
+ assert (status["playwright"]["status"], status["playwright"]["tools"],
+ status["playwright"]["connected"]) == ("lazy", len(cached), False)
+ assert status["eager"]["status"] == "configured" and status["eager"]["tools"] == 0
+ assert status["live"]["status"] == "connected" and status["live"]["tools"] == 2
diff --git a/tui_gateway/contracts/tools_mcp_plugins.py b/tui_gateway/contracts/tools_mcp_plugins.py
index 7e6ef383e8..071b592927 100644
--- a/tui_gateway/contracts/tools_mcp_plugins.py
+++ b/tui_gateway/contracts/tools_mcp_plugins.py
@@ -381,6 +381,7 @@ class McpRuntimeStatus(WireEnum):
disabled = "disabled"
connecting = "connecting"
failed = "failed"
+ lazy = "lazy"
configured = "configured"
diff --git a/ui-tui/src/components/branding.tsx b/ui-tui/src/components/branding.tsx
index 2d9d93fe1d..d82375778a 100644
--- a/ui-tui/src/components/branding.tsx
+++ b/ui-tui/src/components/branding.tsx
@@ -324,6 +324,11 @@ export function SessionPanel({ info, maxWidth, sid, t }: SessionPanelProps) {
disabled
) : s.status === 'connecting' ? (
connecting
+ ) : s.status === 'lazy' ? (
+ // Registered from the schema cache, process not spawned yet: its tools are callable.
+
+ {s.tools} tool{s.tools === 1 ? '' : 's'} (lazy)
+
) : s.status === 'configured' ? (
configured
) : (
diff --git a/ui-tui/src/types.ts b/ui-tui/src/types.ts
index ac0caa5f3f..5e4f42807b 100644
--- a/ui-tui/src/types.ts
+++ b/ui-tui/src/types.ts
@@ -176,7 +176,7 @@ export type SectionVisibility = Partial>
export interface McpServerStatus {
connected: boolean
disabled?: boolean
- status?: 'configured' | 'connecting' | 'connected' | 'disabled' | 'failed'
+ status?: 'configured' | 'connecting' | 'connected' | 'disabled' | 'failed' | 'lazy'
name: string
tools: number
transport: string
diff --git a/website/docs/reference/mcp-config-reference.md b/website/docs/reference/mcp-config-reference.md
index 63fef9414a..5b9e2e5bfa 100644
--- a/website/docs/reference/mcp-config-reference.md
+++ b/website/docs/reference/mcp-config-reference.md
@@ -61,6 +61,7 @@ mcp_servers:
| `skip_preflight` | bool | HTTP | Bypass the fail-fast content-type probe for valid Streamable HTTP endpoints whose HEAD/GET answers a non-MCP content type (default: `false`) |
| `transport` | string | HTTP | Set to `sse` to use the SSE transport instead of Streamable HTTP |
| `keepalive_interval` | number | both | Liveness ping cadence in seconds (default: `180`, floored at 5s). Set below the server's session TTL for servers that GC idle sessions quickly |
+| `lazy` | bool | both | Register the server's tools from the on-disk schema cache at startup and only spawn/connect it on the first tool call (default: `false`). Needs one prior live connect to fill the cache; a missing or stale entry falls back to the normal eager connect. Status surfaces show the server as `lazy` with its cached tool count until first use |
| `idle_timeout_seconds` | number | stdio | Optional stdio server recycle after idle time (`0` disables). May also live under a `lifecycle:` mapping |
| `max_lifetime_seconds` | number | stdio | Optional stdio server recycle after age (`0` disables). May also live under a `lifecycle:` mapping |
| `tools` | mapping | both | Filtering and utility-tool policy |
diff --git a/website/docs/user-guide/features/mcp.md b/website/docs/user-guide/features/mcp.md
index e2e542e1ec..ebf3771104 100644
--- a/website/docs/user-guide/features/mcp.md
+++ b/website/docs/user-guide/features/mcp.md
@@ -394,6 +394,7 @@ Hermes reads MCP config from `~/.hermes/config.yaml` under `mcp_servers`.
| `identity_header` | mapping | Optional per-user identity header for HTTP/SSE servers — `{name, value_from: static\|profile, value}` |
| `timeout` | number | Tool call timeout |
| `connect_timeout` | number | Initial connection timeout (also bounds the MCP `initialize` handshake) |
+| `lazy` | bool | If `true`, register the server's tools from the schema cache at startup and only start/connect it on the first tool call (default `false`). Needs one prior live connect to fill the cache. |
| `idle_timeout_seconds` | number | Recycle a stdio server after this many seconds without a tool call (`0` = never, default). The server restarts transparently on the next tool call. |
| `max_lifetime_seconds` | number | Recycle a stdio server after this total age (`0` = never, default). Restarts transparently on next use. |
| `enabled` | bool | If `false`, Hermes skips the server entirely |
@@ -640,6 +641,10 @@ That keeps the tool list clean.
Hermes discovers MCP servers at startup and registers their tools into the normal tool registry.
+### Lazy start
+
+A server with `lazy: true` is registered from the on-disk schema cache instead: its tools appear in the registry immediately, and the process is spawned (or the HTTP endpoint connected) on the first tool call. The cache is written on every live connect, so the first run of a new or changed server is always eager. The banner and the TUI session panel show such a server as **lazy** with its cached tool count (`3 tool(s) (lazy, starts on first use)`) — it is a working server, not a failed one — and the startup discovery summary counts it as `N lazy, not spawned yet`.
+
### Dynamic Tool Discovery
MCP servers can notify Hermes when their available tools change at runtime by sending a `notifications/tools/list_changed` notification. When Hermes receives this notification, it automatically re-fetches the server's tool list and updates the registry — no manual `/reload-mcp` required.