Two product questions shared metrics could not answer: how long each surface
takes from launch to usable (so startup regressions show per release), and how
current, on which channel and on what class of machine installs run.
hermes.startup.latency {surface, latency_bucket} records once per process start:
cli (process creation -> first rendered prompt, or -q dispatch; Kanban workers
excluded), gateway_boot (-> GatewayRunner.start done), serve_boot (-> hermes
serve listening), and tui / desktop_attach reported by the clients through the
new shared_metrics.startup_latency RPC. The clients declare their surface
because a Desktop on a URL/cloud backend has no HERMES_DESKTOP there; env
detection is the fallback for older clients. In-process surfaces measure from
psutil's process create time, the earliest timestamp available, and hand the
runtime start to a daemon thread under the caller's context so no event loop
waits on it. Everything goes through _emit, so disabled profiles record nothing.
The install snapshot gains release_channel, version_age_bucket, behind_bucket,
ram_bucket, gpu_class and local_model_provider_used. All are read offline:
the installed commit's own date, the channel record / packaged channel / checkout
branch (never the remote URL or branch name), and the update check's existing
cache for this exact revision (never a network call). Rows counted before these
fields existed stay valid as a legacy field set.
1612 lines
70 KiB
Python
1612 lines
70 KiB
Python
"""Hermes Agent — Web UI server: FastAPI app assembly, auth/host middleware, ``start_server``.
|
|
|
|
Route handlers live in ``web_routers/``; their helpers live in the sibling
|
|
``web_server_<concern>`` modules and are re-imported here so ``web_server.<name>``
|
|
stays the single late-binding seam tests monkeypatch (``web_deps.late``).
|
|
Usage: ``python -m hermes_cli.main web [--port 8080]``.
|
|
"""
|
|
|
|
from contextlib import asynccontextmanager
|
|
|
|
import asyncio
|
|
from collections import deque
|
|
import hmac
|
|
import logging
|
|
import os
|
|
import re
|
|
import secrets
|
|
import subprocess
|
|
import sys
|
|
import sysconfig
|
|
import threading
|
|
import time
|
|
import urllib.parse
|
|
|
|
from hermes_cli.install_identity import get_install_id as _shared_get_install_id
|
|
from hermes_cli.process_identity import is_desktop_owned_backend
|
|
from hermes_cli.pty_session import run_reaper
|
|
from pathlib import Path
|
|
from typing import Any, Dict, Optional, Tuple
|
|
|
|
|
|
PROJECT_ROOT = Path(__file__).parent.parent.resolve()
|
|
if str(PROJECT_ROOT) not in sys.path:
|
|
sys.path.insert(0, str(PROJECT_ROOT))
|
|
|
|
from hermes_cli.config import load_config
|
|
from hermes_cli.version_info import get_version_info
|
|
|
|
try:
|
|
from fastapi import FastAPI, HTTPException, Request
|
|
from fastapi.middleware.cors import CORSMiddleware
|
|
from fastapi.responses import JSONResponse
|
|
except ImportError:
|
|
# First try lazy-installing the dashboard extras. Only the user actually
|
|
# running `hermes dashboard` needs fastapi+uvicorn; lazy install keeps
|
|
# them out of every other install path. After install, re-import.
|
|
try:
|
|
from pm import ensure_import
|
|
ensure_import("web")
|
|
from fastapi import (
|
|
FastAPI, HTTPException, Request,
|
|
)
|
|
from fastapi.middleware.cors import CORSMiddleware
|
|
from fastapi.responses import JSONResponse
|
|
except Exception:
|
|
raise SystemExit(
|
|
"Web UI requires fastapi and uvicorn.\n"
|
|
"Run hermes pm repair, then restart Hermes."
|
|
)
|
|
|
|
WEB_DIST = Path(os.environ["HERMES_WEB_DIST"]) if "HERMES_WEB_DIST" in os.environ else Path(__file__).parent / "web_dist"
|
|
_log = logging.getLogger(__name__)
|
|
|
|
|
|
from hermes_cli.web_server_lifecycle import ( # noqa: E402
|
|
PORT_IN_USE_EXIT_CODE,
|
|
_dashboard_forwarded_allow_ips,
|
|
_eager_reconcile_own_session_db,
|
|
_maybe_open_browser,
|
|
_port_bind_conflict,
|
|
_read_bound_port,
|
|
_report_port_in_use,
|
|
_start_parent_death_watchdog,
|
|
_warm_gateway_module,
|
|
_write_dashboard_ready_file,
|
|
_write_machine_sentinel_line,
|
|
)
|
|
|
|
|
|
def _start_desktop_cron_ticker(stop_event: "threading.Event", interval: int = 60) -> None:
|
|
"""Tick the cron scheduler from inside the desktop dashboard backend.
|
|
|
|
The desktop spawns a ``hermes dashboard`` backend, not a gateway, so without
|
|
this a cron created in the app would never fire (no live adapters; delivery
|
|
falls back to the per-platform send path). The primary backend outlives the
|
|
per-profile pool (reaped after ~10 idle minutes), so it ticks EVERY local
|
|
profile's store like a multiplex gateway; external providers keep the
|
|
single-store behavior (registries are not profile-scoped). Cross-process
|
|
safe: the built-in tick takes the per-store ``cron/.tick.lock``.
|
|
|
|
Every local profile's store is ticked, not just this backend's own (#69377's desktop sibling): the
|
|
desktop pools per-profile backends and reaps them after ~10 idle minutes, so a secondary profile's
|
|
ticker dies with its backend and that profile's jobs silently stop firing until the user next opens it
|
|
("tasks on the sleeping profile could be idle" — community report, Aug 2026).
|
|
"""
|
|
from cron.scheduler_provider import InProcessCronScheduler, resolve_cron_scheduler
|
|
|
|
# A live gateway on THIS backend's HERMES_HOME owns cron delivery with live platform
|
|
# adapters (#52202): let it tick, and start nothing here. Without this, the fail-open
|
|
# paths below (profile enumeration failure, empty served set, external provider) start
|
|
# an ungated single-store ticker that races the gateway's tick-lock; when the desktop
|
|
# wins, delivery has no live adapter and the cold send hangs until script_timeout.
|
|
try:
|
|
from hermes_constants import get_hermes_home
|
|
from hermes_cli.profiles import _check_gateway_running
|
|
|
|
if _check_gateway_running(Path(get_hermes_home())):
|
|
_log.info(
|
|
"Desktop cron scheduler not started: live gateway owns cron on this "
|
|
"HERMES_HOME; the gateway ticks with live adapters"
|
|
)
|
|
return
|
|
except Exception:
|
|
# Liveness probe failed: fall through to the existing per-tick gating, which
|
|
# still stands down profile-by-profile for gateway-owned homes.
|
|
_log.warning("Desktop cron: gateway-ownership probe failed; using per-tick gating only", exc_info=True)
|
|
|
|
provider = resolve_cron_scheduler()
|
|
|
|
start_kwargs: dict = {"interval": interval}
|
|
if isinstance(provider, InProcessCronScheduler):
|
|
try:
|
|
from hermes_cli.profiles import (
|
|
_check_gateway_running, _served_by_running_multiplexer, profiles_to_serve)
|
|
|
|
# Same served set as the multiplexer: default + every live profile under profiles/.
|
|
# The ticker re-enumerates this callable every cycle. Passing a
|
|
# startup snapshot leaves deleted profiles in the scheduler until
|
|
# restart, which both writes their removed stores and keeps stale
|
|
# profiles alive in Desktop's background work.
|
|
profile_homes = lambda: list(profiles_to_serve(multiplex=True))
|
|
initial_profile_homes = profile_homes()
|
|
if initial_profile_homes:
|
|
# Even one profile needs the per-tick gateway gate; otherwise
|
|
# Desktop races its dedicated gateway for the same cron store.
|
|
start_kwargs["profile_homes"] = profile_homes
|
|
# Stand down, per tick, for a profile already owned by a gateway — its OWN
|
|
# process, or the live default multiplexer (a served satellite has no gateway.pid
|
|
# of its own). That gateway ticks with live adapters; winning the tick-lock race
|
|
# here would deliver through the standalone path (#100489, #107485).
|
|
start_kwargs["profile_gate"] = lambda name, home: not (
|
|
_check_gateway_running(Path(home))
|
|
or (name != "default" and _served_by_running_multiplexer(name)))
|
|
from hermes_logging import enable_profile_log_routing
|
|
|
|
enable_profile_log_routing(initial_profile_homes)
|
|
_log.info(
|
|
"Desktop cron scheduler will tick %d profile(s): %s",
|
|
len(initial_profile_homes),
|
|
[name for name, _home in initial_profile_homes],
|
|
)
|
|
except Exception:
|
|
# Fail open to the single-store ticker so the active profile keeps firing.
|
|
_log.exception("Desktop cron: profile enumeration failed; ticking active profile only")
|
|
|
|
_log.info("Desktop cron scheduler started (provider=%s, interval=%ds)", provider.name, interval)
|
|
provider.start(stop_event, **start_kwargs)
|
|
|
|
|
|
# Desktop `serve` only (start_server(start_mcp_discovery_after_bind=True)):
|
|
# seconds after the READY sentinel before the MCP discovery thread starts.
|
|
_DESKTOP_MCP_DISCOVERY_DELAY_S = 1.0
|
|
|
|
|
|
@asynccontextmanager
|
|
async def _lifespan(app: "FastAPI"):
|
|
app.state.event_channels = {} # dict[str, set]
|
|
app.state.event_lock = asyncio.Lock()
|
|
app.state.pty_active_session_files = {} # dict[str, Path]
|
|
# Serializes chat-argv resolution so concurrent /api/pty connections don't
|
|
# overlap ``npm install`` / ``npm run build``. Locks live on app.state (not
|
|
# module globals) so they bind to the running loop, not the import-time one.
|
|
app.state.chat_argv_lock = asyncio.Lock()
|
|
|
|
# Bring state.db schema current BEFORE the first session-list poll
|
|
# (#79531/#80037): a store left behind by `hermes update` otherwise 500s
|
|
# every poll while the read-probe heal loses to sibling lock contention.
|
|
# Off-thread so a locked store never delays the socket (Desktop
|
|
# ready-probe times out at 10s, GH-73083). NOT a daemon, and joined at
|
|
# shutdown: its sqlite connection must be closed by the thread that is
|
|
# stepping it. A daemon copy that outlived the lifespan had its
|
|
# connection closed from the main thread mid-probe (pytest's leaked-DB
|
|
# sweep) and segfaulted the interpreter. The worker is time-bounded by
|
|
# SessionDB's lock patience, so the join cannot hang shutdown.
|
|
eager_reconcile_thread = threading.Thread(
|
|
target=_eager_reconcile_own_session_db,
|
|
name="statedb-eager-reconcile",
|
|
)
|
|
eager_reconcile_thread.start()
|
|
|
|
# Import hermes_cli.gateway *before* the yield: on Windows + 3.11 the
|
|
# import holds the GIL, so run_in_executor still froze the loop 15-22s and
|
|
# the Desktop's 10s ready-probe timed out (GH-73083).
|
|
_warm_gateway_module()
|
|
|
|
# Snapshot the checkout revision so lazy-import paths (model picker) can
|
|
# refuse with "restart required" after `hermes update` replaced the code
|
|
# (#86207); the update flow does not reliably restart the dashboard.
|
|
from gateway.code_skew import record_boot_fingerprint
|
|
|
|
record_boot_fingerprint()
|
|
|
|
# Hosted Bot rooms belong to the backend process. Recovery may need a
|
|
# contended state.db migration, so keep it off the pre-yield path: Group
|
|
# Chat must degrade on its own rather than block every Desktop feature.
|
|
from tui_gateway import methods_groups as _hosted_groups
|
|
import tui_gateway.server # noqa: F401
|
|
|
|
try:
|
|
tui_gateway.server.install_tui_message_injector()
|
|
except Exception:
|
|
_log.warning("TUI message injector did not install", exc_info=True)
|
|
|
|
hosted_room_start_cancel = threading.Event()
|
|
|
|
def _start_hosted_rooms() -> None:
|
|
try:
|
|
_hosted_groups.start_hosted_room_service()
|
|
except Exception:
|
|
_log.exception("Hosted Group Chat recovery failed during backend startup")
|
|
finally:
|
|
if hosted_room_start_cancel.is_set():
|
|
_hosted_groups.stop_hosted_room_service(timeout=1.0)
|
|
|
|
hosted_room_start_thread = threading.Thread(
|
|
target=_start_hosted_rooms,
|
|
daemon=True,
|
|
name="hosted-room-startup",
|
|
)
|
|
hosted_room_start_thread.start()
|
|
|
|
# Desktop-spawned backends fire cron jobs themselves, since the app has no
|
|
# gateway running the scheduler. Server `hermes dashboard` is unaffected —
|
|
# it relies on its own gateway.
|
|
cron_stop: "threading.Event | None" = None
|
|
cron_thread: "threading.Thread | None" = None
|
|
desktop_owned = is_desktop_owned_backend()
|
|
if desktop_owned:
|
|
# Reap an orphaned gateway from an abnormal previous exit (reparented to
|
|
# launchd, still holding the platform WebSocket) before forking a fresh
|
|
# one that would race the same credential (#77276). Runs
|
|
# unconditionally; protection of a healthy standalone gateway lives
|
|
# INSIDE the reaper (registration probed with cleanup_stale=False).
|
|
# Startup grace: spare a gateway still claiming gateway.pid/lock (#122533).
|
|
try:
|
|
from hermes_cli.dashboard_procs import _REAP_MIN_AGE_SECONDS
|
|
from hermes_cli.gateway import _reap_unsupervised_gateway_orphans
|
|
|
|
_reap_unsupervised_gateway_orphans(min_age_s=_REAP_MIN_AGE_SECONDS)
|
|
except Exception:
|
|
_log.exception("Desktop startup: orphan gateway reap failed")
|
|
|
|
cron_stop = threading.Event()
|
|
cron_thread = threading.Thread(
|
|
target=_start_desktop_cron_ticker,
|
|
args=(cron_stop,),
|
|
daemon=True,
|
|
name="desktop-cron-ticker",
|
|
)
|
|
cron_thread.start()
|
|
|
|
# Reap idle/dead keep-alive PTY sessions (30-min TTL).
|
|
pty_reaper_task = asyncio.create_task(run_reaper(PTY_REGISTRY))
|
|
# Periodic authenticated self-test feeding the ``dashboard`` component on /api/status.
|
|
selftest_task = asyncio.create_task(_dashboard_selftest_loop())
|
|
# Live auto-archive timer, independent of list requests.
|
|
auto_archive_task = asyncio.create_task(_auto_archive_ticker_loop())
|
|
|
|
# Managed local runtime (local_runtime.enabled): bring llama-server back so a
|
|
# restart doesn't strand a llamacpp main model. Off-thread and best-effort;
|
|
# failure falls back to cloud providers like a cold start. Server only —
|
|
# models load on first inference (an empty router holds no VRAM).
|
|
def _boot_local_runtime():
|
|
try:
|
|
from hermes_cli.config import load_config
|
|
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
|
|
|
|
ensure_local_runtime(load_config())
|
|
except Exception as exc: # noqa: BLE001
|
|
logging.getLogger(__name__).warning("local runtime boot failed: %s", exc)
|
|
|
|
threading.Thread(target=_boot_local_runtime, daemon=True, name="local-runtime-boot").start()
|
|
|
|
# Nous free tier: the ONE place its identity is created. Inventories credentials, mints only
|
|
# when HERMES_GUEST_ONBOARDING=1, records the answer for setup.status / free_tier.status and
|
|
# broadcasts `setup.ready`. Off-thread so a slow portal never delays the socket; the desktop's
|
|
# first setup.status waits on the record (bounded) instead.
|
|
from hermes_cli.free_tier_bootstrap import start_background_bootstrap
|
|
|
|
start_background_bootstrap()
|
|
|
|
try:
|
|
yield
|
|
finally:
|
|
try:
|
|
tui_gateway.server.clear_tui_message_injector()
|
|
except Exception:
|
|
_log.debug("TUI message injector clear skipped", exc_info=True)
|
|
hosted_room_start_cancel.set()
|
|
_hosted_groups.stop_hosted_room_service(timeout=5.0)
|
|
hosted_room_start_thread.join(timeout=1.0)
|
|
if cron_stop is not None:
|
|
cron_stop.set()
|
|
pty_reaper_task.cancel()
|
|
selftest_task.cancel()
|
|
auto_archive_task.cancel()
|
|
await PTY_REGISTRY.close_all()
|
|
# Stop the managed llama-server with its parent (an orphan pins VRAM).
|
|
try:
|
|
from hermes_cli.local_runtime.bootstrap import shutdown_local_runtime
|
|
|
|
shutdown_local_runtime()
|
|
except Exception: # noqa: BLE001
|
|
pass
|
|
if desktop_owned:
|
|
_terminate_desktop_managed_gateway()
|
|
eager_reconcile_thread.join()
|
|
|
|
|
|
def _app_state_default(app: "FastAPI", name: str, factory):
|
|
"""Return ``app.state.<name>``, lazily creating it for non-``with`` TestClient usages.
|
|
|
|
The lifespan normally initialises these on the running event loop (an
|
|
asyncio.Lock created at import time binds to whatever loop was active then).
|
|
"""
|
|
try:
|
|
return getattr(app.state, name)
|
|
except AttributeError:
|
|
value = factory()
|
|
setattr(app.state, name, value)
|
|
return value
|
|
|
|
|
|
def _get_chat_argv_lock(app: "FastAPI") -> asyncio.Lock:
|
|
return _app_state_default(app, "chat_argv_lock", asyncio.Lock)
|
|
|
|
|
|
def _get_pty_active_session_files(app: "FastAPI") -> dict[str, Path]:
|
|
return _app_state_default(app, "pty_active_session_files", dict)
|
|
|
|
|
|
app = FastAPI(title="Hermes Agent", version=get_version_info().base_version, lifespan=_lifespan)
|
|
|
|
|
|
# Memory-provider OAuth connect routes live in the memory layer, not here.
|
|
from hermes_cli.memory_oauth import router as _memory_oauth_router # noqa: E402
|
|
|
|
app.include_router(_memory_oauth_router)
|
|
|
|
# Session token for sensitive endpoints. The desktop shell mints it via
|
|
# HERMES_DASHBOARD_SESSION_TOKEN; otherwise fresh per server start. It dies with
|
|
# the process and is injected into the SPA HTML so only the web UI can use it.
|
|
def _resolve_session_token() -> str:
|
|
return os.environ.get("HERMES_DASHBOARD_SESSION_TOKEN") or secrets.token_urlsafe(32)
|
|
|
|
|
|
_SESSION_TOKEN = _resolve_session_token()
|
|
_SESSION_HEADER_NAME = "X-Hermes-Session-Token"
|
|
_SSH_OWNER_NONCE: Optional[str] = None
|
|
_SSH_RUNTIME_PURELIB: Optional[Tuple[str, int, int]] = None
|
|
_SSH_RUNTIME_MARKER: Optional[str] = None
|
|
|
|
|
|
def _apply_ssh_session_token(token: str) -> None:
|
|
global _SESSION_TOKEN
|
|
if token:
|
|
_SESSION_TOKEN = token
|
|
|
|
|
|
def _apply_ssh_owner_nonce(nonce: Optional[str]) -> None:
|
|
global _SSH_OWNER_NONCE, _SSH_RUNTIME_PURELIB, _SSH_RUNTIME_MARKER
|
|
_SSH_OWNER_NONCE = nonce
|
|
_SSH_RUNTIME_PURELIB = None
|
|
_SSH_RUNTIME_MARKER = None
|
|
if nonce:
|
|
try:
|
|
purelib = sysconfig.get_paths()["purelib"]
|
|
except (KeyError, OSError):
|
|
return
|
|
# Primary identity: a marker FILE in site-packages. A replaced venv
|
|
# loses it deterministically; pip installs leave it. A bare (dev, ino)
|
|
# snapshot alone is NOT enough: ext4 reuses directory inodes at once,
|
|
# so `rm -rf venv && uv venv` can land on the same inode undetected.
|
|
try:
|
|
marker = os.path.join(purelib, f".hermes-ssh-runtime-{nonce}")
|
|
with open(marker, "w", encoding="utf-8") as fh:
|
|
fh.write(f"pid={os.getpid()}\n")
|
|
_SSH_RUNTIME_MARKER = marker
|
|
except OSError:
|
|
pass # read-only site-packages — fall back to the stat snapshot
|
|
try:
|
|
st = os.stat(purelib)
|
|
_SSH_RUNTIME_PURELIB = (purelib, st.st_dev, st.st_ino)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _ssh_runtime_intact() -> bool:
|
|
if _SSH_RUNTIME_MARKER is not None:
|
|
return os.path.isfile(_SSH_RUNTIME_MARKER)
|
|
# Fallback (read-only site-packages): directory identity snapshot — weaker
|
|
# (inode reuse) but catches cross-device moves and version-bump paths.
|
|
if _SSH_RUNTIME_PURELIB is None:
|
|
return True
|
|
purelib, device, inode = _SSH_RUNTIME_PURELIB
|
|
try:
|
|
st = os.stat(purelib)
|
|
except OSError:
|
|
return False
|
|
return (st.st_dev, st.st_ino) == (device, inode)
|
|
|
|
|
|
# In-browser Chat tab (/chat, /api/pty, /api/ws): always enabled. A module
|
|
# constant (not an inlined True) so the WS endpoints and SPA token injection
|
|
# share one testable seam.
|
|
_DASHBOARD_EMBEDDED_CHAT_ENABLED = True
|
|
|
|
# Desktop file.attach sends a whole base64 data URL in one JSON-RPC frame;
|
|
# uvicorn's 16 MiB default rejects files under the 256 MiB raw attach cap.
|
|
_DESKTOP_ATTACHMENT_WS_MAX_BYTES = 384 * 1024 * 1024
|
|
|
|
|
|
# CORS: localhost origins only — allow_origins=["*"] on 0.0.0.0 would let any
|
|
# website read/modify config and secrets.
|
|
app.add_middleware(
|
|
CORSMiddleware,
|
|
allow_origin_regex=r"^https?://(localhost|127\.0\.0\.1)(:\d+)?$",
|
|
allow_methods=["*"],
|
|
allow_headers=["*"],
|
|
)
|
|
|
|
# Endpoints that do NOT require the session token; everything else under /api/
|
|
# is gated below. Shared with the OAuth gate so the two allowlists cannot
|
|
# drift (/api/status once 401'd under the OAuth gate, breaking the portal probe).
|
|
from hermes_cli.dashboard_auth.public_paths import PUBLIC_API_PATHS as _PUBLIC_API_PATHS
|
|
|
|
|
|
def _has_valid_session_token(request: Request) -> bool:
|
|
"""True if the request carries a valid dashboard session token.
|
|
|
|
The dedicated header avoids collisions with reverse proxies that already use
|
|
``Authorization`` (Caddy ``basic_auth``); the legacy Bearer path stays for
|
|
older dashboard bundles.
|
|
"""
|
|
session_header = request.headers.get(_SESSION_HEADER_NAME, "")
|
|
if session_header and hmac.compare_digest(session_header.encode(), _SESSION_TOKEN.encode()):
|
|
return True
|
|
auth = request.headers.get("authorization", "")
|
|
return hmac.compare_digest(auth.encode(), f"Bearer {_SESSION_TOKEN}".encode())
|
|
|
|
|
|
# Routes that may also authenticate via ``?token=`` (download links opened by
|
|
# the OS shell / a new tab, where no header can be set). Kept narrow.
|
|
_QUERY_TOKEN_API_PATHS: frozenset[str] = frozenset({"/api/files/download"})
|
|
|
|
|
|
def _has_valid_query_token(request: Request, path: str) -> bool:
|
|
if path not in _QUERY_TOKEN_API_PATHS:
|
|
return False
|
|
token = request.query_params.get("token", "")
|
|
return bool(token) and hmac.compare_digest(token.encode(), _SESSION_TOKEN.encode())
|
|
|
|
|
|
def _require_token(request: Request) -> None:
|
|
"""Authorize a sensitive endpoint, raising 401 if the caller isn't allowed.
|
|
|
|
Loopback mode (``auth_required`` False): validate the SPA-injected
|
|
``_SESSION_TOKEN``. Gated mode: the token is NOT injected (cookie auth), and
|
|
``gated_auth_middleware`` already 401'd anything without a verified
|
|
``request.state.session`` — requiring the absent token here would make every
|
|
``_require_token`` endpoint unreachable behind the gate, so defer to it.
|
|
"""
|
|
if getattr(request.app.state, "auth_required", False):
|
|
ok = getattr(request.state, "session", None) is not None
|
|
else:
|
|
ok = _has_valid_session_token(request)
|
|
if not ok:
|
|
raise HTTPException(status_code=401, detail="Unauthorized")
|
|
|
|
|
|
# Accepted Host values for loopback binds. DNS rebinding TTL-flips an attacker
|
|
# hostname to 127.0.0.1 so the browser treats it as same-origin; validating Host
|
|
# at the app layer rejects it. See GHSA-ppp5-vxwm-4cf7.
|
|
_LOOPBACK_HOST_VALUES: frozenset = frozenset({"localhost", "127.0.0.1", "::1"})
|
|
|
|
|
|
def _dashboard_public_hosts() -> frozenset[str]:
|
|
"""Return the exact hostname declared by ``dashboard.public_url``.
|
|
|
|
One source of truth for OAuth redirects, Host and WS Origin validation.
|
|
Malformed or unset values fail closed as an empty set.
|
|
"""
|
|
from hermes_cli.dashboard_auth.prefix import resolve_public_url
|
|
|
|
public_url = resolve_public_url()
|
|
try:
|
|
hostname = urllib.parse.urlparse(public_url).hostname if public_url else None
|
|
except ValueError:
|
|
hostname = None
|
|
return frozenset({hostname.lower()}) if hostname else frozenset()
|
|
|
|
|
|
def should_require_auth(host: str, allow_public: bool = False) -> bool:
|
|
"""True iff the auth gate must be active: any non-loopback bind.
|
|
|
|
RFC1918 / CGNAT / link-local are deliberately PUBLIC — a hostile LAN device
|
|
is the threat model. ``allow_public`` (legacy ``--insecure``) is accepted for
|
|
old launch scripts but IGNORED since the June 2026 hermes-0day campaign.
|
|
"""
|
|
return host not in _LOOPBACK_HOST_VALUES
|
|
|
|
|
|
def should_require_dashboard_auth(
|
|
host: str,
|
|
trusted_public_hosts: Optional[frozenset[str]] = None,
|
|
) -> bool:
|
|
"""Gate required for a non-loopback bind OR a non-loopback ``dashboard.public_url``.
|
|
|
|
Callers may pass the already-resolved host set so startup and request
|
|
validation share one snapshot.
|
|
"""
|
|
if trusted_public_hosts is None:
|
|
trusted_public_hosts = _dashboard_public_hosts()
|
|
return should_require_auth(host) or any(h not in _LOOPBACK_HOST_VALUES for h in trusted_public_hosts)
|
|
|
|
|
|
def _desktop_loopback_auth_exempt(
|
|
host: str,
|
|
ssh_session_token: Optional[str] = None,
|
|
ssh_owner_nonce: Optional[str] = None,
|
|
) -> bool:
|
|
"""True for a Desktop-owned loopback backend (#96490).
|
|
|
|
A non-loopback ``dashboard.public_url`` would otherwise engage the
|
|
ticket-only gate for the private loopback backends Desktop spawns, whose
|
|
per-spawn session token the gate's WS path refuses — Desktop could not boot.
|
|
The public dashboard is a separate non-loopback process that stays gated, so
|
|
this never opens the public surface. Requires ALL of: loopback bind,
|
|
``HERMES_DESKTOP=1``, and an operator-minted credential (env token, SSH
|
|
session token, or owner nonce).
|
|
"""
|
|
return (
|
|
host in _LOOPBACK_HOST_VALUES
|
|
and os.environ.get("HERMES_DESKTOP") == "1"
|
|
and bool(os.environ.get("HERMES_DASHBOARD_SESSION_TOKEN") or ssh_session_token or ssh_owner_nonce)
|
|
)
|
|
|
|
|
|
def _host_header_hostname(host_header: str) -> str:
|
|
"""Return a normalized hostname from a valid HTTP Host authority.
|
|
|
|
Host headers are authorities, not full URLs. Reject ambiguous ports,
|
|
malformed IPv6 brackets, and URL syntax so validation always fails closed.
|
|
"""
|
|
value = (host_header or "").strip()
|
|
if not value or "://" in value or any(c in value for c in '"\'<> \n\r\t/?#@'):
|
|
return ""
|
|
|
|
if value.startswith("["):
|
|
close = value.find("]")
|
|
if close == -1:
|
|
return ""
|
|
hostname = value[1:close]
|
|
# Bracket notation is reserved for IPv6 literals.
|
|
if ":" not in hostname:
|
|
return ""
|
|
suffix = value[close + 1:]
|
|
if suffix and not re.fullmatch(r":\d+", suffix):
|
|
return ""
|
|
return hostname.lower()
|
|
|
|
# Unbracketed IPv6 authorities are ambiguous with a port separator.
|
|
if value.count(":") > 1:
|
|
return ""
|
|
if ":" in value:
|
|
hostname, port = value.rsplit(":", 1)
|
|
if not hostname or not port.isdigit():
|
|
return ""
|
|
return hostname.lower()
|
|
return value.lower()
|
|
|
|
|
|
def _is_accepted_host(
|
|
host_header: str,
|
|
bound_host: str,
|
|
trusted_public_hosts: frozenset[str] = frozenset(),
|
|
) -> bool:
|
|
"""True if the Host header targets the interface we bound to.
|
|
|
|
Accepts:
|
|
- Exact bound host (with or without port suffix)
|
|
- Loopback aliases when bound to loopback
|
|
- Exact operator-declared public hosts (with or without port suffix)
|
|
- Any host when bound to 0.0.0.0 (explicit opt-in to non-loopback,
|
|
no protection possible at this layer)
|
|
"""
|
|
host_only = _host_header_hostname(host_header)
|
|
if not host_only:
|
|
return False
|
|
# All-interfaces bind: no Host-layer defence is possible; rely on operator
|
|
# network controls.
|
|
if host_only in trusted_public_hosts or bound_host in {"0.0.0.0", "::"}:
|
|
return True
|
|
bound_lc = bound_host.lower()
|
|
if bound_lc in _LOOPBACK_HOST_VALUES:
|
|
return host_only in _LOOPBACK_HOST_VALUES
|
|
return host_only == bound_lc
|
|
|
|
|
|
@app.middleware("http")
|
|
async def host_header_middleware(request: Request, call_next):
|
|
"""Reject requests whose Host header doesn't match the bound interface (DNS rebinding, GHSA-ppp5-vxwm-4cf7)."""
|
|
# app.state.bound_host is set by start_server() at listen time.
|
|
bound_host = getattr(app.state, "bound_host", None)
|
|
if bound_host and not _is_accepted_host(
|
|
request.headers.get("host", ""), bound_host, getattr(app.state, "trusted_public_hosts", frozenset())
|
|
):
|
|
return JSONResponse(
|
|
status_code=400,
|
|
content={
|
|
"detail": (
|
|
"Invalid Host header. Dashboard requests must use the "
|
|
"bound hostname or the configured public hostname."
|
|
),
|
|
},
|
|
)
|
|
return await call_next(request)
|
|
|
|
|
|
@app.middleware("http")
|
|
async def _plugin_api_runtime_gate(request: Request, call_next):
|
|
"""Block requests to disabled plugin API routes at request time.
|
|
|
|
:func:`_mount_plugin_api_routes` gates at import time; a plugin disabled
|
|
while running keeps its router mounted until restart, so enforce on every
|
|
``/api/plugins/{name}/...`` request. Registered BEFORE the auth middlewares
|
|
(runs AFTER them): an unauthenticated caller must get auth's 401, never this
|
|
404, or the status code becomes a plugin-name oracle.
|
|
"""
|
|
path = request.url.path
|
|
# parts: ['', 'api', 'plugins', '<name>', ...]
|
|
parts = path.split("/")
|
|
plugin_name = parts[3] if path.startswith("/api/plugins/") and len(parts) >= 4 else ""
|
|
# Only gate authenticated requests. Unauthenticated ones fall through so
|
|
# auth_middleware / the OAuth gate return 401 first and this route can't
|
|
# be used as a plugin-name oracle.
|
|
if plugin_name and (
|
|
getattr(request.state, "token_authenticated", False)
|
|
or getattr(request.app.state, "auth_required", False)
|
|
or _has_valid_session_token(request)
|
|
or _has_valid_query_token(request, path)
|
|
):
|
|
try:
|
|
# Gate: only serve user plugins that are in plugins.enabled and not in plugins.disabled. This
|
|
# prevents the frontend from loading JS/CSS from plugins the user has not explicitly activated.
|
|
# (#46435)
|
|
from hermes_cli.plugins_cmd import _get_enabled_set, _get_disabled_set
|
|
enabled_set = _get_enabled_set()
|
|
disabled_set = _get_disabled_set()
|
|
except Exception:
|
|
enabled_set = set()
|
|
disabled_set = set()
|
|
# Source from the cached plugin list; unknown => user plugin (safe default — blocks).
|
|
plugin = next((p for p in _get_dashboard_plugins() if p.get("name") == plugin_name), None)
|
|
source = plugin.get("source") if plugin else "user"
|
|
blocked = plugin_name in disabled_set or (source == "user" and plugin_name not in enabled_set)
|
|
if blocked and source in ("user", "bundled"):
|
|
return JSONResponse(status_code=404, content={"detail": "Plugin not found"})
|
|
return await call_next(request)
|
|
|
|
|
|
@app.middleware("http")
|
|
async def _dashboard_auth_gate(request: Request, call_next):
|
|
"""OAuth gate — active only when start_server flags ``auth_required``; pass-through on loopback.
|
|
|
|
Registered between host_header and auth_middleware: host check → cookie auth → token auth.
|
|
"""
|
|
from hermes_cli.dashboard_auth.middleware import gated_auth_middleware
|
|
return await gated_auth_middleware(request, call_next)
|
|
|
|
|
|
@app.middleware("http")
|
|
async def auth_middleware(request: Request, call_next):
|
|
"""Require the session token on all /api/ routes except the public list.
|
|
|
|
Skipped for requests the token-auth seam already authenticated
|
|
(``token_authenticated``) and when the OAuth gate is active — cookie auth is
|
|
then authoritative and the loopback-only token path must not override it.
|
|
"""
|
|
path = request.url.path
|
|
if (
|
|
not getattr(request.state, "token_authenticated", False)
|
|
and not getattr(request.app.state, "auth_required", False)
|
|
and path.startswith("/api/")
|
|
and path not in _PUBLIC_API_PATHS
|
|
and not path.startswith("/api/mcp/oauth/callback/")
|
|
and not _has_valid_session_token(request)
|
|
and not _has_valid_query_token(request, path)
|
|
):
|
|
return JSONResponse(status_code=401, content={"detail": "Unauthorized"})
|
|
return await call_next(request)
|
|
|
|
|
|
@app.middleware("http")
|
|
async def _token_auth_seam(request: Request, call_next):
|
|
"""Outermost auth seam: bearer-token auth for opted-in routes (registered LAST = runs FIRST).
|
|
|
|
A registered token route is owned here — authenticate, attach the principal
|
|
+ ``token_authenticated`` so downstream gates skip enforcement. Non-token
|
|
routes pass through untouched.
|
|
"""
|
|
from hermes_cli.dashboard_auth.token_auth import token_auth_middleware
|
|
return await token_auth_middleware(request, call_next)
|
|
|
|
|
|
_DASHBOARD_HEALTH_WINDOW_SECONDS = 300.0
|
|
|
|
|
|
class DashboardHealth:
|
|
"""Dashboard-process health: rolling unhandled-error/5xx window + periodic self-test result.
|
|
|
|
Feeds ``components`` on the PUBLIC ``/api/status``, so :meth:`snapshot`
|
|
exports counts and enums only — never ``last_error_type``/``last_error_path``.
|
|
"""
|
|
|
|
def __init__(self, window_seconds: float = _DASHBOARD_HEALTH_WINDOW_SECONDS) -> None:
|
|
self.window_seconds = window_seconds
|
|
self._error_times: "deque[float]" = deque(maxlen=256)
|
|
self.last_error_type: Optional[str] = None
|
|
self.last_error_path: Optional[str] = None # internal-only, never serialized
|
|
self.last_error_at: Optional[float] = None
|
|
self.selftest_status: str = "unknown" # unknown | ok | failing
|
|
self.selftest_http_status: Optional[int] = None
|
|
self.selftest_at: Optional[float] = None
|
|
|
|
def record_error(self, exc_type: str, path: str) -> None:
|
|
now = time.time()
|
|
self._error_times.append(now)
|
|
self.last_error_type = exc_type
|
|
self.last_error_path = path
|
|
self.last_error_at = now
|
|
|
|
def record_selftest(self, passed: bool, http_status: Optional[int]) -> None:
|
|
self.selftest_status = "ok" if passed else "failing"
|
|
self.selftest_http_status = http_status
|
|
self.selftest_at = time.time()
|
|
|
|
def recent_error_count(self) -> int:
|
|
cutoff = time.time() - self.window_seconds
|
|
while self._error_times and self._error_times[0] < cutoff:
|
|
self._error_times.popleft()
|
|
return len(self._error_times)
|
|
|
|
def snapshot(self) -> Dict[str, Any]:
|
|
"""Public component payload: status enum + counts + timestamps only."""
|
|
errors = self.recent_error_count()
|
|
status = "degraded" if (errors or self.selftest_status == "failing") else "ok"
|
|
return {
|
|
"status": status,
|
|
"recent_unhandled_errors": errors,
|
|
"last_error_at": self.last_error_at,
|
|
"selftest": self.selftest_status,
|
|
}
|
|
|
|
|
|
DASHBOARD_HEALTH = DashboardHealth()
|
|
|
|
|
|
@app.middleware("http")
|
|
async def _dashboard_health_middleware(request: Request, call_next):
|
|
"""Outermost middleware (registered last): count unhandled exceptions and 5xx; re-raises, never alters."""
|
|
try:
|
|
response = await call_next(request)
|
|
except Exception as exc:
|
|
DASHBOARD_HEALTH.record_error(type(exc).__name__, request.url.path)
|
|
raise
|
|
if response.status_code >= 500:
|
|
DASHBOARD_HEALTH.record_error(f"http_{response.status_code}", request.url.path)
|
|
return response
|
|
|
|
|
|
# Authenticated-route self-test: one in-process request per minute against a
|
|
# cheap DB-touching route, catching "liveness fine but every authed request 500s".
|
|
_DASHBOARD_SELFTEST_INTERVAL_SECONDS = 60.0
|
|
_DASHBOARD_SELFTEST_ROUTE = "/api/sessions?limit=1"
|
|
|
|
|
|
async def _dashboard_selftest_once() -> None:
|
|
"""Run one authenticated in-process self-test request and record it."""
|
|
try:
|
|
import httpx
|
|
except ImportError:
|
|
return # optional dependency — leave status "unknown"
|
|
try:
|
|
# Loopback base_url so the Host-header middleware accepts the request.
|
|
async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://127.0.0.1") as client:
|
|
resp = await client.get(_DASHBOARD_SELFTEST_ROUTE, headers={_SESSION_HEADER_NAME: _SESSION_TOKEN})
|
|
DASHBOARD_HEALTH.record_selftest(resp.status_code == 200, resp.status_code)
|
|
except Exception:
|
|
DASHBOARD_HEALTH.record_selftest(False, None)
|
|
|
|
|
|
async def _dashboard_selftest_loop() -> None:
|
|
"""Periodic self-test driver started from the lifespan."""
|
|
try:
|
|
import httpx # noqa: F401
|
|
except ImportError:
|
|
_log.debug("httpx unavailable — dashboard self-test disabled")
|
|
return
|
|
while True:
|
|
await asyncio.sleep(_DASHBOARD_SELFTEST_INTERVAL_SECONDS)
|
|
# OAuth-gated binds don't honour the session token; the probe would false-alarm 401.
|
|
if getattr(app.state, "auth_required", False):
|
|
continue
|
|
await _dashboard_selftest_once()
|
|
|
|
|
|
|
|
|
|
# Action registries/spawner are owned by web_server_gateway; routers and tests reach them
|
|
# there, so this module reads them through the module too (one patch seam).
|
|
from hermes_cli import web_server_gateway as _gateway_mod # noqa: E402
|
|
from hermes_cli.web_server_gateway import _ACTION_LOG_FILES, _terminate_desktop_managed_gateway # noqa: E402
|
|
from hermes_cli.web_server_sessions import _auto_archive_ticker_loop # noqa: E402
|
|
from hermes_cli.web_server_chat import PTY_REGISTRY # noqa: E402
|
|
from hermes_cli.web_server_dashboard import ( # noqa: E402
|
|
_discover_dashboard_plugins, _mount_plugin_api_routes, mount_spa,
|
|
)
|
|
|
|
|
|
_GATEWAY_HEALTH_URL = os.getenv("GATEWAY_HEALTH_URL")
|
|
_GATEWAY_HEALTH_TIMEOUT_MAX = 1.0
|
|
try:
|
|
_GATEWAY_HEALTH_TIMEOUT = float(os.getenv("GATEWAY_HEALTH_TIMEOUT", "1"))
|
|
except (ValueError, TypeError):
|
|
_log.warning(
|
|
"Invalid GATEWAY_HEALTH_TIMEOUT value %r — using default 1.0s",
|
|
os.getenv("GATEWAY_HEALTH_TIMEOUT"),
|
|
)
|
|
_GATEWAY_HEALTH_TIMEOUT = 1.0
|
|
if _GATEWAY_HEALTH_TIMEOUT <= 0:
|
|
_log.warning(
|
|
"Invalid non-positive GATEWAY_HEALTH_TIMEOUT value %.3fs — using default 1.0s",
|
|
_GATEWAY_HEALTH_TIMEOUT,
|
|
)
|
|
_GATEWAY_HEALTH_TIMEOUT = 1.0
|
|
elif _GATEWAY_HEALTH_TIMEOUT > _GATEWAY_HEALTH_TIMEOUT_MAX:
|
|
_log.warning(
|
|
"Capping GATEWAY_HEALTH_TIMEOUT %.3fs to %.3fs for dashboard liveness probes",
|
|
_GATEWAY_HEALTH_TIMEOUT,
|
|
_GATEWAY_HEALTH_TIMEOUT_MAX,
|
|
)
|
|
_GATEWAY_HEALTH_TIMEOUT = _GATEWAY_HEALTH_TIMEOUT_MAX
|
|
|
|
|
|
_MANAGED_FILE_MAX_BYTES = 100 * 1024 * 1024
|
|
_FS_DATA_URL_MAX_BYTES = 16 * 1024 * 1024
|
|
# Multipart uploads stream to a temp file in fixed chunks and rename into
|
|
# place: constant memory, no base64 inflation, no proxy body-size 502s (NS-501).
|
|
_UPLOAD_CHUNK_BYTES = 1024 * 1024
|
|
|
|
# Stable install identity for /api/status: one uuid4 hex per physical install,
|
|
# persisted under the ROOT Hermes home (not the profile HERMES_HOME) so every
|
|
# profile reports the same id and the desktop can collapse duplicate roster rows
|
|
# for one backend. Must never change across restarts, so cached per process.
|
|
_INSTALL_ID_CACHE: Dict[str, Optional[str]] = {"root": None, "value": None}
|
|
|
|
|
|
def get_install_id() -> Optional[str]:
|
|
"""Process-lifetime-cached stable install id."""
|
|
return _shared_get_install_id(cache=_INSTALL_ID_CACHE)
|
|
|
|
|
|
# Serializes config.yaml read-modify-write cycles for handlers on worker threads
|
|
# (asyncio.to_thread): config.py's _CONFIG_LOCK covers each load/save call, not
|
|
# the span between them, so two off-loop updates could drop each other's writes.
|
|
# RLock so nested helpers that also take it can't self-deadlock.
|
|
_CONFIG_MUTATION_LOCK = threading.RLock()
|
|
|
|
# A finished ``gateway-restart`` child does not mean the gateway is back (it
|
|
# exits once the restart is handed off), so in-flight reuse stops coalescing
|
|
# exactly when a stale frontend re-fires every few seconds (#89034: 77 restarts,
|
|
# state.db corrupted mid-FTS5-write). MAINTAINER DECISION: a fixed window, not
|
|
# "until healthy" — a gateway that never returns must not leave the action
|
|
# inert. 10s is above the ~3.5s storm spacing and below an operator's retry.
|
|
GATEWAY_RESTART_COOLDOWN_SECONDS = 10.0
|
|
|
|
# ``(monotonic spawn time, Popen, command)`` of the last restart. Deliberately
|
|
# NOT read from ``_ACTION_PROCS``: entries there vanish when the child exits.
|
|
_LAST_GATEWAY_RESTART: Optional[Tuple[float, subprocess.Popen, Tuple[str, ...]]] = None
|
|
|
|
|
|
def _spawn_gateway_restart(profile: Optional[str] = None) -> Tuple[subprocess.Popen, bool]:
|
|
"""Spawn ``hermes gateway restart``, reusing an in-flight or recent restart.
|
|
|
|
Concurrent children race each other on the kill-and-start path, so a live
|
|
child is reused; requests within ``GATEWAY_RESTART_COOLDOWN_SECONDS`` for the
|
|
same profile coalesce onto the last spawn too (#89034). This process stops
|
|
nothing itself: the child decides whether the restart is even allowed
|
|
(``_cmd_restart``'s multiplexer guard), so the orphan reap (#77276) runs
|
|
there, after that guard — reaping here killed the profile's gateway and then
|
|
the child refused to start a replacement (#125394). Returns ``(proc, reused)``.
|
|
"""
|
|
global _LAST_GATEWAY_RESTART
|
|
|
|
subcommand = _gateway_mod._gateway_subcommand(profile, "restart")
|
|
existing = _gateway_mod._ACTION_PROCS.get("gateway-restart")
|
|
if existing is not None and existing.poll() is None:
|
|
existing_command = _gateway_mod._ACTION_COMMANDS.get("gateway-restart")
|
|
if existing_command is None or existing_command == tuple(subcommand):
|
|
return existing, True
|
|
raise RuntimeError("gateway restart already in progress for another profile")
|
|
|
|
recent = _LAST_GATEWAY_RESTART
|
|
if recent is not None:
|
|
spawned_at, recent_proc, recent_command = recent
|
|
age = time.monotonic() - spawned_at if recent_command == tuple(subcommand) else None
|
|
if age is not None and age < GATEWAY_RESTART_COOLDOWN_SECONDS:
|
|
_log.info(
|
|
"Coalescing gateway restart: one was started %.1fs ago "
|
|
"(pid %s) and the gateway may still be coming back; not "
|
|
"spawning another (#89034).",
|
|
age,
|
|
getattr(recent_proc, "pid", "?"),
|
|
)
|
|
return recent_proc, True
|
|
|
|
proc = _gateway_mod._spawn_hermes_action(subcommand, "gateway-restart")
|
|
_LAST_GATEWAY_RESTART = (time.monotonic(), proc, tuple(subcommand))
|
|
return proc, False
|
|
|
|
|
|
# Collapses repeated identical ElevenLabs voice-list failures (the desktop
|
|
# re-polls on every settings focus) to one log line; re-arms on success or a
|
|
# changed signature.
|
|
_voice_list_last_error: Optional[str] = None
|
|
|
|
|
|
def _voice_list_error_logged_once(signature: Optional[str]) -> bool:
|
|
"""True if ``signature`` is new and should be logged now; ``None`` clears the latch."""
|
|
global _voice_list_last_error
|
|
if signature is None:
|
|
_voice_list_last_error = None
|
|
return False
|
|
if signature == _voice_list_last_error:
|
|
return False
|
|
_voice_list_last_error = signature
|
|
return True
|
|
|
|
|
|
_ACTION_LOG_FILES.setdefault("computer-use-grant", "action-computer-use-grant.log")
|
|
|
|
# Cache discovered plugins per-process (refresh on explicit re-scan).
|
|
_dashboard_plugins_cache: Optional[list] = None
|
|
|
|
|
|
def _get_dashboard_plugins(force_rescan: bool = False) -> list:
|
|
global _dashboard_plugins_cache
|
|
stale = _dashboard_plugins_cache is None or force_rescan or any(
|
|
not Path(p["_dir"]).is_dir() for p in _dashboard_plugins_cache
|
|
)
|
|
if stale:
|
|
_dashboard_plugins_cache = _discover_dashboard_plugins()
|
|
return _dashboard_plugins_cache
|
|
|
|
|
|
# Router mounting. ORDER IS ROUTE-MATCHING ORDER: literal paths must land before
|
|
# templated siblings (e.g. /api/sessions/bulk-delete before /api/sessions/{id}).
|
|
from hermes_cli.web_routers import ( # noqa: E402
|
|
files as _files_routes,
|
|
git as _git_routes,
|
|
local_models as _local_models_routes,
|
|
status as _status_routes,
|
|
actions as _actions_routes,
|
|
audio as _audio_routes,
|
|
display as _display_routes,
|
|
sessions as _sessions_routes,
|
|
profiles as _profiles_routes,
|
|
memory_providers as _memory_providers_routes,
|
|
config_env as _config_env_routes,
|
|
models as _models_routes,
|
|
messaging as _messaging_routes,
|
|
oauth as _oauth_routes,
|
|
cron as _cron_routes,
|
|
mcp as _mcp_routes,
|
|
ops as _ops_routes,
|
|
skills as _skills_routes,
|
|
tools as _tools_routes,
|
|
analytics as _analytics_routes,
|
|
chat_ws as _chat_ws_routes,
|
|
chat_workspaces as _chat_workspaces_routes,
|
|
dashboard_ui as _dashboard_ui_routes,
|
|
)
|
|
|
|
app.include_router(_files_routes.router)
|
|
app.include_router(_git_routes.router)
|
|
app.include_router(_local_models_routes.router)
|
|
app.include_router(_status_routes.router)
|
|
app.include_router(_actions_routes.router)
|
|
app.include_router(_audio_routes.router)
|
|
app.include_router(_display_routes.router)
|
|
app.include_router(_actions_routes.status_router)
|
|
app.include_router(_sessions_routes.list_router)
|
|
app.include_router(_profiles_routes.sessions_router)
|
|
app.include_router(_sessions_routes.search_router)
|
|
app.include_router(_memory_providers_routes.router)
|
|
app.include_router(_config_env_routes.config_router)
|
|
app.include_router(_models_routes.router)
|
|
app.include_router(_config_env_routes.router)
|
|
app.include_router(_messaging_routes.router)
|
|
app.include_router(_oauth_routes.router)
|
|
app.include_router(_sessions_routes.manage_router)
|
|
app.include_router(_status_routes.logs_router)
|
|
app.include_router(_cron_routes.router)
|
|
app.include_router(_mcp_routes.router)
|
|
app.include_router(_ops_routes.router)
|
|
app.include_router(_skills_routes.hub_router)
|
|
app.include_router(_profiles_routes.router)
|
|
app.include_router(_skills_routes.router)
|
|
app.include_router(_tools_routes.router)
|
|
app.include_router(_analytics_routes.router)
|
|
app.include_router(_chat_ws_routes.router)
|
|
app.include_router(_chat_workspaces_routes.router)
|
|
app.include_router(_dashboard_ui_routes.router)
|
|
|
|
# Plugin API routes and the dashboard auth routes (/login, /auth/*, /api/auth/*)
|
|
# mount before the SPA catch-all so /{full_path:path} doesn't swallow them. Auth
|
|
# routes are always mounted — the gate middleware decides enforcement.
|
|
_mount_plugin_api_routes()
|
|
from hermes_cli.dashboard_auth.routes import router as _dashboard_auth_router # noqa: E402
|
|
|
|
app.include_router(_dashboard_auth_router)
|
|
mount_spa(app)
|
|
|
|
|
|
def _no_auth_provider_message(host: str) -> str:
|
|
"""Actionable SystemExit text for a gated bind with no registered auth provider.
|
|
|
|
Names the exact trigger: on a loopback bind the ONLY trigger is
|
|
dashboard.public_url, so print the offending URL and the remove-it exit.
|
|
Bundled providers expose ``LAST_SKIP_REASON`` so an installed-but-
|
|
unconfigured provider is not reported as merely "no providers".
|
|
"""
|
|
skip_reasons: list[str] = []
|
|
try:
|
|
from plugins.dashboard_auth import nous as _nous_plugin
|
|
|
|
if _nous_plugin.LAST_SKIP_REASON:
|
|
skip_reasons.append(f" • nous: {_nous_plugin.LAST_SKIP_REASON}")
|
|
except Exception:
|
|
pass
|
|
|
|
if host in _LOOPBACK_HOST_VALUES:
|
|
public_url = ""
|
|
try:
|
|
from hermes_cli.dashboard_auth.prefix import resolve_public_url
|
|
|
|
public_url = resolve_public_url()
|
|
except Exception:
|
|
pass
|
|
gate_reason = (
|
|
f"dashboard.public_url is set to "
|
|
f"{public_url or '<a non-loopback URL>'} — an "
|
|
f"operator-declared external URL engages the auth gate "
|
|
f"even on a loopback bind"
|
|
)
|
|
fix_hint = (
|
|
"If this dashboard should be LOCAL-ONLY (no reverse "
|
|
"proxy), remove dashboard.public_url from config.yaml "
|
|
"(and unset HERMES_DASHBOARD_PUBLIC_URL) to restore the "
|
|
"unauthenticated loopback mode.\n"
|
|
)
|
|
else:
|
|
gate_reason = f"the auth gate engages on non-loopback binds ({host})"
|
|
fix_hint = ""
|
|
|
|
fix_hint += (
|
|
"Configure an auth provider before exposing the dashboard:\n"
|
|
" • Password: set dashboard.basic_auth.username + "
|
|
"password_hash in config.yaml\n"
|
|
" (hash with: python -c \"from "
|
|
"plugins.dashboard_auth.basic import hash_password; "
|
|
"print(hash_password('your-password'))\")\n"
|
|
" • OAuth: run `hermes dashboard register` (Nous Portal) or "
|
|
"install a DashboardAuthProvider plugin.\n"
|
|
"There is no unauthenticated public-dashboard option. For "
|
|
"local-only use, bind 127.0.0.1 and leave dashboard.public_url "
|
|
"unset; a configured external public URL requires auth even "
|
|
"when a local reverse proxy reaches a loopback backend."
|
|
)
|
|
# Credentials exist but the bundled provider is disabled (#54489). Basic
|
|
# auth needs a username AND a credential; a half-configured block is silent.
|
|
try:
|
|
from hermes_cli.config import load_config as _load_cfg
|
|
from hermes_cli.plugins_cmd import _BASIC_AUTH_PLUGIN_KEYS
|
|
|
|
cfg = _load_cfg()
|
|
ba = (cfg.get("dashboard") or {}).get("basic_auth") or {}
|
|
disabled = (cfg.get("plugins") or {}).get("disabled") or []
|
|
has_creds = bool(ba.get("username")) and bool(ba.get("password_hash") or ba.get("password"))
|
|
if has_creds and (set(disabled) & _BASIC_AUTH_PLUGIN_KEYS):
|
|
fix_hint = (
|
|
"The 'basic' dashboard-auth plugin is in "
|
|
"plugins.disabled but dashboard.basic_auth is "
|
|
"configured.\n"
|
|
"Remove 'basic' from plugins.disabled (or run "
|
|
"`hermes plugins enable basic`), then restart the "
|
|
"dashboard.\n\n"
|
|
) + fix_hint
|
|
except Exception:
|
|
pass
|
|
msg = (
|
|
f"Refusing to bind dashboard to {host} — {gate_reason}, "
|
|
f"but no auth providers are registered.\n\n"
|
|
)
|
|
if skip_reasons:
|
|
msg += "Bundled providers reported these issues:\n" + "\n".join(skip_reasons) + "\n\n"
|
|
return msg + fix_hint
|
|
|
|
|
|
def _configure_auth_gate(
|
|
host: str,
|
|
allow_public: bool,
|
|
ssh_session_token: Optional[str],
|
|
ssh_owner_nonce: Optional[str],
|
|
) -> None:
|
|
"""Resolve the trusted public hosts + auth-gate flag onto ``app.state``.
|
|
|
|
Fails closed (``SystemExit`` with an actionable message) when the gate
|
|
engages but no dashboard auth provider is registered.
|
|
"""
|
|
# dashboard.public_url is also the exact Host/Origin trust declaration for
|
|
# reverse-proxy deployments; resolved once so middleware never reloads
|
|
# config. A non-loopback public hostname engages the gate even on a loopback
|
|
# backend, else the SPA's local session token becomes remotely reachable.
|
|
app.state.trusted_public_hosts = _dashboard_public_hosts()
|
|
# auth_required drives middleware, SPA-token injection, WS auth, the
|
|
# startup refusal, the gate-on banner and uvicorn proxy_headers.
|
|
if _desktop_loopback_auth_exempt(host, ssh_session_token, ssh_owner_nonce):
|
|
# public_url describes the operator's PUBLIC deployment, not this
|
|
# Desktop-owned loopback backend (#96490), which authenticates with the
|
|
# per-spawn session token the ticket-only gate would refuse.
|
|
app.state.auth_required = should_require_auth(host)
|
|
_log.info(
|
|
"Desktop-owned loopback backend: dashboard.public_url does not "
|
|
"engage the ticket gate for this process; the public deployment "
|
|
"keeps its own gate.",
|
|
)
|
|
else:
|
|
app.state.auth_required = should_require_dashboard_auth(host, app.state.trusted_public_hosts)
|
|
|
|
# ``--insecure`` no longer disables the gate (June 2026 hermes-0day
|
|
# hardening); warn that it is a no-op rather than silently ignore it.
|
|
if allow_public and host not in _LOOPBACK_HOST_VALUES:
|
|
_log.warning(
|
|
"--insecure no longer bypasses dashboard authentication. A "
|
|
"non-loopback bind (%s) now ALWAYS requires an auth provider "
|
|
"(OAuth or the bundled password provider). Configure one — see "
|
|
"below — or bind to 127.0.0.1 and reach it over an SSH tunnel / "
|
|
"Tailscale.", host,
|
|
)
|
|
|
|
if app.state.auth_required:
|
|
# No escape hatch serves a gated dashboard without a provider.
|
|
from hermes_cli.dashboard_auth import list_providers
|
|
if not list_providers():
|
|
raise SystemExit(_no_auth_provider_message(host))
|
|
_log.info(
|
|
"Dashboard binding to %s with auth gate enabled. Providers: %s",
|
|
host,
|
|
", ".join(p.name for p in list_providers()),
|
|
)
|
|
|
|
|
|
def _build_uvicorn_server(host: str, port: int, *, ssh_isolated: bool = False):
|
|
"""Build the uvicorn ``Config`` + ``Server`` for this bind (reads ``app.state.auth_required``).
|
|
|
|
uvicorn.Server is driven directly (not uvicorn.run) so startup is split from
|
|
the main loop: after startup() the socket is bound and held by uvicorn, so the
|
|
OS-assigned port can be read with no pre-bind-then-close TOCTOU. Explicit
|
|
taken ports are caught by the #93608 preflight probe; uvicorn's own bind
|
|
error stays the fallback for races.
|
|
"""
|
|
import uvicorn
|
|
|
|
# WS keepalive ping runs ON the agent event loop; a GIL-holding worker call
|
|
# can starve it for minutes, so uvicorn misses the pong and drops a healthy
|
|
# local socket (#53773/#48445/#50005). The ping only detects half-open
|
|
# connections (proxy 524, dropped tunnels), impossible on loopback where a
|
|
# dead client sends a real FIN/RST -> WebSocketDisconnect. So: no ping on
|
|
# loopback; non-loopback sits behind a Cloudflare Tunnel (~100s idle) and
|
|
# keeps a config-driven cadence (dashboard.ws_ping_interval/_timeout,
|
|
# #79635) defaulting to 20/20.
|
|
_is_loopback = host in _LOOPBACK_HOST_VALUES
|
|
try:
|
|
_dash_cfg = load_config().get("dashboard") or {}
|
|
except Exception:
|
|
_dash_cfg = {}
|
|
|
|
def _ws_ping_setting(key: str, default: float = 20.0) -> float:
|
|
try:
|
|
return float(_dash_cfg.get(key, default))
|
|
except (TypeError, ValueError):
|
|
return default
|
|
|
|
# A Desktop-owned SSH-isolated backend is loopback on the SERVER, but the client sits at the far
|
|
# end of a tunnel: the local socket stays healthy while the laptop sleeps, so only a slow WS ping
|
|
# notices the half-open tunnel (#101626). Its client count is tracked at the ASGI boundary so
|
|
# the idle watchdog can retire the backend once nothing is connected.
|
|
served_app = app
|
|
ping_interval, ping_timeout = (None, None) if _is_loopback else (
|
|
_ws_ping_setting("ws_ping_interval"), _ws_ping_setting("ws_ping_timeout"))
|
|
if ssh_isolated:
|
|
from hermes_cli.web_server_idle_exit import (
|
|
TUNNEL_WS_PING_INTERVAL_S, TUNNEL_WS_PING_TIMEOUT_S, IdleClientTracker, wrap_asgi_with_ws_tracking)
|
|
app.state.ssh_isolated_clients = IdleClientTracker()
|
|
served_app = wrap_asgi_with_ws_tracking(app, app.state.ssh_isolated_clients)
|
|
ping_interval, ping_timeout = TUNNEL_WS_PING_INTERVAL_S, TUNNEL_WS_PING_TIMEOUT_S
|
|
|
|
config = uvicorn.Config(
|
|
served_app, host=host, port=port, log_level="warning",
|
|
# Off by default so _ws_client_is_allowed sees the real peer, not
|
|
# X-Forwarded-For. Gated mode runs behind a TLS terminator and needs
|
|
# X-Forwarded-Proto for cookie Secure flags.
|
|
proxy_headers=bool(app.state.auth_required),
|
|
# Loopback-only unless the operator trusts a bounded upstream proxy, so
|
|
# spoofed X-Forwarded-* from arbitrary callers is never honoured.
|
|
forwarded_allow_ips=_dashboard_forwarded_allow_ips(_dash_cfg),
|
|
ws_ping_interval=ping_interval,
|
|
ws_ping_timeout=ping_timeout,
|
|
ws_max_size=_DESKTOP_ATTACHMENT_WS_MAX_BYTES,
|
|
)
|
|
return config, uvicorn.Server(config)
|
|
|
|
|
|
def _best_effort(what: str, fn) -> None:
|
|
"""Run a best-effort startup step; any failure (import included) is a debug line."""
|
|
try:
|
|
fn()
|
|
except Exception as exc:
|
|
_log.debug("%s skipped: %s", what, exc)
|
|
|
|
|
|
def _publish_host_rendezvous(host: str, port: int) -> None:
|
|
"""Publish this backend's host record: ``ROLE_SERVE`` for the machine-level owner,
|
|
``ROLE_DESKTOP_SERVE`` for a Desktop-owned child."""
|
|
# Desktop-spawned backends (flag + per-spawn credential; the bare flag is inherited by every
|
|
# Desktop shell) are loopback, random-port and per-profile. Recording one as the HOST owner
|
|
# made a later independently supervised `dashboard --host 0.0.0.0 --port N` refuse behind
|
|
# the private child on every restart (#119824): the attach/refuse ladder reads ROLE_SERVE
|
|
# only. They still publish under their own role so `hermes plugins install` from a terminal
|
|
# can reach the backend hosting the open chats on a Desktop-only box (#119644).
|
|
from gateway import host_rendezvous as hr
|
|
|
|
desktop_child = is_desktop_owned_backend()
|
|
role = hr.ROLE_DESKTOP_SERVE if desktop_child else hr.ROLE_SERVE
|
|
|
|
outcome, error = hr.claim_host_lock(role)
|
|
if outcome is hr.HostLockOutcome.COULD_NOT_OPEN:
|
|
_log.warning(
|
|
"Host backend lock could not be opened (%s); this backend is not discoverable. "
|
|
"This is NOT another backend holding it.", error)
|
|
return
|
|
if outcome is hr.HostLockOutcome.HELD_BY_OTHER:
|
|
owner = hr.read_record(role)
|
|
if desktop_child:
|
|
# A second pool child is Desktop's own topology, not a conflict.
|
|
_log.debug("another Desktop backend holds the %s record (%s)", role,
|
|
hr.describe(owner) if owner else "owner unknown")
|
|
return
|
|
_log.warning(
|
|
"Another backend already owns this host (%s); this one bound anyway "
|
|
"(observe-only). Multiplex-only expects exactly one backend per host.",
|
|
hr.describe(owner) if owner else "owner unknown",
|
|
)
|
|
return
|
|
app.state.host_role = role
|
|
hr.publish_record(
|
|
role,
|
|
host=host,
|
|
port=port,
|
|
profiles=hr.served_profiles(),
|
|
# The live session token, so an attaching client of the same OS user can
|
|
# authenticate even when the backend is gated and `GET /` withholds it.
|
|
token=_SESSION_TOKEN,
|
|
)
|
|
# SIGTERM included: it is the normal stop, and it does not run atexit here.
|
|
hr.cleanup_on_exit(role)
|
|
|
|
|
|
def _on_server_started(
|
|
server,
|
|
*,
|
|
host: str,
|
|
port: int,
|
|
headless: bool,
|
|
isolated: bool,
|
|
open_browser: bool,
|
|
initial_profile: str,
|
|
start_mcp_discovery_after_bind: bool,
|
|
) -> None:
|
|
"""Post-bind arming on the serving loop right after ``server.startup()``.
|
|
|
|
Reap prior corpses, parent-death watchdog, process identity, READY
|
|
announcement, browser open, deferred MCP discovery, loop-noise filter,
|
|
loop heartbeat.
|
|
"""
|
|
# Clear corpses from a previous unclean Desktop exit (crash/SIGKILL/update
|
|
# handoff leaves an orphaned backend + its MCP subtree) before stacking a
|
|
# new tree (EMFILE / missing tabs). The watchdog only protects *this*
|
|
# process going forward.
|
|
def _reap_desktop_serves() -> None:
|
|
from hermes_cli.dashboard_procs import _reap_orphaned_desktop_local_serves
|
|
|
|
_reap_orphaned_desktop_local_serves()
|
|
|
|
def _reap_mcp_helpers() -> None:
|
|
from hermes_cli.process_identity import reap_orphaned_mcp_helpers
|
|
|
|
reap_orphaned_mcp_helpers()
|
|
|
|
if is_desktop_owned_backend():
|
|
_best_effort("orphan desktop-local serve reap", _reap_desktop_serves)
|
|
# Same sweep for stdio MCP helpers (#61514): positive identity only (spawn
|
|
# ledger + spawner provably dead); anything alive or unprovable is untouched.
|
|
_best_effort("orphan MCP helper reap", _reap_mcp_helpers)
|
|
|
|
# No-op for standalone `hermes serve` (no HERMES_PARENT_PID).
|
|
_start_parent_death_watchdog()
|
|
# SSH-isolated backends are detached from any parent on purpose (#91668); their liveness signal
|
|
# is "does a client still hold a WebSocket" (#101626).
|
|
if getattr(app.state, "ssh_isolated_clients", None) is not None:
|
|
from hermes_cli.web_server_idle_exit import DEFAULT_IDLE_GRACE_S, start_idle_watchdog
|
|
try:
|
|
grace = float((load_config().get("dashboard") or {}).get("ssh_isolated_idle_grace_s", DEFAULT_IDLE_GRACE_S))
|
|
except (TypeError, ValueError):
|
|
grace = DEFAULT_IDLE_GRACE_S
|
|
start_idle_watchdog(server, app.state.ssh_isolated_clients, grace_s=grace)
|
|
# A connected client keeps the idle watchdog quiet forever, and the host's updater may not
|
|
# restart this backend, so it retires itself (between turns) when the install moves on.
|
|
from hermes_cli.web_server_skew_exit import start_code_skew_watchdog
|
|
|
|
start_code_skew_watchdog(server)
|
|
|
|
actual_port = _read_bound_port(server, fallback=port)
|
|
app.state.bound_port = actual_port
|
|
# Published by /api/host/identity: an attaching `hermes dashboard` must never be routed to a
|
|
# headless backend (a URL with no UI behind it).
|
|
app.state.serves_spa = not headless
|
|
|
|
# Positive process identity in the machine spawn ledger (+ Windows
|
|
# kill-on-close job). Registered AFTER the bind so the entry carries the
|
|
# ACTUAL port — what lets `hermes update` relaunch a manually-started serve
|
|
# on its real endpoint (#63206).
|
|
def _register_identity() -> None:
|
|
from hermes_cli.process_identity import attach_self_to_kill_on_close_job, register_self
|
|
|
|
register_self(
|
|
"serve" if headless else "dashboard",
|
|
detail={"host": host, "port": actual_port, "profile": initial_profile or "", "isolated": isolated},
|
|
)
|
|
attach_self_to_kill_on_close_job()
|
|
|
|
_best_effort("process-identity registration", _register_identity)
|
|
|
|
# Host rendezvous (multiplex-only): the host lock + record that let a SECOND `hermes serve`
|
|
# for any profile find this process and attach instead of binding a second port. Published
|
|
# after the bind so the record carries the real port, and beside — not instead of — the
|
|
# spawn-ledger entry above, which Desktop's attach ladder reads.
|
|
_best_effort("host rendezvous publish", lambda: _publish_host_rendezvous(host, actual_port))
|
|
|
|
_write_dashboard_ready_file(actual_port)
|
|
# Port-discovery sentinel parsed by the Desktop spawn. Written to fd 1:
|
|
# tui_gateway.server redirects sys.stdout to stderr at import, and the
|
|
# Desktop watches child.stdout (#96282). A headless `serve` announces the
|
|
# neutral token FIRST and the legacy HERMES_DASHBOARD_READY one after it:
|
|
# a packaged Desktop artifact whose parser predates the neutral token
|
|
# (#60772) still matches the legacy sentinel, while current parsers match
|
|
# either. The legacy `dashboard` backend keeps its own single token.
|
|
if headless:
|
|
_write_machine_sentinel_line(f"HERMES_BACKEND_READY port={actual_port}")
|
|
_write_machine_sentinel_line(f"HERMES_DASHBOARD_READY port={actual_port}")
|
|
else:
|
|
_write_machine_sentinel_line(f"HERMES_DASHBOARD_READY port={actual_port}")
|
|
if headless:
|
|
# Auth-gated JSON-RPC/WS only — announce the bind, not a URL. flush:
|
|
# a piped stdout otherwise surfaces this minutes after the sentinel.
|
|
print(f" Hermes backend listening on {host}:{actual_port}", flush=True)
|
|
else:
|
|
print(f" Hermes Web UI → http://{host}:{actual_port}")
|
|
_maybe_open_browser(host, actual_port, open_browser, initial_profile)
|
|
|
|
if start_mcp_discovery_after_bind:
|
|
# Desktop `serve`: the ~350ms `mcp` SDK import holds the GIL while the
|
|
# renderer does its WS handshake + first hydration reads, so arm it one
|
|
# second later when the shell is painted and idle. An agent build inside
|
|
# that second fires the deferred start itself (wait_for_mcp_discovery).
|
|
try:
|
|
from hermes_cli.mcp_startup import defer_background_mcp_discovery
|
|
|
|
defer_background_mcp_discovery(
|
|
logger=_log,
|
|
thread_name="dashboard-mcp-discovery",
|
|
delay=_DESKTOP_MCP_DISCOVERY_DELAY_S,
|
|
)
|
|
except Exception:
|
|
_log.debug("Deferred MCP discovery arm failed", exc_info=True)
|
|
|
|
# Collapse the peer-hangup teardown flood (#50005): 50+ identical WinError
|
|
# 10054 tracebacks per Desktop disconnect become one debug line.
|
|
def _install_noise_filter() -> None:
|
|
from tui_gateway.loop_noise import install_loop_noise_filter
|
|
|
|
install_loop_noise_filter(asyncio.get_running_loop())
|
|
|
|
_best_effort("loop noise filter install", _install_noise_filter)
|
|
|
|
# Loop heartbeat watchdog (CF-1): a 2s call_later tick whose drift equals
|
|
# any GIL stall, so a stalled-loop WS drop is diagnosable from the log.
|
|
# call_later (not a task) dies with the loop — nothing to cancel.
|
|
_hb_interval = 2.0
|
|
_hb_stall_threshold = 5.0
|
|
_hb_loop = asyncio.get_running_loop()
|
|
|
|
def _loop_heartbeat(expected: float) -> None:
|
|
now = _hb_loop.time()
|
|
drift = now - expected
|
|
if drift > _hb_stall_threshold:
|
|
_log.warning("event loop stalled %.1fs (GIL pressure suspected)", drift)
|
|
_hb_loop.call_later(_hb_interval, _loop_heartbeat, now + _hb_interval)
|
|
|
|
_hb_loop.call_later(_hb_interval, _loop_heartbeat, _hb_loop.time() + _hb_interval)
|
|
|
|
|
|
def _windows_serve_loop_factory(config):
|
|
"""Loop factory for serve on Windows: always a selector loop.
|
|
|
|
uvicorn 0.41's ``asyncio_loop_factory`` returns ProactorEventLoop on
|
|
win32, on which uvicorn's socket stack binds-but-never-accepts (READY
|
|
prints, then WinError 10014 accept failures, exit 1, desktop
|
|
ECONNREFUSED — #120164, regression of #50641). A factory that already
|
|
yields selector loops (older uvicorn, explicit ``--loop``) passes through.
|
|
"""
|
|
factory = config.get_loop_factory()
|
|
if factory is None or factory is asyncio.ProactorEventLoop: # type: ignore[attr-defined]
|
|
return asyncio.SelectorEventLoop
|
|
return factory
|
|
|
|
|
|
def _run_serve(serve, config, host: str, port: int) -> None:
|
|
"""Drive ``serve()`` on the loop uvicorn expects.
|
|
|
|
POSIX keeps ``asyncio.run`` (already a SelectorEventLoop / uvloop). On
|
|
Windows ``asyncio.run`` defaults to a ProactorEventLoop, on which uvicorn
|
|
binds a socket that never accepts (#50641), so mirror uvicorn's own runner +
|
|
loop factory there (hand-installed selector policy for uvicorn < 0.36).
|
|
Ctrl+C -> clean return; probe-to-bind port race -> sentinel + exit code.
|
|
"""
|
|
runner = asyncio.run
|
|
runner_kwargs: dict = {}
|
|
if sys.platform == "win32":
|
|
# Resolved FIRST; the serve call is outside this try so genuine
|
|
# serve-time errors (port in use) propagate instead of double-running.
|
|
try:
|
|
from uvicorn._compat import asyncio_run as runner
|
|
|
|
runner_kwargs = {"loop_factory": _windows_serve_loop_factory(config)}
|
|
except Exception:
|
|
runner = asyncio.run
|
|
runner_kwargs = {}
|
|
try:
|
|
asyncio.set_event_loop_policy(
|
|
asyncio.WindowsSelectorEventLoopPolicy() # type: ignore[attr-defined]
|
|
)
|
|
except Exception:
|
|
pass
|
|
|
|
# ``capture_signals()`` re-raises the captured signal after graceful
|
|
# shutdown; console Ctrl+C lands as KeyboardInterrupt = clean exit.
|
|
# (Re-raised SIGTERM/SIGBREAK keep their terminate disposition.)
|
|
try:
|
|
runner(serve(), **runner_kwargs)
|
|
except KeyboardInterrupt:
|
|
return
|
|
except SystemExit as exc:
|
|
# Probe-to-bind race (#93608): uvicorn's bind_socket() exits 1 — re-check
|
|
# and translate a confirmed conflict into the sentinel + distinct code.
|
|
if exc.code == 1 and _port_bind_conflict(host, port):
|
|
_report_port_in_use(host, port)
|
|
raise SystemExit(PORT_IN_USE_EXIT_CODE) from None
|
|
raise
|
|
|
|
|
|
def start_server(
|
|
host: str = "127.0.0.1",
|
|
port: int = 9119,
|
|
open_browser: bool = True,
|
|
allow_public: bool = False,
|
|
initial_profile: str = "",
|
|
headless: bool = False,
|
|
isolated: bool = False,
|
|
ssh_session_token: Optional[str] = None,
|
|
ssh_owner_nonce: Optional[str] = None,
|
|
start_mcp_discovery_after_bind: bool = False,
|
|
):
|
|
"""Start the web UI server.
|
|
|
|
``initial_profile`` is appended to the auto-opened URL as ``?profile=<name>``
|
|
(profile alias ``<profile> dashboard``). ``headless`` is the ``serve`` path:
|
|
JSON-RPC/WS backend, no UI build, no SPA mount (``HERMES_SERVE_HEADLESS``).
|
|
``isolated`` (``--isolated``) is recorded in the spawn ledger so attach-first
|
|
discovery never adopts this process.
|
|
``ssh_session_token``/``ssh_owner_nonce`` are process-local Desktop SSH
|
|
bootstrap state, never persisted or exported to children.
|
|
``start_mcp_discovery_after_bind`` (Desktop ``serve``) defers MCP discovery
|
|
until the ready sentinel is written so its SDK import can't hold the GIL
|
|
against the pre-bind path.
|
|
"""
|
|
_apply_ssh_session_token(ssh_session_token or "")
|
|
_apply_ssh_owner_nonce(ssh_owner_nonce)
|
|
|
|
# Dashboard-mode starts don't route through main.py's `serve` path, which
|
|
# applies the same RLIMIT_NOFILE floor (policy in resource_limits, #81547).
|
|
from hermes_cli.resource_limits import apply_nofile_soft_limit
|
|
|
|
apply_nofile_soft_limit()
|
|
|
|
import uvicorn # noqa: F401 — fail fast (before any side effects) when the dashboard extra is missing
|
|
|
|
try:
|
|
from hermes_cli.nous_auth_keepalive import start_nous_auth_keepalive
|
|
|
|
start_nous_auth_keepalive()
|
|
except Exception as exc:
|
|
_log.debug("Nous auth keepalive did not start: %s", exc)
|
|
|
|
_configure_auth_gate(host, allow_public, ssh_session_token, ssh_owner_nonce)
|
|
|
|
# host_header_middleware validates Host against this (DNS rebinding,
|
|
# GHSA-ppp5-vxwm-4cf7).
|
|
app.state.bound_host = host
|
|
# The SPA bootstrap reads this so profile-less deep links (/chat?resume=<id>) inherit the
|
|
# launcher's preselected profile instead of silently running in the launch scope (#73085).
|
|
app.state.initial_profile = str(initial_profile or "")
|
|
|
|
config, server = _build_uvicorn_server(host, port, ssh_isolated=bool(ssh_session_token))
|
|
|
|
# Flush-on-kill guard (#94724): chaining SIGTERM/SIGINT handlers persist
|
|
# in-memory transcripts to state.db before shutdown. Installed BEFORE
|
|
# uvicorn's capture_signals() so uvicorn re-raises into them as the
|
|
# "original" handlers — kills outside the serve window are covered too.
|
|
try:
|
|
from tui_gateway.server import install_exit_flush_signal_handlers
|
|
|
|
install_exit_flush_signal_handlers()
|
|
except Exception as exc:
|
|
_log.debug("exit-flush signal handlers not installed: %s", exc)
|
|
|
|
# #93608: uvicorn's bind_socket() would exit 1 with a bare ERROR line,
|
|
# indistinguishable from "backend broken". Probe first so a conflict
|
|
# surfaces as the BACKEND_PORT_IN_USE sentinel + distinct exit code.
|
|
# ``--port 0`` is skipped by the probe.
|
|
if _port_bind_conflict(host, port):
|
|
_report_port_in_use(host, port)
|
|
raise SystemExit(PORT_IN_USE_EXIT_CODE)
|
|
|
|
# LAST boot step, deliberately. One host process serves every profile and this one can be asked
|
|
# for any of them via ``?profile=``, so the decision is made here instead of on the first such
|
|
# request — activation is one-way, and everything the backend had already done by then
|
|
# (idle-reaper flushes, hosted rooms, cron) stayed on single-profile assumptions. It runs after
|
|
# the keepalive / auth gate / uvicorn build because activation FREEZES ``os.environ`` as the
|
|
# launch profile's credentials, and that snapshot is the only source for launch keys with no
|
|
# ``.env`` to rebuild from (systemd ``Environment=``, ``op run``, Compose): anything injected or
|
|
# rotated by a later boot step would otherwise be invisible for the process lifetime. No-op on a
|
|
# single-profile host; `gateway.multiplex_profiles: false` is retired and no longer skips it.
|
|
try:
|
|
from tui_gateway.launch_profile_policy import activate_multi_profile_hosting_eagerly
|
|
|
|
activate_multi_profile_hosting_eagerly()
|
|
except Exception:
|
|
_log.warning("eager multi-profile activation failed", exc_info=True)
|
|
|
|
async def _serve():
|
|
# startup split from main_loop so the bound (ephemeral) port is readable.
|
|
if not config.loaded:
|
|
config.load()
|
|
server.lifespan = config.lifespan_class(config)
|
|
with server.capture_signals():
|
|
await server.startup()
|
|
if server.should_exit:
|
|
return
|
|
|
|
_on_server_started(
|
|
server,
|
|
host=host,
|
|
port=port,
|
|
headless=headless,
|
|
isolated=isolated,
|
|
open_browser=open_browser,
|
|
initial_profile=initial_profile,
|
|
start_mcp_discovery_after_bind=start_mcp_discovery_after_bind,
|
|
)
|
|
if headless:
|
|
from hermes_cli.observability.shared_metrics_startup import record_process_ready
|
|
record_process_ready("serve_boot", background=True)
|
|
|
|
await server.main_loop()
|
|
if server.started:
|
|
await server.shutdown()
|
|
|
|
_run_serve(_serve, config, host, port)
|