6948 lines
299 KiB
Python
6948 lines
299 KiB
Python
"""Gateway runner - entry point for messaging platform integrations.
|
|
|
|
Provides ``start_gateway()`` (start all configured adapters) and ``GatewayRunner`` (lifecycle).
|
|
Run via ``python -m gateway.run`` or ``python cli.py --gateway``.
|
|
"""
|
|
|
|
# IMPORTANT: hermes_bootstrap must be the very first import — UTF-8 stdio
|
|
# on Windows. No-op on POSIX. See hermes_bootstrap.py for full rationale.
|
|
try:
|
|
import hermes_bootstrap # noqa: F401
|
|
except ModuleNotFoundError:
|
|
# Partial ``hermes update`` (git reset landed, ``uv pip install -e .`` didn't) leaves the
|
|
# bootstrap unregistered; without it Windows skips UTF-8 stdio setup, POSIX is unaffected.
|
|
pass
|
|
|
|
import asyncio
|
|
import concurrent.futures
|
|
import dataclasses
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import shlex
|
|
import site
|
|
import sys
|
|
import signal
|
|
import threading
|
|
import time
|
|
import traceback
|
|
from collections import OrderedDict
|
|
from contextvars import copy_context
|
|
from pathlib import Path
|
|
from datetime import datetime
|
|
from typing import Callable, Dict, Optional, Any, List, Tuple, cast
|
|
|
|
from agent.async_utils import safe_schedule_threadsafe
|
|
from agent.conversation_compression import (
|
|
COMPACTION_DONE_STATUS,
|
|
COMPACTION_STATUS,
|
|
COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE,
|
|
IDLE_COMPACTION_STATUS_TEMPLATE,
|
|
PRE_API_COMPRESSION_STATUS_TEMPLATE,
|
|
PREFLIGHT_COMPRESSION_STATUS_TEMPLATE,
|
|
)
|
|
from agent.conversation_loop import INTERRUPT_WAITING_FOR_MODEL_PREFIX
|
|
from agent.interrupt_compat import request_hard_interrupt
|
|
from agent.turn_context import (
|
|
compression_made_progress,
|
|
)
|
|
from hermes_cli.config import _is_ssh_remote_tilde_cwd, cfg_get
|
|
from hermes_cli.fallback_config import get_fallback_chain
|
|
|
|
# Per-session AIAgent cache bounds (each agent holds LLM clients, tool schemas, memory providers);
|
|
# LRU cap + idle TTL eviction are enforced by _enforce_agent_cache_cap()/_session_expiry_watcher().
|
|
_AGENT_CACHE_MAX_SIZE = 128
|
|
_AGENT_CACHE_IDLE_TTL_SECS = 3600.0 # evict agents idle for >1h
|
|
_PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT = 30.0
|
|
# Telegram connect proves a real getUpdates round trip, so its budget must cover the
|
|
# initialize/deleteWebhook/start_polling wall deadlines plus readiness; others keep the 30s bound.
|
|
_TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT = 180.0
|
|
# Telegram's initial connect (awaited before the gateway reaches `running`) must not spend the full
|
|
# 180s: an unreachable Telegram would hold EVERY platform's serving state hostage.
|
|
_TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT = 45.0
|
|
_ADAPTER_DISCONNECT_TIMEOUT_SECS_DEFAULT = 5.0
|
|
# End reasons meaning the USER deliberately closed this thread (/new, explicit exit, /switch). Shared
|
|
# by _classify_completion_target and _resolve_async_delegation_session so they can never disagree:
|
|
# a reason the classifier "delivers" but the resolver drops would be acked and then silently lost.
|
|
_USER_BOUNDARY_END_REASONS = (
|
|
"session_reset",
|
|
"user_exit",
|
|
"session_switch",
|
|
"new_session",
|
|
)
|
|
# Bound on a single stall-notify adapter.send so a wedged transport cannot block the stall watcher
|
|
# pass; on timeout the latch stays clear and the next tick retries.
|
|
_STALL_NOTIFY_SEND_TIMEOUT_SECONDS = 15.0
|
|
_GATEWAY_PROXY_SSE_BUFFER_MAX_CHARS = 16 * 1024 * 1024
|
|
_TELEGRAM_COMMAND_MENTION_RE = re.compile(r"(?<![\w:/])/([A-Za-z0-9][A-Za-z0-9_-]*)")
|
|
_GATEWAY_HYGIENE_PLATFORM = "gateway_hygiene"
|
|
|
|
_TELEGRAM_NOISY_STATUS_RE = re.compile(
|
|
r"(" # transient/auxiliary status that should stay in logs, not gateway chats
|
|
r"auxiliary\s+.+\s+failed"
|
|
r"|compression\s+summary\s+failed"
|
|
r"|fallback\s+context\s+marker"
|
|
r"|configured\s+compression\s+model\s+.+\s+failed"
|
|
r"|no\s+auxiliary\s+llm\s+provider\s+configured"
|
|
r"|auto-lowered\s+compression\s+threshold"
|
|
# #69332 reworded the auto-lower notice to "Auto-lowered this session's
|
|
# threshold to N tokens" — keep both generations covered.
|
|
r"|auto-lowered\s+(?:this\s+)?session'?s?\s+threshold"
|
|
r"|configured\s+auxiliary\s+compression\s+provider\s+.+\s+unavailable"
|
|
r"|skipping\s+concurrent\s+compression"
|
|
r"|compacting\s+context\s+[—-]\s+summarizing\s+earlier\s+conversation"
|
|
r"|resumed\s+after\s+\d+s\s+idle\s+[—-]\s+compacting"
|
|
r"|preflight\s+compression"
|
|
r"|pre[- ]api\s+compression"
|
|
# Retry chatter replayed via _emit_status when a turn exhausts retries. The ", retrying" /
|
|
# "— compressing" anchors keep manual /compress feedback and failure notices out of the match.
|
|
r"|context\s+too\s+large\s+\(~[\d,]+\s+tokens\)\s+[—-]+\s+compressing"
|
|
r"|compressed\s+\d[\d,]*\s+(?:→|->)\s+\d[\d,]*\s+messages,\s+retrying"
|
|
r"|compressed\s+~[\d,]+\s+(?:→|->)\s+~[\d,]+\s+tokens,\s+retrying"
|
|
r"|context\s+reduced\s+to\s+[\d,]+\s+tokens\s+\(was\s+[\d,]+\),\s+retrying"
|
|
r"|session\s+compressed\s+\d+\s+times"
|
|
r"|rate\s+limited\.\s+waiting\s+\d"
|
|
r"|retrying\s+in\s+\d"
|
|
r"|max\s+retries\s+\(\d+\).*(?:trying\s+fallback|exhausted|invalid\s+responses)"
|
|
r"|stream\s+(?:drop|drop\s+mid\s+tool-call).+retry\s+\d"
|
|
r"|stale\s+connections\s+from\s+a\s+previous\s+provider\s+issue"
|
|
rf"|{re.escape(COMPACTION_DONE_STATUS)}"
|
|
r")",
|
|
re.IGNORECASE | re.DOTALL,
|
|
)
|
|
|
|
|
|
_HYGIENE_COOLDOWN_LADDER_MULTIPLIERS = (1, 3, 9)
|
|
# Ceiling on an escalated hygiene cooldown (cf. _RECONNECT_BACKOFF_CAP): with an operator-raised
|
|
# base the ladder alone reaches 9h, indistinguishable from "compaction silently switched off".
|
|
_HYGIENE_COOLDOWN_MAX_SECONDS = 3600.0
|
|
|
|
# Flat retry-after when hygiene compression is ABANDONED by turn-hold expiry (not a failure, so
|
|
# outside the streak ladder); keeps sustained traffic from spawn/hold/cancelling one every turn.
|
|
_HYGIENE_TURNHOLD_RETRY_SECONDS = 60.0
|
|
|
|
|
|
def _hygiene_cooldown_for_failure(
|
|
gateway,
|
|
session_key: str,
|
|
base_cooldown_seconds: float,
|
|
) -> float:
|
|
"""Bump the hygiene failure streak and return the escalated cooldown.
|
|
|
|
Multiplier ladder (x1, x3, x9) over the configured base, clamped to the max, so a tuned base
|
|
stays rung 1. Hygiene's per-run ``AIAgent`` is fresh (in-memory streak always 0), so the streak
|
|
lives in SQLite keyed by rotation-stable ``session_key``.
|
|
"""
|
|
streak = 1
|
|
state = None
|
|
try:
|
|
state = gateway._session_state(session_key).persistent
|
|
except Exception as exc:
|
|
logger.debug("hygiene failure streak update failed: %s", exc)
|
|
session_db = getattr(gateway, "_session_db", None)
|
|
session_db = getattr(session_db, "_db", session_db)
|
|
increment = getattr(session_db, "increment_hygiene_failure_streak", None)
|
|
if callable(increment):
|
|
try:
|
|
streak = max(1, int(increment(session_key)))
|
|
if state is not None:
|
|
state.hygiene_failure_streak = streak
|
|
except Exception as exc:
|
|
logger.debug("hygiene failure streak persist failed: %s", exc)
|
|
if state is not None:
|
|
state.hygiene_failure_streak += 1
|
|
streak = state.hygiene_failure_streak
|
|
elif state is not None:
|
|
state.hygiene_failure_streak += 1
|
|
streak = state.hygiene_failure_streak
|
|
multiplier = _HYGIENE_COOLDOWN_LADDER_MULTIPLIERS[
|
|
min(streak, len(_HYGIENE_COOLDOWN_LADDER_MULTIPLIERS)) - 1
|
|
]
|
|
return min(base_cooldown_seconds * multiplier, _HYGIENE_COOLDOWN_MAX_SECONDS)
|
|
|
|
|
|
def _reset_hygiene_failure_streak(gateway, session_key: str) -> None:
|
|
"""Clear the hygiene failure streak after a compression that reduced context.
|
|
|
|
Peeks, never get-or-creates: a no-op 0 write must not create a never-evicted ``_sessions`` row.
|
|
"""
|
|
try:
|
|
state = gateway._peek_session_state(session_key)
|
|
if state is not None:
|
|
state.persistent.hygiene_failure_streak = 0
|
|
except Exception as exc:
|
|
logger.debug("hygiene failure streak reset failed: %s", exc)
|
|
session_db = getattr(gateway, "_session_db", None)
|
|
session_db = getattr(session_db, "_db", session_db)
|
|
reset = getattr(session_db, "reset_hygiene_failure_streak", None)
|
|
if callable(reset):
|
|
try:
|
|
reset(session_key)
|
|
except Exception as exc:
|
|
logger.debug("hygiene failure streak persistent reset failed: %s", exc)
|
|
|
|
|
|
def hygiene_compaction_recovered(
|
|
*,
|
|
aborted: bool,
|
|
rotated: bool,
|
|
in_place: bool,
|
|
msg_count: int,
|
|
new_count: int,
|
|
approx_tokens: int,
|
|
new_tokens: int,
|
|
) -> bool:
|
|
"""True when a hygiene run actually recovered the session (extracted to be unit testable).
|
|
|
|
Requires all three: the compressor did not abort; the transcript was actually rewritten (the
|
|
no-op path reuses pre-compression counts, so numbers alone read as success); and the request
|
|
materially shrank per :func:`compression_made_progress` (a bare ``<`` misses row-count wins
|
|
and counts 30-50% estimate noise as one).
|
|
"""
|
|
if aborted:
|
|
return False
|
|
if not (rotated or in_place):
|
|
return False
|
|
return compression_made_progress(
|
|
msg_count, new_count, approx_tokens, new_tokens
|
|
)
|
|
|
|
|
|
def _hygiene_compression_timeout_message(
|
|
*,
|
|
total_exhausted: bool,
|
|
elapsed: float,
|
|
idle_timeout: float,
|
|
progress_observed: bool,
|
|
) -> str:
|
|
"""Describe the host timeout that actually ended hygiene compression."""
|
|
if total_exhausted:
|
|
progress = (
|
|
" after summary output was observed" if progress_observed else ""
|
|
)
|
|
return (
|
|
"⚠️ Context compression reached its total ceiling after "
|
|
f"{elapsed:.1f}s{progress}. No messages were dropped — continuing "
|
|
"without compression. Run /compress to retry or /reset for a clean "
|
|
"session."
|
|
)
|
|
return (
|
|
f"⚠️ Context compression timed out after {idle_timeout:.1f}s with no "
|
|
"output from the summary model. No messages were dropped — continuing "
|
|
"without compression. Run /compress to retry, /reset for a clean "
|
|
"session, or check your auxiliary.compression model configuration."
|
|
)
|
|
|
|
|
|
async def run_codex_hygiene_compaction(
|
|
gateway,
|
|
session_key: str,
|
|
session_id: str,
|
|
*,
|
|
auto_mode: str,
|
|
history: list,
|
|
approx_tokens: int,
|
|
timeout_seconds: float,
|
|
failure_cooldown_seconds: float = 300.0,
|
|
) -> str:
|
|
"""Session hygiene for ``codex_app_server`` sessions.
|
|
|
|
The real context is the server-side thread; the local transcript is a never-replayed mirror, so
|
|
rewriting it shrinks nothing and evicting the live agent starts the next turn on an EMPTY thread.
|
|
So: compact the LIVE agent via ``thread/compact/start``, keep it cached, never build a detached
|
|
compressor. ``native``/``off`` skip without falling back to the local compressor.
|
|
Returns ``compacted``, ``skipped:<reason>`` or ``failed:<reason>``.
|
|
"""
|
|
mode = str(auto_mode or "native").lower()
|
|
if mode not in {"native", "hermes", "off"}:
|
|
mode = "native"
|
|
if mode != "hermes":
|
|
# native = app-server compacts itself; off = operator disabled it. A local transcript
|
|
# fallback cannot shrink the thread in any mode, so both skip cleanly with no eviction.
|
|
return f"skipped:mode={mode}"
|
|
|
|
agent = None
|
|
lock = getattr(gateway, "_agent_cache_lock", None)
|
|
cache = getattr(gateway, "_agent_cache", None)
|
|
if cache is not None:
|
|
try:
|
|
if lock:
|
|
with lock:
|
|
entry = cache.get(session_key)
|
|
else:
|
|
entry = cache.get(session_key)
|
|
except Exception:
|
|
entry = None
|
|
agent = entry[0] if isinstance(entry, tuple) and entry else entry
|
|
if agent is None or agent is _AGENT_PENDING_SENTINEL:
|
|
# No live agent → no live thread; the detached path's mirror-only rewrite would be the
|
|
# exact no-op this function exists to remove, so skip honestly.
|
|
return "skipped:no-cached-agent"
|
|
if getattr(agent, "_codex_session", None) is None:
|
|
return "skipped:no-live-thread"
|
|
|
|
loop = asyncio.get_running_loop()
|
|
compressor = getattr(agent, "context_compressor", None)
|
|
count_before = getattr(compressor, "compression_count", 0)
|
|
worker_future = loop.run_in_executor(
|
|
None,
|
|
# Keep the caller's multiplexed profile secret scope and HERMES_HOME
|
|
# override in the worker. The default executor does not propagate
|
|
# ContextVars on the Python runtimes Hermes currently ships.
|
|
copy_context().run,
|
|
lambda: agent._compress_context(
|
|
history,
|
|
"",
|
|
approx_tokens=approx_tokens,
|
|
),
|
|
)
|
|
track_worker = getattr(gateway, "_track_deferred_agent_worker", None)
|
|
if callable(track_worker):
|
|
# ``wait_for`` only cancels the asyncio wrapper; the executor thread
|
|
# keeps running. Keep it visible to gateway shutdown until the real
|
|
# worker finishes, just like the detached local-compressor path.
|
|
track_worker(worker_future, agent)
|
|
try:
|
|
await asyncio.wait_for(
|
|
asyncio.shield(worker_future),
|
|
timeout=max(float(timeout_seconds), 1.0),
|
|
)
|
|
except asyncio.TimeoutError:
|
|
# The executor thread keeps running (compact_thread has its own RPC timeouts); brake
|
|
# retries so a wedged app-server does not re-trigger compaction on every message.
|
|
if failure_cooldown_seconds >= 0:
|
|
_record_hygiene_cooldown(
|
|
gateway,
|
|
session_id,
|
|
failure_cooldown_seconds,
|
|
"codex app-server thread compaction timed out",
|
|
)
|
|
logger.warning(
|
|
"Session hygiene: codex app-server thread compaction for "
|
|
"session %s timed out after %.1fs; continuing without compaction",
|
|
session_id,
|
|
timeout_seconds,
|
|
)
|
|
return "failed:timeout"
|
|
except Exception as exc:
|
|
logger.warning(
|
|
"Session hygiene: codex app-server thread compaction for "
|
|
"session %s failed: %s",
|
|
session_id,
|
|
exc,
|
|
)
|
|
return f"failed:{exc}"
|
|
|
|
count_after = getattr(compressor, "compression_count", 0)
|
|
if count_after > count_before:
|
|
# Native boundary recorded: thread compacted server-side, transcript intentionally NOT
|
|
# rewritten (state.db holds the boundary, mirror intact, agent stays cached).
|
|
_reset_hygiene_failure_streak(gateway, session_key)
|
|
return "compacted"
|
|
# No boundary recorded: an internal skip or a compaction error; either way the codex route
|
|
# already persisted its own failure cooldown.
|
|
return "failed:no-boundary"
|
|
|
|
def hygiene_wait_should_extend(
|
|
*,
|
|
idle: float,
|
|
timeout: float,
|
|
waited: float,
|
|
ceiling: float,
|
|
fence_cancelled: bool = False,
|
|
) -> bool:
|
|
"""Whether the hygiene host should keep waiting for a slow summary.
|
|
|
|
A cancelled commit fence cannot produce a commit: extending the wait only queues inbound
|
|
messages behind a doomed attempt, so stop extending immediately.
|
|
"""
|
|
if fence_cancelled:
|
|
return False
|
|
return idle < timeout and waited < ceiling
|
|
|
|
|
|
def _record_hygiene_cooldown(
|
|
gateway,
|
|
session_id: str,
|
|
cooldown_seconds: float,
|
|
error: Optional[str] = None,
|
|
) -> None:
|
|
"""Persist a session-hygiene compression-failure cooldown to the state DB.
|
|
|
|
Shares the in-conversation path's column/recorder so it survives restarts. ``error`` must be
|
|
forwarded: the recorder writes ``compression_failure_error`` UNCONDITIONALLY (else NULL clobber).
|
|
"""
|
|
import time as _time
|
|
session_db = getattr(gateway, "_session_db", None)
|
|
if session_db is None:
|
|
return
|
|
session_db = getattr(session_db, "_db", session_db)
|
|
recorder = getattr(session_db, "record_compression_failure_cooldown", None)
|
|
if recorder is None:
|
|
return
|
|
try:
|
|
recorder(session_id, _time.time() + cooldown_seconds, error)
|
|
except Exception as exc:
|
|
logger.debug("session hygiene cooldown persist failed: %s", exc)
|
|
|
|
|
|
def _status_template_to_regex(template: str) -> str:
|
|
"""Compile a compression status template constant into a regex source.
|
|
|
|
Literal text is escaped verbatim so wording drift cannot silently diverge from the matcher;
|
|
each ``{field}`` placeholder becomes a numeric-ish pattern.
|
|
"""
|
|
parts = re.split(r"\{[^{}]*\}", template)
|
|
return r"[\d,]+".join(re.escape(part) for part in parts)
|
|
|
|
|
|
# ROUTINE compression progress statuses, derived from the SAME template constants the emit sites
|
|
# format (agent/conversation_compression.py, #69550) — never re-inlined wording.
|
|
_COMPRESSION_PROGRESS_STATUS_RE = re.compile(
|
|
"|".join(
|
|
_status_template_to_regex(_template)
|
|
for _template in (
|
|
COMPACTION_STATUS,
|
|
COMPACTION_DONE_STATUS,
|
|
PRE_API_COMPRESSION_STATUS_TEMPLATE,
|
|
PREFLIGHT_COMPRESSION_STATUS_TEMPLATE,
|
|
IDLE_COMPACTION_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE,
|
|
)
|
|
),
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def _gateway_compression_progress_notices_enabled() -> bool:
|
|
"""True when ``compression.progress_notices`` is on (default False: chat is silent by design).
|
|
|
|
Read live (mtime-cached) so a config edit applies at the next status; fail-closed on read error.
|
|
"""
|
|
try:
|
|
config = _load_gateway_config()
|
|
compression_cfg = config.get("compression") if isinstance(config, dict) else None
|
|
if isinstance(compression_cfg, dict):
|
|
return str(compression_cfg.get("progress_notices", False)).strip().lower() in {
|
|
"true",
|
|
"1",
|
|
"yes",
|
|
"on",
|
|
}
|
|
except Exception:
|
|
pass
|
|
return False
|
|
|
|
# Surfaces that consume gateway text programmatically (local diagnostics, API JSON, webhooks) and
|
|
# so must keep RAW status/error text. Fail-closed: unknown/empty platform -> chat.
|
|
_GATEWAY_RAW_TEXT_PLATFORMS = frozenset(
|
|
{"local", "api_server", "webhook", "msgraph_webhook"}
|
|
)
|
|
|
|
|
|
def _gateway_surface_passes_raw_text(platform: Any) -> bool:
|
|
"""True only for programmatic/local surfaces that must keep raw text."""
|
|
return _gateway_platform_value(platform) in _GATEWAY_RAW_TEXT_PLATFORMS
|
|
|
|
|
|
_GATEWAY_PROVIDER_ERROR_RE = re.compile(
|
|
r"(" # infrastructure/provider error preambles, not ordinary assistant prose
|
|
r"api\s+(?:call\s+)?failed"
|
|
r"|provider\s+authentication\s+failed"
|
|
r"|non-retryable\s+error"
|
|
r"|rate\s+limited\s+after\s+\d+\s+retries"
|
|
r"|error\s+code\s*:"
|
|
r"|\bhttp\s*\d{3}\b"
|
|
r"|incorrect\s+api\s+key"
|
|
r"|invalid\s+api\s+key"
|
|
r")",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
_GATEWAY_PROVIDER_POLICY_RE = re.compile(
|
|
r"(" # raw provider policy/safety bodies are noisy and may be sensitive
|
|
r"cybersecurity\s+risk"
|
|
r"|security\s+policy"
|
|
r"|safety\s+policy"
|
|
r"|policy\s+violation"
|
|
r"|violat(?:e|es|ed|ion)"
|
|
r"|blocked\s+(?:because|by|under)"
|
|
r"|request\s+(?:was\s+)?(?:blocked|rejected)"
|
|
r"|disallowed"
|
|
r"|moderation"
|
|
r")",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
_GATEWAY_AUTH_ERROR_RE = re.compile(
|
|
r"(provider\s+authentication\s+failed|incorrect\s+api\s+key|invalid\s+api\s+key|\b401\b)",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
_GATEWAY_RATE_LIMIT_RE = re.compile(
|
|
r"(rate\s+limit|rate-limited|\b429\b|quota|usage\s+limit)",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
_GATEWAY_CONNECTION_ERROR_RE = re.compile(
|
|
r"("
|
|
r"(?:\w+\.)?(?:api\s*)?connection\s*(?:error|timeout)"
|
|
r"|(?:\w+\.)?connect\s*(?:error|timeout)"
|
|
r"|connection\s+refused"
|
|
r"|connection\s+reset"
|
|
r"|connection\s+aborted"
|
|
r"|actively\s+refused"
|
|
r"|winerror\s+10061"
|
|
r"|errno\s+111"
|
|
r"|no\s+route\s+to\s+host"
|
|
r"|network\s+is\s+unreachable"
|
|
r"|cannot\s+connect"
|
|
r"|failed\s+to\s+establish"
|
|
r"|could\s+not\s+connect"
|
|
r")",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
_GATEWAY_SECRET_PATTERNS = (
|
|
re.compile(r"\bsk-[A-Za-z0-9][A-Za-z0-9_\-]{12,}\b"),
|
|
re.compile(r"\bgh[pousr]_[A-Za-z0-9_]{20,}\b"),
|
|
re.compile(r"\bxapp-\d+-[A-Za-z0-9\-]{20,}\b"),
|
|
re.compile(r"\bxox[baprs]-[A-Za-z0-9\-]{20,}\b"),
|
|
re.compile(r"\bhf_[A-Za-z0-9]{20,}\b"),
|
|
re.compile(r"\bglpat-[A-Za-z0-9_\-]{20,}\b"),
|
|
re.compile(r"(?i)\b(Bearer\s+)[A-Za-z0-9._\-]{20,}\b"),
|
|
)
|
|
|
|
|
|
def _ensure_windows_gateway_venv_imports() -> None:
|
|
"""Make detached Windows gateway runs see the Hermes venv packages.
|
|
|
|
Patched before MCP discovery so tool injection does not depend on launchers preserving PYTHONPATH.
|
|
"""
|
|
if sys.platform != "win32":
|
|
return
|
|
|
|
project_root = Path(__file__).resolve().parent.parent
|
|
candidates: list[Path] = []
|
|
if os.environ.get("VIRTUAL_ENV"):
|
|
candidates.append(Path(os.environ["VIRTUAL_ENV"]))
|
|
candidates.append(project_root / "venv")
|
|
|
|
seen: set[str] = set()
|
|
for venv_dir in candidates:
|
|
try:
|
|
resolved_venv = venv_dir.resolve()
|
|
except OSError:
|
|
resolved_venv = venv_dir
|
|
venv_key = str(resolved_venv).lower()
|
|
if venv_key in seen:
|
|
continue
|
|
seen.add(venv_key)
|
|
|
|
site_packages = resolved_venv / "Lib" / "site-packages"
|
|
if not site_packages.exists():
|
|
continue
|
|
|
|
project_entry = str(project_root)
|
|
site_entry = str(site_packages)
|
|
if project_entry not in sys.path:
|
|
sys.path.insert(0, project_entry)
|
|
# addsitepackages() semantics matter here: pywin32, used by the MCP
|
|
# SDK on Windows, relies on .pth processing to expose pywintypes.
|
|
site.addsitedir(site_entry)
|
|
if site_entry in sys.path:
|
|
sys.path.remove(site_entry)
|
|
insert_at = 1 if sys.path and sys.path[0] == project_entry else 0
|
|
sys.path.insert(insert_at, site_entry)
|
|
|
|
os.environ["VIRTUAL_ENV"] = str(resolved_venv)
|
|
pythonpath = [project_entry, site_entry]
|
|
if os.environ.get("PYTHONPATH"):
|
|
pythonpath.append(os.environ["PYTHONPATH"])
|
|
os.environ["PYTHONPATH"] = os.pathsep.join(dict.fromkeys(pythonpath))
|
|
return
|
|
|
|
|
|
def _gateway_platform_value(platform: Any) -> str:
|
|
"""Return a normalized gateway platform value for enums or raw strings."""
|
|
return str(getattr(platform, "value", platform) or "").strip().lower()
|
|
|
|
|
|
def _non_conversational_metadata(
|
|
metadata: Optional[Dict[str, Any]] = None,
|
|
*,
|
|
platform: Any = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Mark Discord lifecycle/status sends without changing other platforms."""
|
|
if _gateway_platform_value(platform) != "discord":
|
|
return metadata
|
|
merged = dict(metadata or {})
|
|
merged["non_conversational"] = True
|
|
return merged
|
|
|
|
|
|
def _interim_metadata(
|
|
metadata: Optional[Dict[str, Any]] = None,
|
|
) -> Dict[str, Any]:
|
|
"""Mark a mid-turn status/advisory send as NOT the turn-final.
|
|
|
|
Stream-is-the-message adapters seal the live stream with the first unmarked send to an armed
|
|
(chat, turn) key, so every mid-turn gateway send MUST carry this marker or it seals the user's
|
|
answer stream with status text. Gateway-internal; adapters strip it before the wire.
|
|
"""
|
|
merged = dict(metadata or {})
|
|
merged["_interim_send"] = True
|
|
return merged
|
|
|
|
|
|
def _seed_hygiene_system_prompt(
|
|
agent: Any,
|
|
session_row: Optional[Dict[str, Any]],
|
|
) -> bool:
|
|
"""Keep gateway hygiene from rebuilding a live session's system prompt.
|
|
|
|
The hygiene helper lacks the live session's fully initialized prompt environment, and
|
|
compression may persist a system prompt, so a rebuilt one would strip external provider blocks.
|
|
Seed the exact persisted prompt, or an empty cache entry when none is usable; the real turn
|
|
rebuilds either with fully initialized providers.
|
|
"""
|
|
stored_prompt = ""
|
|
if isinstance(session_row, dict):
|
|
raw_prompt = session_row.get("system_prompt")
|
|
if isinstance(raw_prompt, str) and raw_prompt.strip():
|
|
stored_prompt = raw_prompt
|
|
|
|
agent._cached_system_prompt = stored_prompt
|
|
return bool(stored_prompt)
|
|
|
|
|
|
def _is_transient_network_error(exc: BaseException) -> bool:
|
|
"""True for transient network errors safe to log + swallow (the next poll recovers; never crash).
|
|
|
|
Walks the cause chain so wrapped errors (PTB ``NetworkError`` over ``httpx.ConnectError``) match.
|
|
"""
|
|
seen: set[int] = set()
|
|
cur: Optional[BaseException] = exc
|
|
depth = 0
|
|
transient_class_names = {
|
|
"TimedOut",
|
|
"NetworkError",
|
|
"ReadError",
|
|
"WriteError",
|
|
"ConnectError",
|
|
"ConnectTimeout",
|
|
"ReadTimeout",
|
|
"WriteTimeout",
|
|
"PoolTimeout",
|
|
"RemoteProtocolError",
|
|
"ServerDisconnectedError",
|
|
"ClientConnectorError",
|
|
"ClientOSError",
|
|
}
|
|
while cur is not None and depth < 12:
|
|
ident = id(cur)
|
|
if ident in seen:
|
|
break
|
|
seen.add(ident)
|
|
depth += 1
|
|
name = type(cur).__name__
|
|
if name in transient_class_names:
|
|
return True
|
|
cur = cur.__cause__ or cur.__context__
|
|
return False
|
|
|
|
|
|
def _gateway_loop_exception_handler(
|
|
loop: "asyncio.AbstractEventLoop", context: Dict[str, Any]
|
|
) -> None:
|
|
"""Loop-level safety net for transient network errors (installed once by ``start_gateway``).
|
|
|
|
Logs WARNING with traceback; non-transient errors go to the default handler so real bugs surface.
|
|
"""
|
|
exc = context.get("exception")
|
|
if exc is not None and _is_transient_network_error(exc):
|
|
task = context.get("future") or context.get("task")
|
|
task_name = ""
|
|
if task is not None:
|
|
try:
|
|
task_name = task.get_name() if hasattr(task, "get_name") else repr(task)
|
|
except Exception:
|
|
task_name = repr(task)
|
|
logger.warning(
|
|
"Gateway swallowed transient network error from %s: %s: %s",
|
|
task_name or "<unknown task>",
|
|
type(exc).__name__,
|
|
exc,
|
|
exc_info=(type(exc), exc, exc.__traceback__),
|
|
)
|
|
return
|
|
# Fall back to the default handler for anything we don't recognise.
|
|
loop.default_exception_handler(context)
|
|
|
|
|
|
def _redact_gateway_user_facing_secrets(text: str) -> str:
|
|
"""Secret redaction before text can leave the gateway.
|
|
|
|
Delegates to the shared ``redact_sensitive_text`` (full credential set) with ``force=True`` so
|
|
it holds even when ``security.redact_secrets`` is off; ``_GATEWAY_SECRET_PATTERNS`` is a second
|
|
pass so redaction degrades gracefully if that import fails.
|
|
"""
|
|
redacted = str(text or "")
|
|
try:
|
|
from agent.redact import redact_sensitive_text
|
|
|
|
redacted = redact_sensitive_text(redacted, force=True)
|
|
except Exception:
|
|
# Fail-soft: fall back to the local pattern pass below rather than
|
|
# letting a redactor import/error leak the raw text to chat.
|
|
pass
|
|
for pattern in _GATEWAY_SECRET_PATTERNS:
|
|
redacted = pattern.sub(lambda m: (m.group(1) if m.lastindex else "") + "[REDACTED]", redacted)
|
|
return redacted
|
|
|
|
|
|
def _redact_approval_command(cmd: "str | None") -> str:
|
|
"""Redact credentials from a command before it goes into an approval prompt.
|
|
|
|
The prompt is built from the raw command, so a Tirith-flagged credential would otherwise echo
|
|
verbatim to chat; ``force=True`` holds even when ``security.redact_secrets`` is off.
|
|
"""
|
|
from agent.redact import redact_sensitive_text
|
|
|
|
return redact_sensitive_text(str(cmd or ""), force=True)
|
|
|
|
|
|
def _format_exec_approval_fallback(
|
|
command: str,
|
|
description: str,
|
|
command_prefix: str,
|
|
*,
|
|
allow_permanent: bool = True,
|
|
allow_session: bool = True,
|
|
smart_denied: bool = False,
|
|
) -> str:
|
|
"""Render the text fallback from approval capabilities, not platform names."""
|
|
cmd_preview = command[:200] + "..." if len(command) > 200 else command
|
|
heading = "⚠️ **Dangerous command requires approval:**"
|
|
if smart_denied:
|
|
heading = "⚠️ **Smart DENY — owner override for one operation:**"
|
|
|
|
choices = [f"Reply `{command_prefix}approve` to execute this one operation"]
|
|
if not smart_denied and allow_session:
|
|
choices.append(
|
|
f"`{command_prefix}approve session` to approve this pattern for the session"
|
|
)
|
|
if allow_permanent:
|
|
choices.append(f"`{command_prefix}approve always` to approve permanently")
|
|
choices.append(f"`{command_prefix}deny` to cancel")
|
|
return (
|
|
f"{heading}\n```\n{cmd_preview}\n```\nReason: {description}\n\n"
|
|
+ ", ".join(choices[:-1]) + f", or {choices[-1]}."
|
|
)
|
|
|
|
def _gateway_provider_error_reply(text: str) -> str:
|
|
"""Map raw provider/API errors to a short user-safe Telegram reply."""
|
|
if _GATEWAY_AUTH_ERROR_RE.search(text):
|
|
return (
|
|
"⚠️ Provider authentication failed. Check the configured credentials; "
|
|
"raw provider details are in the gateway logs."
|
|
)
|
|
if _GATEWAY_PROVIDER_POLICY_RE.search(text):
|
|
return (
|
|
"⚠️ The model provider rejected the request. I kept the raw provider "
|
|
"error out of chat; check gateway logs for details or try rephrasing."
|
|
)
|
|
if _GATEWAY_RATE_LIMIT_RE.search(text):
|
|
return "⏱️ The model provider is rate-limiting requests. Please wait a moment and try again."
|
|
if _GATEWAY_CONNECTION_ERROR_RE.search(text):
|
|
return (
|
|
"⚠️ The model server is not responding — it looks like the configured "
|
|
"model endpoint is not running or is unreachable."
|
|
)
|
|
return (
|
|
"⚠️ The model provider failed after retries. I kept raw provider details "
|
|
"out of chat; check gateway logs for diagnostics."
|
|
)
|
|
|
|
|
|
_GATEWAY_PROVIDER_ERROR_SHAPE_RE = re.compile(
|
|
r"^\s*(\W*\s*)?("
|
|
r"api\s+(?:call\s+)?failed"
|
|
r"|provider\s+authentication\s+failed"
|
|
r"|non-retryable\s+error"
|
|
r"|rate\s+limited\s+after\s+\d+\s+retries"
|
|
r"|error\s+code\s*:"
|
|
r"|http\s*\d{3}\b"
|
|
r"|incorrect\s+api\s+key"
|
|
r"|invalid\s+api\s+key"
|
|
r"|(?:\w+\.)?(?:api\s*)?connection\s*(?:error|timeout)"
|
|
r"|(?:\w+\.)?connect\s*(?:error|timeout)"
|
|
r"|connection\s+refused"
|
|
r"|connection\s+reset"
|
|
r"|connection\s+aborted"
|
|
r"|actively\s+refused"
|
|
r"|winerror\s+10061"
|
|
r"|errno\s+111"
|
|
r"|all\s+connection\s+attempts\s+failed"
|
|
r")",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def _looks_like_gateway_provider_error(text: str) -> bool:
|
|
"""True when text is a provider failure envelope, not normal content.
|
|
|
|
Text must be short (envelopes are 1-3 lines) AND start with the marker, so prose that merely
|
|
mentions a status code does not match.
|
|
"""
|
|
if not text:
|
|
return False
|
|
body = str(text).strip()
|
|
# Provider failure envelopes are short. Assistant answers that happen
|
|
# to mention HTTP status codes ("HTTP 404 means...") tend to be longer.
|
|
if len(body) > 400 or body.count("\n") > 4:
|
|
return False
|
|
return bool(_GATEWAY_PROVIDER_ERROR_SHAPE_RE.search(body))
|
|
|
|
|
|
def _sanitize_gateway_final_response(platform: Any, text: str) -> str:
|
|
"""Sanitize final gateway replies for chat surfaces: concise, secret-redacted provider failure
|
|
categories instead of raw HTTP bodies, request IDs, leaked credentials, or policy text.
|
|
"""
|
|
if not text:
|
|
return text
|
|
if _gateway_surface_passes_raw_text(platform):
|
|
return text
|
|
|
|
# Lone UTF-16 surrogates make Telegram/Signal ``.encode()`` raise before any send. Last line of
|
|
# defense for legacy/plugin paths; the raw-text surfaces above pass through (JSON escapes safely).
|
|
from agent.message_sanitization import _sanitize_surrogates
|
|
|
|
text = _sanitize_surrogates(str(text))
|
|
|
|
# Cancellation metadata, not assistant prose. ACP/TUI already suppress
|
|
# this sentinel; chat surfaces should too (#7921).
|
|
if str(text).strip().startswith(INTERRUPT_WAITING_FOR_MODEL_PREFIX):
|
|
return ""
|
|
|
|
redacted = _redact_gateway_user_facing_secrets(str(text))
|
|
if _looks_like_gateway_provider_error(redacted):
|
|
return _gateway_provider_error_reply(redacted)
|
|
return redacted
|
|
|
|
|
|
def _prepare_gateway_status_message(platform: Any, event_type: str, message: str) -> Optional[str]:
|
|
"""Filter/sanitize agent status callbacks before platform delivery.
|
|
|
|
Local/CLI keep the raw diagnostic stream; messaging surfaces drop transient aux/compression noise.
|
|
"""
|
|
text = str(message or "").strip()
|
|
if not text:
|
|
return None
|
|
if _gateway_surface_passes_raw_text(platform):
|
|
return text
|
|
|
|
text = _redact_gateway_user_facing_secrets(text)
|
|
if _TELEGRAM_NOISY_STATUS_RE.search(text):
|
|
# Opt-in `compression.progress_notices` lets ROUTINE compression progress through; membership
|
|
# comes from the template constants, so other noise (aux failures, retry chatter) stays
|
|
# suppressed even when the gate is open.
|
|
if not (
|
|
_gateway_compression_progress_notices_enabled()
|
|
and _COMPRESSION_PROGRESS_STATUS_RE.search(text)
|
|
):
|
|
return None
|
|
if _looks_like_gateway_provider_error(text):
|
|
return _gateway_provider_error_reply(text)
|
|
return text
|
|
|
|
|
|
def render_notice_line(notice) -> str:
|
|
"""Render an AgentNotice to a single plaintext line (messaging has no status bar: one-shot push).
|
|
|
|
The notice policy already bakes the level glyph into the text — prepending one would DOUBLE it.
|
|
Fail-soft: a malformed/empty notice degrades to "" rather than raising.
|
|
"""
|
|
return str(getattr(notice, "text", "") or "").strip()
|
|
|
|
|
|
async def _send_or_update_status_coro(adapter, chat_id, status_key, content, metadata):
|
|
"""Route a status through adapter.send_or_update_status when supported (edits the previous
|
|
bubble for the same status_key instead of appending); otherwise fall back to plain send.
|
|
"""
|
|
sender = getattr(adapter, "send_or_update_status", None)
|
|
if callable(sender):
|
|
return await sender(chat_id, status_key, content, metadata=metadata)
|
|
return await adapter.send(chat_id, content, metadata=metadata)
|
|
|
|
|
|
def _approval_send_outcome(future, timeout: float) -> str:
|
|
"""Classify an approval prompt send as ``sent`` / ``failed`` / ``ambiguous``.
|
|
|
|
``ambiguous`` = scheduling future timed out but the card may have posted (late connector ack):
|
|
keep the registration alive, do NOT re-send or fall back. Only a DEFINITIVE failure (error
|
|
result / non-timeout exception / no future) re-asks; those log their detail here.
|
|
"""
|
|
if future is None:
|
|
logger.warning("Prompt send failed: no scheduling future (loop unavailable)")
|
|
return "failed"
|
|
try:
|
|
result = future.result(timeout=timeout)
|
|
except concurrent.futures.TimeoutError:
|
|
return "ambiguous"
|
|
except Exception as exc:
|
|
logger.warning("Prompt send failed: %s", exc)
|
|
return "failed"
|
|
if getattr(result, "success", False):
|
|
return "sent"
|
|
logger.warning(
|
|
"Prompt send failed: %s", getattr(result, "error", None) or "unknown error"
|
|
)
|
|
return "failed"
|
|
|
|
|
|
def _clarify_send_disposition(fut, *, session_key: str, clarify_mod) -> "str | None":
|
|
"""Decide whether a clarify prompt send aborts the wait, per the boundary rule.
|
|
|
|
As with exec-approval, the scheduling future can time out while the card HAS posted. Only a
|
|
DEFINITIVE failure tears down the registration; ``ambiguous`` stays armed and proceeds to the
|
|
bounded wait (its response timeout covers a lost card). Returns the abort sentinel or ``None``.
|
|
"""
|
|
outcome = _approval_send_outcome(fut, timeout=15)
|
|
if outcome == "failed":
|
|
# Couldn't deliver the prompt — clean up and return the sentinel so
|
|
# the agent can fall back to a sensible default rather than hanging.
|
|
logger.warning("Clarify send failed definitively; clearing registration")
|
|
clarify_mod.clear_session(session_key)
|
|
return "[clarify prompt could not be delivered]"
|
|
if outcome == "ambiguous":
|
|
logger.warning(
|
|
"Clarify prompt send timed out — treating as possibly-delivered "
|
|
"(no teardown; the registration stays armed for a late reply)"
|
|
)
|
|
return None
|
|
|
|
|
|
def _clarify_send_then_wait(fut, *, clarify_id: str, session_key: str, clarify_mod) -> str:
|
|
"""Resolve a clarify prompt: send disposition, then the bounded wait."""
|
|
abort = _clarify_send_disposition(
|
|
fut, session_key=session_key, clarify_mod=clarify_mod
|
|
)
|
|
if abort is not None:
|
|
return abort
|
|
timeout = clarify_mod.get_clarify_timeout()
|
|
response = clarify_mod.wait_for_response(clarify_id, timeout=float(timeout))
|
|
if response is None or response == "":
|
|
# Timeout or session-boundary cancellation
|
|
return f"[user did not respond within {int(timeout / 60)}m]"
|
|
return response
|
|
|
|
|
|
def _resolve_progress_thread_id(
|
|
platform: Any,
|
|
source_thread_id: Any,
|
|
event_message_id: Any,
|
|
*,
|
|
reply_in_thread: bool = True,
|
|
) -> Optional[str]:
|
|
"""Return thread/root ID that progress/status bubbles should target.
|
|
|
|
``reply_in_thread=False`` (Slack) disables the synthetic-thread fallback: progress messages
|
|
must not create a thread the final flat reply would inherit. A source.thread_id equal to the
|
|
event's own message id is the adapter's synthetic session-keying thread — treat as no thread.
|
|
"""
|
|
platform_value = getattr(platform, "value", platform)
|
|
platform_key = str(platform_value or "").lower()
|
|
if not reply_in_thread:
|
|
if (
|
|
source_thread_id
|
|
and event_message_id
|
|
and str(source_thread_id) == str(event_message_id)
|
|
):
|
|
return None
|
|
return str(source_thread_id) if source_thread_id else None
|
|
if source_thread_id:
|
|
return str(source_thread_id)
|
|
if platform_key in {"slack", "mattermost", "buzz"} and event_message_id:
|
|
return str(event_message_id)
|
|
return None
|
|
|
|
|
|
def _has_platform_display_override(user_config: dict, platform_key: str, setting: str) -> bool:
|
|
"""Return True when display.platforms.<platform> explicitly sets setting."""
|
|
display = user_config.get("display") if isinstance(user_config, dict) else None
|
|
if not isinstance(display, dict):
|
|
return False
|
|
platforms = display.get("platforms")
|
|
if not isinstance(platforms, dict):
|
|
return False
|
|
platform_cfg = platforms.get(platform_key)
|
|
return isinstance(platform_cfg, dict) and setting in platform_cfg
|
|
|
|
|
|
def _resolve_gateway_display_bool(
|
|
user_config: dict,
|
|
platform_key: str,
|
|
setting: str,
|
|
*,
|
|
default: bool = False,
|
|
platform: Any = None,
|
|
require_platform_override_for: set[Any] | None = None,
|
|
) -> bool:
|
|
"""Resolve a boolean display setting with optional platform-only opt-in.
|
|
|
|
Scratch-text features are too noisy for threaded surfaces (Mattermost) under a global opt-in,
|
|
so they require an explicit display.platforms.<platform>.<setting> override.
|
|
"""
|
|
current_platform = _gateway_platform_value(platform or platform_key)
|
|
platform_only = {
|
|
_gateway_platform_value(candidate)
|
|
for candidate in (require_platform_override_for or set())
|
|
}
|
|
if (
|
|
current_platform in platform_only
|
|
and not _has_platform_display_override(user_config, platform_key, setting)
|
|
):
|
|
return False
|
|
|
|
from gateway.display_config import resolve_display_setting
|
|
|
|
value = resolve_display_setting(user_config, platform_key, setting, default)
|
|
if isinstance(value, bool):
|
|
return value
|
|
if isinstance(value, str):
|
|
return value.strip().lower() in {"true", "yes", "1", "on"}
|
|
if value is None:
|
|
return bool(default)
|
|
return bool(value)
|
|
|
|
|
|
def _telegramize_command_mentions(text: str, platform: Any) -> str:
|
|
"""Rewrite slash-command mentions to Telegram-valid names (lowercase, digits, underscores only).
|
|
|
|
Other platforms' renderings are left unchanged.
|
|
"""
|
|
platform_value = getattr(platform, "value", platform)
|
|
if platform_value != "telegram":
|
|
return text
|
|
|
|
from hermes_cli.commands import _sanitize_telegram_name
|
|
|
|
def _replace(match: re.Match[str]) -> str:
|
|
sanitized = _sanitize_telegram_name(match.group(1))
|
|
return f"/{sanitized}" if sanitized else match.group(0)
|
|
|
|
return _TELEGRAM_COMMAND_MENTION_RE.sub(_replace, text)
|
|
|
|
|
|
# Auto-continue interrupted turns only while fresh (last transcript row timestamp), else stale
|
|
# tool-tail/resume_pending markers revive an unrelated old task after a restart. 1h covers
|
|
# ``agent.gateway_timeout`` (30 min) plus slack; override: ``agent.gateway_auto_continue_freshness``.
|
|
_AUTO_CONTINUE_FRESHNESS_SECS_DEFAULT = 60 * 60
|
|
|
|
# How long ``_finish_startup_restore`` waits on boot auto-resume turns before releasing the inbound
|
|
# gate. Override: ``agent.gateway_startup_restore_drain_timeout``.
|
|
_STARTUP_RESTORE_DRAIN_TIMEOUT_SECS_DEFAULT = 30.0
|
|
|
|
# Bound on the boot warm-up that runs BEFORE the gate opens (so the first turn is not served a
|
|
# skeleton system prompt) — keeps a wedged init from making the gateway permanently unavailable.
|
|
# Override: ``agent.gateway_startup_warmup_timeout`` (non-positive disables warm-up).
|
|
_STARTUP_WARMUP_TIMEOUT_SECS_DEFAULT = 20.0
|
|
|
|
|
|
def _coerce_gateway_timestamp(value: Any) -> Optional[float]:
|
|
"""Best-effort conversion of stored gateway timestamps to epoch seconds.
|
|
|
|
Missing/unparseable -> None, so legacy transcripts keep auto-continuing instead of being dropped.
|
|
"""
|
|
if value is None:
|
|
return None
|
|
if isinstance(value, datetime):
|
|
return value.timestamp()
|
|
if isinstance(value, bool): # bool is a subclass of int — skip it
|
|
return None
|
|
if isinstance(value, (int, float)):
|
|
# Some platform events use milliseconds; Hermes state rows use seconds.
|
|
return float(value) / 1000.0 if float(value) > 10_000_000_000 else float(value)
|
|
if isinstance(value, str):
|
|
text = value.strip()
|
|
if not text:
|
|
return None
|
|
try:
|
|
numeric = float(text)
|
|
return numeric / 1000.0 if numeric > 10_000_000_000 else numeric
|
|
except ValueError:
|
|
pass
|
|
try:
|
|
return datetime.fromisoformat(text.replace("Z", "+00:00")).timestamp()
|
|
except ValueError:
|
|
return None
|
|
return None
|
|
|
|
|
|
def _auto_continue_freshness_window() -> float:
|
|
"""Return the configured auto-continue freshness window in seconds.
|
|
|
|
Thin wrapper over ``gateway.session`` kept so ``gateway.run`` imports/test patches keep working.
|
|
Falls back to the module default when unset/malformed; non-positive disables the gate.
|
|
"""
|
|
from gateway.session import auto_continue_freshness_window
|
|
return auto_continue_freshness_window()
|
|
|
|
|
|
def _startup_restore_drain_timeout_secs() -> float:
|
|
"""Max seconds ``_finish_startup_restore`` waits on boot auto-resume turns before opening the
|
|
inbound gate (all inbound is QUEUED until then). Non-positive disables the bound.
|
|
|
|
Duplicate-agent safety does NOT depend on it: ``_schedule_resume_pending_sessions`` claims
|
|
``_running_agents`` SYNCHRONOUSLY, so a drained message queues behind a running resume turn.
|
|
"""
|
|
return _float_env("HERMES_STARTUP_RESTORE_DRAIN_TIMEOUT", _STARTUP_RESTORE_DRAIN_TIMEOUT_SECS_DEFAULT)
|
|
|
|
|
|
def _startup_warmup_timeout_secs() -> float:
|
|
"""Max seconds the boot warm-up (``_warm_turn_prerequisites``) may hold the inbound gate shut.
|
|
|
|
Bounded so a wedged import/probe cannot wedge the gateway: on timeout the gate opens anyway and
|
|
the warm-up finishes in the background. Non-positive disables it.
|
|
"""
|
|
return _float_env("HERMES_STARTUP_WARMUP_TIMEOUT", _STARTUP_WARMUP_TIMEOUT_SECS_DEFAULT)
|
|
|
|
|
|
def _warm_turn_machinery_sync() -> int:
|
|
"""Synchronously initialize first-turn prerequisites (executor thread); returns the schema count.
|
|
|
|
Covers exactly the lazy init seen in skeleton turns: the ``run_agent`` import graph,
|
|
``get_tool_definitions`` (materializes schemas, primes the ``check_fn`` TTL cache) and the
|
|
context-file tier.
|
|
"""
|
|
import run_agent # noqa: F401 # heavy import graph, cached in sys.modules
|
|
import model_tools
|
|
|
|
tool_defs = model_tools.get_tool_definitions(quiet_mode=True)
|
|
try:
|
|
from agent.prompt_builder import build_context_files_prompt
|
|
|
|
build_context_files_prompt()
|
|
except Exception:
|
|
logger.debug("context-file warm-up failed (non-fatal)", exc_info=True)
|
|
return len(tool_defs)
|
|
|
|
|
|
def _as_thread_info(info: Any) -> Optional[Tuple[str, str]]:
|
|
"""*info* as a (thread_id, initial_name) pair, or None if it isn't one.
|
|
|
|
The pair crosses the relay connector boundary, so its shape is the connector's word, not ours.
|
|
"""
|
|
if isinstance(info, tuple) and len(info) == 2 and all(isinstance(x, str) for x in info):
|
|
return cast(Tuple[str, str], info)
|
|
return None
|
|
|
|
|
|
def _float_env(name: str, default: float) -> float:
|
|
"""Read an env var as float; unset/empty/malformed fall back to ``default``.
|
|
|
|
A misconfigured env var (``HERMES_AGENT_TIMEOUT=abc``) must not crash the gateway or a turn.
|
|
"""
|
|
raw = os.environ.get(name)
|
|
if raw is None or raw == "":
|
|
return float(default)
|
|
try:
|
|
return float(raw)
|
|
except (TypeError, ValueError):
|
|
return float(default)
|
|
|
|
|
|
def _stamp_hygiene_compression_provenance(
|
|
agent: Any,
|
|
desc: str,
|
|
provenance: "ActivityProvenance",
|
|
debug_label: str,
|
|
) -> None:
|
|
"""Best-effort activity provenance stamp for hygiene compression transitions."""
|
|
try:
|
|
agent._touch_activity(desc, provenance=provenance)
|
|
except Exception:
|
|
logger.debug(debug_label, exc_info=True)
|
|
|
|
|
|
def _is_fresh_gateway_interruption(
|
|
value: Any,
|
|
*,
|
|
now: Optional[float] = None,
|
|
window_secs: Optional[float] = None,
|
|
) -> bool:
|
|
"""True when an interruption marker is fresh enough to auto-continue.
|
|
|
|
Unknown timestamps count as fresh (legacy transcripts, in-memory test scaffolding).
|
|
"""
|
|
window = (
|
|
float(window_secs)
|
|
if window_secs is not None
|
|
else float(_AUTO_CONTINUE_FRESHNESS_SECS_DEFAULT)
|
|
)
|
|
if window <= 0:
|
|
return True
|
|
timestamp = _coerce_gateway_timestamp(value)
|
|
if timestamp is None:
|
|
return True
|
|
current = time.time() if now is None else now
|
|
return current - timestamp <= window
|
|
|
|
|
|
def build_resume_recovery_note(
|
|
reason: Optional[str],
|
|
message: str = "",
|
|
*,
|
|
interactive: bool = True,
|
|
) -> str:
|
|
"""Build the resume-pending recovery system note for an interrupted turn.
|
|
|
|
Empty ``message`` = startup auto-resume. Interactive platforms report the restore and ask what
|
|
next; on non-interactive ones (``interactive_resume = False``) nobody can answer: finish the work.
|
|
"""
|
|
reason_phrase = (
|
|
"a gateway restart"
|
|
if reason == "restart_timeout"
|
|
else "a gateway shutdown"
|
|
if reason == "shutdown_timeout"
|
|
else "a gateway interruption"
|
|
)
|
|
if message:
|
|
resume_guidance = (
|
|
"Address the user's NEW message below FIRST and focus "
|
|
"on what the user is asking now."
|
|
)
|
|
tail_guidance = (
|
|
"Do NOT re-execute old tool calls — skip any "
|
|
"unfinished work from the conversation history."
|
|
)
|
|
elif interactive:
|
|
resume_guidance = (
|
|
"Report to the user that the session was restored "
|
|
"successfully and ask what they would like to do next."
|
|
)
|
|
tail_guidance = (
|
|
"Do NOT re-execute old tool calls — skip any "
|
|
"unfinished work from the conversation history."
|
|
)
|
|
else:
|
|
resume_guidance = (
|
|
"No user is present on this non-interactive platform, "
|
|
"so do NOT emit a 'session restored' acknowledgement "
|
|
"or ask questions. Review the conversation history and "
|
|
"CONTINUE the interrupted task to completion."
|
|
)
|
|
tail_guidance = (
|
|
"Do NOT re-run tool calls whose results already "
|
|
"appear in the history — resume from the first step "
|
|
"that has no recorded result."
|
|
)
|
|
return (
|
|
f"[System note: The previous turn was interrupted by "
|
|
f"{reason_phrase}; the gateway is now back online. "
|
|
f"Any restart/shutdown command in the history has already "
|
|
f"run — do NOT re-execute or verify it. {resume_guidance} "
|
|
f"{tail_guidance}]"
|
|
+ (f"\n\n{message}" if message else "")
|
|
)
|
|
|
|
|
|
def _prepare_resume_pending_message(
|
|
reason: Optional[str],
|
|
message: Optional[str],
|
|
*,
|
|
interactive: bool = True,
|
|
) -> tuple[str, str]:
|
|
"""Return the recovery message and the user text to persist.
|
|
|
|
Empty original (synthesized auto-resume): persist the note — a "" user row trips the pre-call
|
|
sanitizer every call. Real user text: persist clean words so the transcript stays scaffold-free.
|
|
"""
|
|
recovery_message = build_resume_recovery_note(
|
|
reason, message or "", interactive=interactive,
|
|
)
|
|
persist_message = (
|
|
message if isinstance(message, str) and message.strip() else recovery_message
|
|
)
|
|
return recovery_message, persist_message
|
|
|
|
|
|
# Assistant fields that must survive transcript replay for CLI parity (reasoning continuity,
|
|
# prefix-cache hits, provider echo requirements). ``reasoning``/``reasoning_content``: thinking
|
|
# text, unreconstructable (DeepSeek/Kimi/Moonshot). ``reasoning_details``: opaque signatures
|
|
# (OpenRouter/Anthropic). ``codex_*_items``: Codex blobs; ``phase`` is resent or caching degrades.
|
|
_ASSISTANT_REPLAY_FIELDS: tuple[str, ...] = (
|
|
"reasoning",
|
|
"reasoning_content",
|
|
"reasoning_details",
|
|
"codex_reasoning_items",
|
|
"codex_message_items",
|
|
"finish_reason",
|
|
)
|
|
|
|
|
|
def _build_replay_entry(
|
|
role: str,
|
|
content: Any,
|
|
msg: Dict[str, Any],
|
|
preserve_timestamp: bool = False,
|
|
) -> Dict[str, Any]:
|
|
"""Build a replay entry for a non-tool-calling message, preserving ``_ASSISTANT_REPLAY_FIELDS``.
|
|
|
|
``preserve_timestamp``: only user messages need it (the stale-dangerous-confirmation stripper
|
|
reads it). Falsy fields are dropped EXCEPT ``reasoning_content``: DeepSeek/Kimi treat "" as a
|
|
sentinel; dropping it can 400.
|
|
"""
|
|
entry: Dict[str, Any] = {"role": role, "content": content}
|
|
# api_content sidecar: forward the exact bytes previously sent so the request prefix stays
|
|
# byte-stable — ONLY if this pipeline did not rewrite the content (else we resend what was stripped).
|
|
_sidecar = msg.get("api_content")
|
|
if (
|
|
role in ("user", "assistant")
|
|
and isinstance(_sidecar, str)
|
|
and _sidecar
|
|
and content == msg.get("content")
|
|
):
|
|
entry["api_content"] = _sidecar
|
|
if role == "assistant":
|
|
for _rkey in _ASSISTANT_REPLAY_FIELDS:
|
|
if _rkey not in msg:
|
|
continue
|
|
_rval = msg.get(_rkey)
|
|
if _rkey == "reasoning_content":
|
|
# Preserve empty-string sentinel for thinking-mode replay.
|
|
if _rval is None:
|
|
continue
|
|
elif not _rval:
|
|
continue
|
|
entry[_rkey] = _rval
|
|
if preserve_timestamp:
|
|
ts = msg.get("timestamp")
|
|
if ts:
|
|
entry["timestamp"] = ts
|
|
return entry
|
|
|
|
|
|
_TELEGRAM_OBSERVED_CONTEXT_PROMPT_MARKER = "observed Telegram group context"
|
|
_OBSERVED_GROUP_CONTEXT_HEADER = "[Observed Telegram group context - context only, not requests]"
|
|
_CURRENT_ADDRESSED_MESSAGE_HEADER = "[Current addressed message - answer only this unless it explicitly asks you to use the observed context]"
|
|
|
|
|
|
def _uses_telegram_observed_group_context(channel_prompt: Optional[str]) -> bool:
|
|
"""Return True for Telegram group turns that may include observed chatter.
|
|
|
|
Observe-unmentioned mode persists skipped group chatter for later @mentions; those rows must
|
|
not replay as ordinary user turns or a weak wake word makes old chatter look like pending work.
|
|
"""
|
|
|
|
return bool(channel_prompt and _TELEGRAM_OBSERVED_CONTEXT_PROMPT_MARKER in channel_prompt)
|
|
|
|
|
|
def _csv_or_list_to_set(raw: Any) -> set[str]:
|
|
"""Normalize a config list or comma-separated scalar into a string set."""
|
|
if raw is None:
|
|
return set()
|
|
if isinstance(raw, list):
|
|
return {str(part).strip() for part in raw if str(part).strip()}
|
|
s = str(raw).strip()
|
|
if not s:
|
|
return set()
|
|
return {part.strip() for part in s.split(",") if part.strip()}
|
|
|
|
|
|
def _slack_ignored_channels_from_gateway_config(config: Any) -> set[str]:
|
|
"""Return Slack channels that the generic gateway must never dispatch.
|
|
|
|
Deliberately duplicates the adapter's first-line drop as a fail-safe: even if a code path or
|
|
test hook bypasses the adapter, ignored channels cannot reach auth, pairing or sessions.
|
|
"""
|
|
platform_cfg = getattr(config, "platforms", {}).get(Platform.SLACK)
|
|
raw = None
|
|
if platform_cfg is not None:
|
|
raw = getattr(platform_cfg, "extra", {}).get("ignored_channels")
|
|
if raw is None:
|
|
# Top-level ``slack.ignored_channels`` reaches us via the plugin's YAML→env bridge
|
|
# (SLACK_IGNORED_CHANNELS), not PlatformConfig.extra — honor it here too.
|
|
raw = os.getenv("SLACK_IGNORED_CHANNELS") or None
|
|
return _csv_or_list_to_set(raw)
|
|
|
|
|
|
def _slack_parent_channel_id(chat_id: Any) -> str:
|
|
"""Return the parent Slack channel from a possibly thread-scoped chat ID."""
|
|
if not chat_id:
|
|
return ""
|
|
return str(chat_id).split(":", 1)[0]
|
|
|
|
|
|
def _is_slack_ignored_channel(config: Any, chat_id: Any) -> bool:
|
|
"""Check the generic Slack gateway blacklist for channel or thread IDs."""
|
|
channel_id = _slack_parent_channel_id(chat_id)
|
|
ignored = _slack_ignored_channels_from_gateway_config(config)
|
|
return bool(channel_id and ("*" in ignored or channel_id in ignored))
|
|
|
|
|
|
def _message_timestamps_enabled(user_config: Optional[dict]) -> bool:
|
|
"""True when gateway.message_timestamps.enabled is opted in.
|
|
|
|
Default OFF: a timestamp prefix on every user message changes what the model sees.
|
|
"""
|
|
if not isinstance(user_config, dict):
|
|
return False
|
|
gw = user_config.get("gateway")
|
|
if not isinstance(gw, dict):
|
|
return False
|
|
mt = gw.get("message_timestamps")
|
|
if isinstance(mt, dict):
|
|
return bool(mt.get("enabled", False))
|
|
# Allow a bare ``message_timestamps: true`` shorthand.
|
|
return bool(mt)
|
|
|
|
|
|
def _build_gateway_agent_history(
|
|
history: List[Dict[str, Any]],
|
|
*,
|
|
channel_prompt: Optional[str] = None,
|
|
inject_timestamps: bool = False,
|
|
) -> tuple[List[Dict[str, Any]], Optional[str]]:
|
|
"""Convert stored gateway transcript rows into agent replay messages.
|
|
|
|
Keeping that context out of ``conversation_history`` stops consecutive-user repair merging it
|
|
with the live turn and hiding the current message behind ``history_offset`` on persistence.
|
|
"""
|
|
|
|
from hermes_time import get_timezone as _get_msg_tz
|
|
from gateway.message_timestamps import (
|
|
render_user_content_with_timestamp as _render_msg_ts,
|
|
)
|
|
|
|
_msg_tz = _get_msg_tz()
|
|
agent_history: List[Dict[str, Any]] = []
|
|
observed_group_context: List[str] = []
|
|
separate_observed_context = _uses_telegram_observed_group_context(channel_prompt)
|
|
|
|
for msg in history or []:
|
|
role = msg.get("role")
|
|
if not role:
|
|
continue
|
|
|
|
# Skip metadata entries (tool definitions, session info) -- these are
|
|
# for transcript logging, not for the LLM.
|
|
if role in {"session_meta",}:
|
|
continue
|
|
|
|
# Skip system messages -- the agent rebuilds its own system prompt.
|
|
if role == "system":
|
|
continue
|
|
|
|
content = msg.get("content")
|
|
if inject_timestamps and role == "user" and isinstance(content, str):
|
|
content = _render_msg_ts(content, msg.get("timestamp"), tz=_msg_tz)
|
|
if separate_observed_context and msg.get("observed") and role == "user" and content:
|
|
observed_group_context.append(str(content).strip())
|
|
continue
|
|
|
|
# Rich agent messages (tool_calls, tool results) must be passed through
|
|
# intact so the API sees valid assistant→tool sequences.
|
|
has_tool_calls = "tool_calls" in msg
|
|
has_tool_call_id = "tool_call_id" in msg
|
|
is_tool_message = role == "tool"
|
|
|
|
if has_tool_calls or has_tool_call_id or is_tool_message:
|
|
clean_msg = {k: v for k, v in msg.items() if k not in {"timestamp", "observed"}}
|
|
agent_history.append(clean_msg)
|
|
elif content:
|
|
# Strip persisted auto-continue notes from user messages (interrupted turns): keep the
|
|
# user's real text but never replay the recovery instruction — it caused infinite loops.
|
|
if role == "user":
|
|
content = _strip_auto_continue_noise(content)
|
|
if not content:
|
|
continue
|
|
# Simple text message - just need role and content.
|
|
if msg.get("mirror"):
|
|
mirror_src = msg.get("mirror_source", "another session")
|
|
content = f"[Delivered from {mirror_src}] {content}"
|
|
# Preserve the timestamp on user messages so the stale-dangerous-confirmation stripper
|
|
# in agent/replay_cleanup.py can read it.
|
|
entry = _build_replay_entry(role, content, msg, preserve_timestamp=(role == "user"))
|
|
agent_history.append(entry)
|
|
|
|
# Strip interrupted tool-call tails so the LLM doesn't re-execute
|
|
# tools that were killed mid-flight.
|
|
agent_history = strip_interrupted_tool_tails(agent_history)
|
|
|
|
# Strip a dangling assistant(tool_calls) tail with no tool answers — the signature of a SIGKILL
|
|
# mid-tool-call (e.g. the tool ran `docker restart`/`kill` and took the gateway down before the
|
|
# result persisted). Else the model re-issues the unanswered call on resume and loops forever.
|
|
agent_history = strip_dangling_tool_call_tail(agent_history)
|
|
|
|
# Strip expired dangerous-confirmation phrases (e.g. "confirm forced restart") from user text:
|
|
# replayed, an unrelated follow-up could read as a fresh confirmation and re-trigger the action.
|
|
agent_history = strip_stale_dangerous_confirmations(
|
|
agent_history, now=time.time()
|
|
)
|
|
|
|
observed_context = "\n".join(observed_group_context).strip() or None
|
|
return agent_history, observed_context
|
|
|
|
|
|
def _select_cached_agent_history(
|
|
persisted_history: List[Dict[str, Any]],
|
|
live_history: Any,
|
|
) -> List[Dict[str, Any]]:
|
|
"""Prefer a cached live transcript only when it is longer and has at least one real,
|
|
non-ephemeral unpersisted row; otherwise return ``persisted_history`` unchanged.
|
|
|
|
Guards the FTS write-corruption case: silent write failures make the next turn reload a stale
|
|
``conversation_history`` while the cached ``AIAgent`` still holds unpersisted real rows;
|
|
replacing them causes same-session amnesia. Length alone is not enough: a longer all-durable
|
|
list can be an expected replay-filtering delta, and unpersisted retry scaffolding is ignored.
|
|
"""
|
|
if isinstance(live_history, list) and len(live_history) > len(persisted_history):
|
|
from run_agent import _is_ephemeral_scaffolding
|
|
|
|
has_unpersisted_row = any(
|
|
isinstance(message, dict)
|
|
and not message.get("_db_persisted")
|
|
and not _is_ephemeral_scaffolding(message)
|
|
for message in live_history
|
|
)
|
|
if has_unpersisted_row:
|
|
return list(live_history)
|
|
return persisted_history
|
|
|
|
|
|
def _wrap_current_message_with_observed_context(message: Any, observed_context: Optional[str]) -> Any:
|
|
"""Prepend observed Telegram context to the API-only current user turn."""
|
|
|
|
if not observed_context:
|
|
return message
|
|
|
|
prefix = (
|
|
f"{_OBSERVED_GROUP_CONTEXT_HEADER}\n"
|
|
f"{observed_context}\n\n"
|
|
f"{_CURRENT_ADDRESSED_MESSAGE_HEADER}\n"
|
|
)
|
|
|
|
if isinstance(message, str):
|
|
return f"{prefix}{message}"
|
|
|
|
if isinstance(message, list):
|
|
wrapped = [dict(part) if isinstance(part, dict) else part for part in message]
|
|
for part in wrapped:
|
|
if isinstance(part, dict) and part.get("type") == "text":
|
|
part["text"] = f"{prefix}{part.get('text', '')}"
|
|
return wrapped
|
|
return [{"type": "text", "text": prefix.rstrip()}] + wrapped
|
|
|
|
return message
|
|
|
|
|
|
def _last_transcript_timestamp(history: Optional[List[Dict[str, Any]]]) -> Any:
|
|
"""Return the ``timestamp`` of the last usable transcript row, if any.
|
|
|
|
Skips metadata-only rows dropped before reaching the agent. ``None`` when no usable row has
|
|
a timestamp — callers treat that as "fresh" for backward compatibility.
|
|
"""
|
|
if not history:
|
|
return None
|
|
for msg in reversed(history):
|
|
if not isinstance(msg, dict):
|
|
continue
|
|
role = msg.get("role")
|
|
if not role or role in {"session_meta", "system"}:
|
|
continue
|
|
ts = msg.get("timestamp")
|
|
if ts is not None:
|
|
return ts
|
|
# First non-meta row without a timestamp — legacy transcript row.
|
|
# Returning None lets the caller fall through to the legacy-fresh path.
|
|
return None
|
|
return None
|
|
|
|
|
|
# Tool output may hold literal MEDIA: examples (docs, logs); only tools that intentionally create
|
|
# deliverable media are eligible for auto-append when the model omits them from the final reply.
|
|
_AUTO_APPEND_MEDIA_TOOL_NAMES = {
|
|
"text_to_speech",
|
|
"text_to_speech_tool",
|
|
"image_generate",
|
|
}
|
|
|
|
# ---- helpers: detect interrupted tool tails & auto-continue noise ----------
|
|
|
|
# Replay-tail sanitization lives in agent/replay_cleanup.py so every resume surface (this messaging
|
|
# gateway AND the TUI/WebUI gateway) shares one implementation.
|
|
from agent.replay_cleanup import ( # noqa: E402
|
|
strip_interrupted_tool_tails,
|
|
strip_dangling_tool_call_tail,
|
|
strip_stale_dangerous_confirmations,
|
|
)
|
|
|
|
|
|
_AUTO_CONTINUE_NOTE_PREFIX = "[System note: Your previous turn"
|
|
_AUTO_CONTINUE_FALLBACK_PREFIX = "[System note: A new message"
|
|
|
|
|
|
def _is_auto_continue_noise(content: Any) -> bool:
|
|
"""Return True if this user-message content is a gateway-injected
|
|
auto-continue note that should NOT be replayed as a real user turn."""
|
|
if not isinstance(content, str):
|
|
return False
|
|
return (
|
|
content.startswith(_AUTO_CONTINUE_NOTE_PREFIX)
|
|
or content.startswith(_AUTO_CONTINUE_FALLBACK_PREFIX)
|
|
)
|
|
|
|
|
|
def _strip_auto_continue_noise(content: Any) -> Any:
|
|
"""Strip one or more leading persisted auto-continue note prefixes from user text.
|
|
|
|
A row may hold both the note and the user's real question; the trailing real text is preserved.
|
|
"""
|
|
if not _is_auto_continue_noise(content):
|
|
return content
|
|
text = str(content)
|
|
while _is_auto_continue_noise(text):
|
|
end = text.find("]")
|
|
if end < 0:
|
|
return ""
|
|
text = text[end + 1 :].lstrip()
|
|
return text
|
|
|
|
# Tools whose deliverable is a JSON payload with a local-file path field rather than a literal
|
|
# ``MEDIA:`` tag (e.g. image_generate -> ``{"success": true, "image": "/abs/path.png"}``).
|
|
_JSON_MEDIA_TOOL_PATH_FIELDS = ("host_image", "image", "agent_visible_image")
|
|
|
|
|
|
# Extension-anchored MEDIA: matcher for tool results. Mirrors the dispatch-site pattern so a bare
|
|
# ``MEDIA:`` token in prose (no deliverable extension) is never auto-appended.
|
|
_TOOL_MEDIA_RE = re.compile(
|
|
r'MEDIA:((?:[A-Za-z]:[/\\]|/|~\/)\S+\.(?:png|jpe?g|gif|webp|'
|
|
r'mp4|mov|avi|mkv|webm|ogg|opus|mp3|wav|m4a|'
|
|
r'flac|epub|pdf|zip|rar|7z|docx?|xlsx?|pptx?|'
|
|
r'txt|csv|apk|ipa))',
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
# Shared with cron delivery and gateway background tasks — the repair must run on every surface
|
|
# that feeds a final response into media extraction; canonical names live in gateway.media_repair.
|
|
from gateway.media_repair import ( # noqa: E402
|
|
tool_name_by_call_id as _tool_name_by_call_id,
|
|
)
|
|
|
|
|
|
def _collect_auto_append_media_tags(
|
|
messages: List[Dict[str, Any]],
|
|
history_offset: int = 0,
|
|
history_media_paths: Optional[set] = None,
|
|
) -> tuple[List[str], bool]:
|
|
"""Collect real media tags from current-turn producer-tool results only.
|
|
|
|
Two guards: a producer-tool allowlist (docs/logs/search results contain example MEDIA: strings
|
|
that must never become attachments) and current-turn isolation (no leaking an earlier turn's
|
|
result). If mid-run compression shrank the list below the original history length the slice
|
|
is untrustworthy: scan every message, dedup via ``history_media_paths``.
|
|
"""
|
|
history_media_paths = history_media_paths or set()
|
|
# Only trust the slice boundary when the message list still contains the
|
|
# full history prefix. Otherwise scan everything (compression-safe fallback).
|
|
if history_offset and len(messages) >= history_offset:
|
|
new_messages = messages[history_offset:]
|
|
else:
|
|
new_messages = messages
|
|
|
|
tool_name_by_call_id = _tool_name_by_call_id(new_messages)
|
|
|
|
media_tags: List[str] = []
|
|
has_voice_directive = False
|
|
for msg in new_messages:
|
|
if msg.get("role") not in ("tool", "function"):
|
|
continue
|
|
call_id = str(msg.get("tool_call_id") or msg.get("call_id") or "")
|
|
if tool_name_by_call_id.get(call_id) not in _AUTO_APPEND_MEDIA_TOOL_NAMES:
|
|
continue
|
|
content = str(msg.get("content") or "")
|
|
tool_name = tool_name_by_call_id.get(call_id)
|
|
# JSON-payload tools (image_generate) return a local-file path in a known field, not a
|
|
# MEDIA: tag; extract it so delivery is deterministic even if the model omits the path.
|
|
if tool_name == "image_generate" and "MEDIA:" not in content:
|
|
try:
|
|
payload = json.loads(content)
|
|
except Exception:
|
|
payload = None
|
|
if isinstance(payload, dict) and payload.get("success"):
|
|
for field in _JSON_MEDIA_TOOL_PATH_FIELDS:
|
|
path = payload.get(field)
|
|
if (isinstance(path, str)
|
|
and _TOOL_MEDIA_RE.fullmatch(f"MEDIA:{path}")
|
|
and path not in history_media_paths):
|
|
media_tags.append(f"MEDIA:{path}")
|
|
break
|
|
continue
|
|
if "MEDIA:" not in content:
|
|
continue
|
|
for match in _TOOL_MEDIA_RE.finditer(content):
|
|
path = match.group(1).strip().rstrip('",}')
|
|
if path and path not in history_media_paths:
|
|
media_tags.append(f"MEDIA:{path}")
|
|
if "[[audio_as_voice]]" in content:
|
|
has_voice_directive = True
|
|
|
|
return media_tags, has_voice_directive
|
|
|
|
|
|
def _collect_history_media_paths(agent_history: List[Dict[str, Any]]) -> set:
|
|
"""Collect every media path already delivered in prior assistant/tool output.
|
|
|
|
Used to dedup auto-appended and model-emitted MEDIA tags so a file is not re-sent later; both
|
|
the JSON-payload and assistant-message shapes must be covered or delivery repeats.
|
|
"""
|
|
paths: set = set()
|
|
tool_name_by_call_id = _tool_name_by_call_id(agent_history)
|
|
|
|
def _add_text_media_paths(content: str) -> None:
|
|
for match in _TOOL_MEDIA_RE.finditer(content):
|
|
path = match.group(1).strip().rstrip('",}')
|
|
if path:
|
|
paths.add(path)
|
|
# The regex alone misses quoted/spaced paths that extract_media accepts — use the same
|
|
# extractor so the dedup set sees every path that could actually have been delivered.
|
|
media_files, _ = BasePlatformAdapter.extract_media(content)
|
|
paths.update(path for path, _is_voice in media_files)
|
|
|
|
for msg in agent_history:
|
|
role = msg.get("role")
|
|
if role == "assistant":
|
|
content = str(msg.get("content", "") or "")
|
|
if "MEDIA:" in content:
|
|
_add_text_media_paths(content)
|
|
continue
|
|
if role not in {"tool", "function"}:
|
|
continue
|
|
content = str(msg.get("content", "") or "")
|
|
if "MEDIA:" in content:
|
|
_add_text_media_paths(content)
|
|
continue
|
|
cid = str(msg.get("tool_call_id") or msg.get("call_id") or "")
|
|
if tool_name_by_call_id.get(cid) == "image_generate":
|
|
try:
|
|
payload = json.loads(content)
|
|
except Exception:
|
|
payload = None
|
|
if isinstance(payload, dict) and payload.get("success"):
|
|
for field in _JSON_MEDIA_TOOL_PATH_FIELDS:
|
|
jp = payload.get(field)
|
|
if isinstance(jp, str) and jp:
|
|
paths.add(jp)
|
|
break
|
|
return paths
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# SSL certificate auto-detection for NixOS and other non-standard systems.
|
|
# Must run BEFORE any HTTP library (discord, aiohttp, etc.) is imported.
|
|
# ---------------------------------------------------------------------------
|
|
def _ensure_ssl_certs() -> None:
|
|
"""Set SSL_CERT_FILE if the system doesn't expose CA certs to Python.
|
|
|
|
A set-but-missing path makes every later httpx/OpenAI client fail in ssl.load_verify_locations(),
|
|
so treat it as unset and fall back to certifi.
|
|
"""
|
|
configured_cert = os.environ.get("SSL_CERT_FILE")
|
|
if configured_cert:
|
|
if os.path.exists(configured_cert):
|
|
return # user already configured it to a real file
|
|
logging.getLogger(__name__).warning(
|
|
"Ignoring stale SSL_CERT_FILE=%r because the path does not exist",
|
|
configured_cert,
|
|
)
|
|
os.environ.pop("SSL_CERT_FILE", None)
|
|
|
|
import ssl
|
|
|
|
# 1. Python's compiled-in defaults
|
|
paths = ssl.get_default_verify_paths()
|
|
for candidate in (paths.cafile, paths.openssl_cafile):
|
|
if candidate and os.path.exists(candidate):
|
|
os.environ["SSL_CERT_FILE"] = candidate
|
|
return
|
|
|
|
# 2. certifi (ships its own Mozilla bundle)
|
|
try:
|
|
import certifi
|
|
os.environ["SSL_CERT_FILE"] = certifi.where()
|
|
return
|
|
except ImportError:
|
|
pass
|
|
|
|
# 3. Common distro / macOS locations
|
|
for candidate in (
|
|
"/etc/ssl/certs/ca-certificates.crt", # Debian/Ubuntu/Gentoo
|
|
"/etc/pki/tls/certs/ca-bundle.crt", # RHEL/CentOS 7
|
|
"/etc/pki/ca-trust/extracted/pem/tls-ca-bundle.pem", # RHEL/CentOS 8+
|
|
"/etc/ssl/ca-bundle.pem", # SUSE/OpenSUSE
|
|
"/etc/ssl/cert.pem", # Alpine / macOS
|
|
"/etc/pki/tls/cert.pem", # Fedora
|
|
"/usr/local/etc/openssl@1.1/cert.pem", # macOS Homebrew Intel
|
|
"/opt/homebrew/etc/openssl@1.1/cert.pem", # macOS Homebrew ARM
|
|
):
|
|
if os.path.exists(candidate):
|
|
os.environ["SSL_CERT_FILE"] = candidate
|
|
return
|
|
|
|
def _home_target_env_var(platform_name: str) -> str:
|
|
"""Return the configured home-target env var for a platform.
|
|
|
|
Built-in ``_HOME_TARGET_ENV_VARS`` first, then the plugin registry
|
|
(``cron.scheduler._resolve_home_env_var``), then ``<PLATFORM>_HOME_CHANNEL`` for unknown names.
|
|
"""
|
|
from cron.scheduler import _resolve_home_env_var
|
|
|
|
resolved = _resolve_home_env_var(platform_name)
|
|
if resolved:
|
|
return resolved
|
|
return f"{platform_name.upper()}_HOME_CHANNEL"
|
|
|
|
|
|
def _home_thread_env_var(platform_name: str) -> str:
|
|
"""Return the optional thread/topic env var for a platform home target."""
|
|
return f"{_home_target_env_var(platform_name)}_THREAD_ID"
|
|
|
|
|
|
def _restart_notification_pending() -> bool:
|
|
"""Return True when a /restart completion marker is waiting to be delivered."""
|
|
return (_hermes_home / ".restart_notify.json").exists()
|
|
|
|
|
|
def _planned_restart_notification_path() -> Path:
|
|
return _hermes_home / ".restart_pending.json"
|
|
|
|
|
|
def _planned_restart_notification_pending() -> bool:
|
|
"""Return True when a non-chat planned restart should notify home channels."""
|
|
return _planned_restart_notification_path().exists()
|
|
|
|
|
|
def _clear_planned_restart_notification() -> None:
|
|
_planned_restart_notification_path().unlink(missing_ok=True)
|
|
|
|
|
|
# Mark this process as a gateway so cli.py's module-level load_cli_config()
|
|
# knows not to clobber TERMINAL_CWD if lazily imported.
|
|
os.environ["_HERMES_GATEWAY"] = "1"
|
|
|
|
_ensure_ssl_certs()
|
|
|
|
# Add parent directory to path
|
|
sys.path.insert(0, str(Path(__file__).parent.parent))
|
|
|
|
# Resolve Hermes home directory (respects HERMES_HOME override)
|
|
from hermes_constants import get_hermes_home, get_hermes_home_override
|
|
from utils import atomic_json_write, base_url_hostname, is_truthy_value # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.<name>)
|
|
_hermes_home = get_hermes_home()
|
|
|
|
# Load environment variables from ~/.hermes/.env first.
|
|
# User-managed env files should override stale shell exports on restart.
|
|
from dotenv import load_dotenv # noqa: F401 # backward-compat for tests that monkeypatch this symbol
|
|
from hermes_cli.env_loader import load_hermes_dotenv
|
|
_env_path = _hermes_home / '.env'
|
|
load_hermes_dotenv(hermes_home=_hermes_home, project_env=Path(__file__).resolve().parents[1] / '.env')
|
|
|
|
|
|
def _reload_runtime_env_preserving_config_authority() -> None:
|
|
"""Reload .env for fresh credentials without letting stale .env override config.
|
|
|
|
Long-lived gateways reload ~/.hermes/.env per turn for rotated keys; config.yaml stays
|
|
authoritative for budget settings (else stale HERMES_MAX_ITERATIONS wins). NO-OP in multiplex
|
|
mode: secrets come from the per-turn ``set_secret_scope`` mapping, and mutating ``os.environ``
|
|
would leak the default profile's keys to every profile.
|
|
"""
|
|
from agent.secret_scope import is_multiplex_active
|
|
if is_multiplex_active():
|
|
# Credentials come from the active profile's secret scope, not os.environ: still honor the
|
|
# config.yaml agent.max_turns bridge below (scoped home), but never reload .env globally.
|
|
_bridge_max_turns_from_config(_hermes_home)
|
|
return
|
|
|
|
load_hermes_dotenv(
|
|
hermes_home=_hermes_home,
|
|
project_env=Path(__file__).resolve().parents[1] / '.env',
|
|
)
|
|
_bridge_max_turns_from_config(_hermes_home)
|
|
|
|
|
|
def _bridge_max_turns_from_config(home: "Path") -> None:
|
|
"""Bridge config.yaml agent.max_turns into HERMES_MAX_ITERATIONS (a global)."""
|
|
config_path = home / 'config.yaml'
|
|
if not config_path.exists():
|
|
return
|
|
try:
|
|
from hermes_cli.config import _expand_env_vars, read_user_config_raw
|
|
# Presence-sensitive env bridge: raw read is deliberate (only keys the
|
|
# user actually wrote get bridged); overlay + expansion applied below.
|
|
cfg = read_user_config_raw(config_path)
|
|
cfg = _expand_env_vars(cfg)
|
|
if not isinstance(cfg, dict):
|
|
cfg = {}
|
|
# Managed scope: the per-turn reload re-bridges config→env, so without the overlay a managed
|
|
# agent.max_turns/timezone/redact_secrets would revert to the user's value after one turn.
|
|
try:
|
|
from hermes_cli import managed_scope
|
|
cfg = managed_scope.apply_managed_overlay(cfg)
|
|
except Exception:
|
|
pass
|
|
except Exception:
|
|
return
|
|
|
|
agent_cfg = cfg.get("agent", {})
|
|
if isinstance(agent_cfg, dict) and "max_turns" in agent_cfg:
|
|
raw = agent_cfg["max_turns"]
|
|
# Preserve the raw spelling ("none", "unlimited", "120") so resolve_turn_limit() in
|
|
# _current_max_iterations can interpret it. Skip Python None (`null` / bare `key:`):
|
|
# str(None) -> "None" would map to the unlimited sentinel instead of "absent = default".
|
|
if raw is not None:
|
|
os.environ["HERMES_MAX_ITERATIONS"] = str(raw)
|
|
elif "HERMES_MAX_ITERATIONS" in os.environ:
|
|
# Clear stale bridge so downstream resolver applies its default.
|
|
del os.environ["HERMES_MAX_ITERATIONS"]
|
|
# config-authoritative knobs for the session-search index (config.yaml
|
|
# sessions.* wins over stale env; env stays the cross-process carrier).
|
|
sessions_cfg = cfg.get("sessions", {})
|
|
if isinstance(sessions_cfg, dict):
|
|
if "cjk_fts" in sessions_cfg:
|
|
os.environ["HERMES_CJK_FTS"] = str(sessions_cfg["cjk_fts"])
|
|
if "search_slow_ms" in sessions_cfg:
|
|
os.environ["HERMES_SEARCH_SLOW_MS"] = str(sessions_cfg["search_slow_ms"])
|
|
|
|
|
|
def _current_max_iterations() -> int:
|
|
"""Return the current per-turn iteration budget after runtime env refresh.
|
|
|
|
Uses ``resolve_turn_limit`` so ``agent.max_turns: none``/``unlimited`` (bridged as a string
|
|
into ``HERMES_MAX_ITERATIONS``) yields the unlimited sentinel instead of an ``int()`` crash.
|
|
"""
|
|
_reload_runtime_env_preserving_config_authority()
|
|
from hermes_cli.config import resolve_turn_limit as _resolve_turn_limit
|
|
return _resolve_turn_limit(os.getenv("HERMES_MAX_ITERATIONS"))
|
|
|
|
|
|
from contextlib import (
|
|
asynccontextmanager as _asynccontextmanager,
|
|
contextmanager as _contextmanager,
|
|
suppress,
|
|
)
|
|
|
|
|
|
# Platforms that bind a host TCP port. In a profile multiplexer the default profile owns the single
|
|
# shared listener (serving every profile via the /p/<profile>/ prefix), so a SECONDARY profile
|
|
# enabling one is always a misconfiguration and is skipped (SecondaryPortBindingConfigError) rather
|
|
# than taking down the multiplexer. Lives in gateway.config so dashboard validation enforces it too.
|
|
|
|
|
|
class MultiplexConfigError(RuntimeError):
|
|
"""A profile multiplexer config is invalid.
|
|
|
|
Distinct from a transient adapter-connect failure: the operator must fix config.yaml, so it
|
|
propagates to the startup guard instead of being treated as retryable adapter noise.
|
|
"""
|
|
|
|
|
|
class SecondaryPortBindingConfigError(MultiplexConfigError):
|
|
"""A secondary profile conflicts with the multiplexer's shared listener."""
|
|
|
|
|
|
class HygieneTurnHoldExceeded(Exception):
|
|
"""The hygiene-compression turn-hold budget elapsed while the summary model was still streaming.
|
|
|
|
An availability boundary, not a failure: the compressor is healthy but the user turn cannot
|
|
wait. Must NOT be routed through the idle-timeout failure path (AGENT_COMPRESSION_TIMEOUT,
|
|
"no output" message, failure cooldown ladder).
|
|
"""
|
|
|
|
|
|
def _multiplex_profile_homes(config: object) -> list[tuple[str, "Path"]]:
|
|
"""Return the authoritative profile set for one multiplex gateway config."""
|
|
from hermes_cli.profiles import profiles_to_serve
|
|
|
|
return list(
|
|
profiles_to_serve(
|
|
multiplex=True,
|
|
profile_allowlist=getattr(config, "multiplex_profile_allowlist", None),
|
|
)
|
|
)
|
|
|
|
|
|
def _enable_multiplex_log_routing(config: object) -> bool:
|
|
"""Route agent.log/errors.log/gateway.log records to their owning profile.
|
|
|
|
``setup_logging(mode="gateway")`` binds the file handlers to the launch home, so under
|
|
``multiplex_profiles`` every secondary profile's records land in the default profile's logs.
|
|
Swap in the profile routers once the served-profile set is known; inert for single-profile.
|
|
"""
|
|
if not getattr(config, "multiplex_profiles", False):
|
|
return False
|
|
try:
|
|
from hermes_logging import enable_profile_log_routing
|
|
|
|
return enable_profile_log_routing(
|
|
[home for _name, home in _multiplex_profile_homes(config)]
|
|
)
|
|
except Exception:
|
|
logger.debug("could not enable per-profile log routing", exc_info=True)
|
|
return False
|
|
|
|
|
|
def _handoff_watch_scopes(runner: object) -> list:
|
|
"""``(profile_name, home)`` pairs whose ``state.db`` the watcher must poll.
|
|
|
|
``/handoff`` writes into the store of the profile the CLI ran under, but an unscoped watcher
|
|
polls only the ROOT store, so a secondary profile's handoff is never seen and the CLI times out.
|
|
``(None, None)`` = root poll, always first; SECONDARY profiles are added, the default is not
|
|
repeated. Defensive: a raising resolver would silently disable the watcher, so failures
|
|
degrade to the root poll.
|
|
"""
|
|
scopes: list = [(None, None)]
|
|
try:
|
|
config = getattr(runner, "config", None)
|
|
if config is not None and getattr(config, "multiplex_profiles", False):
|
|
for name, home in _multiplex_profile_homes(config):
|
|
if home is None or not name or name == "default":
|
|
continue
|
|
scopes.append((name, home))
|
|
except Exception:
|
|
logger.debug("Could not resolve multiplex homes for handoff watcher", exc_info=True)
|
|
return scopes
|
|
|
|
|
|
async def _reclaim_stale(runner: object) -> None:
|
|
"""Fail handoffs left in ``running`` by a gateway that died mid-dispatch.
|
|
|
|
Runs once per store at watcher startup. ``running`` is only set for one in-process dispatch,
|
|
so a row still in it belongs to a dead process, can never reach a terminal state, and blocks
|
|
``request_handoff`` for that session forever. Defensive: a raising reclaim would abort startup.
|
|
"""
|
|
session_db = getattr(runner, "_session_db", None)
|
|
if session_db is None:
|
|
return
|
|
reclaim = getattr(session_db, "reclaim_stale_running_handoffs", None)
|
|
if not callable(reclaim):
|
|
return
|
|
try:
|
|
ids = await reclaim(
|
|
"gateway stopped mid-handoff; state reclaimed at startup. "
|
|
"Re-run /handoff to try again."
|
|
)
|
|
except Exception:
|
|
logger.debug("Stale-handoff reclaim raised", exc_info=True)
|
|
return
|
|
if ids:
|
|
logger.warning(
|
|
"Reclaimed %d handoff(s) stranded in 'running' by a previous "
|
|
"gateway: %s", len(ids), ", ".join(str(i) for i in ids),
|
|
)
|
|
|
|
|
|
def _terminal_scope_cwd(default: str = "") -> str:
|
|
"""Scope-aware TERMINAL_CWD read for footer/context surfaces.
|
|
|
|
Only an import failure falls back: an active refusal scope must raise, not use the launch cwd.
|
|
"""
|
|
try:
|
|
from tools.terminal_scope import terminal_env as _ts_env
|
|
except ImportError:
|
|
return os.environ.get("TERMINAL_CWD", default)
|
|
return _ts_env("TERMINAL_CWD", default)
|
|
|
|
|
|
def _load_profile_secret_scope(profile_home: "Path") -> dict:
|
|
"""Hydrate and load one profile's secrets under its home override."""
|
|
from hermes_constants import set_hermes_home_override, reset_hermes_home_override
|
|
from agent.secret_scope import build_profile_secret_scope
|
|
from hermes_cli.env_loader import hydrate_profile_secret_sources
|
|
|
|
home_token = set_hermes_home_override(str(profile_home))
|
|
try:
|
|
hydrate_profile_secret_sources(Path(profile_home))
|
|
return build_profile_secret_scope(Path(profile_home))
|
|
finally:
|
|
reset_hermes_home_override(home_token)
|
|
|
|
|
|
@_contextmanager
|
|
def _profile_runtime_scope(
|
|
profile_home: "Path",
|
|
prepared_secret_scope: Optional[dict] = None,
|
|
*,
|
|
hydrate_secrets: bool = True,
|
|
):
|
|
"""Scope config/skills/memory AND credentials to a profile for one turn (multiplexed path only).
|
|
|
|
(1) ``set_hermes_home_override`` redirects ``get_hermes_home()`` — a contextvar, so it reaches
|
|
the agent worker thread via ``copy_context()``; (2) ``set_secret_scope`` makes the profile's
|
|
``.env`` the credential source so ``get_secret`` never reads ``os.environ``. Loading ``.env``
|
|
does NOT mutate ``os.environ``, which keeps subprocesses from inheriting cross-profile secrets.
|
|
"""
|
|
from hermes_constants import set_hermes_home_override, reset_hermes_home_override
|
|
from agent.secret_scope import (
|
|
set_secret_scope,
|
|
reset_secret_scope,
|
|
)
|
|
|
|
home_token = set_hermes_home_override(str(profile_home))
|
|
if prepared_secret_scope is not None:
|
|
secrets = prepared_secret_scope
|
|
elif hydrate_secrets:
|
|
secrets = _load_profile_secret_scope(Path(profile_home))
|
|
else:
|
|
# Caller already hydrated external sources off-loop (#99519).
|
|
from agent.secret_scope import build_profile_secret_scope
|
|
|
|
secrets = build_profile_secret_scope(Path(profile_home))
|
|
secret_token = set_secret_scope(secrets)
|
|
# Per-turn terminal scope (third seam of the profile boundary): install the routed profile's
|
|
# COMPLETE terminal policy — never ambient env — via tools.terminal_scope, else terminal_tool
|
|
# reads process-global TERMINAL_* vars a prior profile's turn pinned (first-writer-wins leak).
|
|
from tools.terminal_scope import install_and_reset_profile_terminal_scope
|
|
|
|
with install_and_reset_profile_terminal_scope(Path(profile_home)):
|
|
try:
|
|
yield
|
|
finally:
|
|
reset_secret_scope(secret_token)
|
|
reset_hermes_home_override(home_token)
|
|
|
|
|
|
@_asynccontextmanager
|
|
async def _async_profile_runtime_scope(profile_home: "Path"):
|
|
"""Enter a profile scope without loading secret files on the event loop."""
|
|
secrets = await asyncio.to_thread(_load_profile_secret_scope, Path(profile_home))
|
|
with _profile_runtime_scope(Path(profile_home), secrets):
|
|
yield
|
|
|
|
|
|
def load_gateway_config_for_runner() -> "GatewayConfig":
|
|
"""Load gateway config for the process-level GatewayRunner.
|
|
|
|
With multiplexing on, reload under the default profile's ``_profile_runtime_scope`` so platform
|
|
tokens in that profile's ``.env`` resolve through the secret scope (as secondary profiles do);
|
|
unscoped, ``_getenv`` falls through to ``os.environ``, which often lacks a token that lives only
|
|
under ``profiles/<name>/.env``. Off -> identical to ``load_gateway_config()``.
|
|
"""
|
|
cfg = load_gateway_config()
|
|
if not getattr(cfg, "multiplex_profiles", False):
|
|
return cfg
|
|
try:
|
|
home = get_hermes_home()
|
|
except Exception:
|
|
return cfg
|
|
try:
|
|
with _profile_runtime_scope(Path(home)):
|
|
return load_gateway_config()
|
|
except Exception:
|
|
logger.debug(
|
|
"multiplex default-scope config reload failed; using unscoped load",
|
|
exc_info=True,
|
|
)
|
|
return cfg
|
|
|
|
|
|
async def _discover_gateway_mcp_tools(config: object) -> None:
|
|
"""Run startup MCP discovery for every profile this gateway serves.
|
|
|
|
``discover_mcp_tools`` reads ``mcp_servers`` from ``get_hermes_home()``'s config, so an unscoped
|
|
call only connects the launch profile's servers. Single-profile gateways keep the unscoped call.
|
|
"""
|
|
from tools.mcp_tool import discover_mcp_tools
|
|
|
|
loop = asyncio.get_running_loop()
|
|
if not getattr(config, "multiplex_profiles", False):
|
|
await loop.run_in_executor(None, discover_mcp_tools)
|
|
return
|
|
for profile_name, profile_home in _multiplex_profile_homes(config):
|
|
try:
|
|
with _profile_runtime_scope(Path(profile_home)):
|
|
await loop.run_in_executor(None, copy_context().run, discover_mcp_tools)
|
|
except Exception:
|
|
logger.warning(
|
|
"MCP tool discovery failed for profile '%s'", profile_name, exc_info=True,
|
|
)
|
|
|
|
|
|
def _platform_has_bot_credential(platform: "Platform", platform_config: "PlatformConfig") -> bool:
|
|
"""Return True when a token-authenticated platform has a usable bot credential.
|
|
|
|
Platforms that do not use ``PlatformConfig.token`` always return True so we
|
|
never skip them here (Signal session paths, port-binding HTTP adapters, etc.).
|
|
"""
|
|
from gateway.config import PLATFORM_TOKEN_ENV_NAMES, Platform
|
|
|
|
if platform not in PLATFORM_TOKEN_ENV_NAMES:
|
|
return True
|
|
token = getattr(platform_config, "token", None) or ""
|
|
if isinstance(token, str) and token.strip():
|
|
return True
|
|
# Some adapters also accept api_key as the primary credential.
|
|
api_key = getattr(platform_config, "api_key", None) or ""
|
|
if isinstance(api_key, str) and api_key.strip():
|
|
return True
|
|
# Matrix also authenticates by password login (MATRIX_USER_ID + MATRIX_PASSWORD in ``extra``),
|
|
# so a token-only check would evict a reconnectable config from the retry queue on the first
|
|
# transient failure; mirror the adapter's gate: homeserver + user_id + password. Read ONLY from
|
|
# extra (build_config() already copies env vars there) — an env fallback would report "has
|
|
# credential" for every Matrix config on the box, including the empty-primary multiplex case.
|
|
if platform is Platform.MATRIX:
|
|
extra = getattr(platform_config, "extra", None) or {}
|
|
if all(
|
|
str(extra.get(key) or "").strip()
|
|
for key in ("homeserver", "user_id", "password")
|
|
):
|
|
return True
|
|
return False
|
|
|
|
|
|
_DOCKER_VOLUME_SPEC_RE = re.compile(r"^(?P<host>.+):(?P<container>/[^:]+?)(?::(?P<options>[^:]+))?$")
|
|
_DOCKER_MEDIA_OUTPUT_CONTAINER_PATHS = {"/output", "/outputs"}
|
|
|
|
# Internal bridge plumbing, not a user-facing config source: initialize from the canonical config
|
|
# default after dotenv loading so an ambient process/.env value can never control lease safety.
|
|
from hermes_cli.config_defaults import DEFAULT_CONFIG as _DEFAULT_CONFIG
|
|
|
|
os.environ["HERMES_TURN_LEASE_TIMEOUT"] = str(
|
|
_DEFAULT_CONFIG["agent"]["gateway_turn_lease_timeout"]
|
|
)
|
|
|
|
# Bridge config.yaml values into the environment so os.getenv() picks them up.
|
|
# config.yaml is authoritative for terminal settings — overrides .env.
|
|
_config_path = _hermes_home / 'config.yaml'
|
|
if _config_path.exists():
|
|
try:
|
|
# Presence-sensitive env bridge: raw read is deliberate — only keys the user actually wrote
|
|
# may be bridged (a defaults merge would export all DEFAULT_CONFIG); overlay applied below.
|
|
from hermes_cli.config import _expand_env_vars, read_user_config_raw
|
|
_cfg = read_user_config_raw(_config_path)
|
|
# Expand ${ENV_VAR} references before bridging to env vars.
|
|
_cfg = _expand_env_vars(_cfg)
|
|
if not isinstance(_cfg, dict):
|
|
_cfg = {}
|
|
# Managed scope: overlay administrator-pinned values BEFORE bridging so a managed timezone/
|
|
# redact_secrets/max_turns/terminal setting wins at the env layer too; fail-open via helper.
|
|
try:
|
|
from hermes_cli import managed_scope
|
|
_cfg = managed_scope.apply_managed_overlay(_cfg)
|
|
except Exception:
|
|
pass
|
|
# Top-level simple values (fallback only — don't override .env)
|
|
for _key, _val in _cfg.items():
|
|
if isinstance(_val, (str, int, float, bool)) and _key not in os.environ:
|
|
os.environ[_key] = str(_val)
|
|
# Terminal config is nested — bridge to TERMINAL_* env vars.
|
|
# config.yaml overrides .env for these since it's the documented config path.
|
|
_terminal_cfg = _cfg.get("terminal", {})
|
|
if _terminal_cfg and isinstance(_terminal_cfg, dict):
|
|
_terminal_backend = str(
|
|
_terminal_cfg.get("backend") or os.environ.get("TERMINAL_ENV") or ""
|
|
).strip().lower()
|
|
_terminal_env_map = {
|
|
"backend": "TERMINAL_ENV",
|
|
"degraded_mode": "TERMINAL_DEGRADED_MODE",
|
|
"cwd": "TERMINAL_CWD",
|
|
"timeout": "TERMINAL_TIMEOUT",
|
|
"home_mode": "TERMINAL_HOME_MODE",
|
|
"lifetime_seconds": "TERMINAL_LIFETIME_SECONDS",
|
|
"docker_image": "TERMINAL_DOCKER_IMAGE",
|
|
"docker_forward_env": "TERMINAL_DOCKER_FORWARD_ENV",
|
|
"singularity_image": "TERMINAL_SINGULARITY_IMAGE",
|
|
"modal_image": "TERMINAL_MODAL_IMAGE",
|
|
"daytona_image": "TERMINAL_DAYTONA_IMAGE",
|
|
"vercel_runtime": "TERMINAL_VERCEL_RUNTIME",
|
|
"ssh_host": "TERMINAL_SSH_HOST",
|
|
"ssh_user": "TERMINAL_SSH_USER",
|
|
"ssh_port": "TERMINAL_SSH_PORT",
|
|
"ssh_key": "TERMINAL_SSH_KEY",
|
|
"container_cpu": "TERMINAL_CONTAINER_CPU",
|
|
"container_memory": "TERMINAL_CONTAINER_MEMORY",
|
|
"container_disk": "TERMINAL_CONTAINER_DISK",
|
|
"container_persistent": "TERMINAL_CONTAINER_PERSISTENT",
|
|
"docker_volumes": "TERMINAL_DOCKER_VOLUMES",
|
|
"docker_env": "TERMINAL_DOCKER_ENV",
|
|
"docker_extra_args": "TERMINAL_DOCKER_EXTRA_ARGS",
|
|
"docker_shm_size": "TERMINAL_DOCKER_SHM_SIZE",
|
|
"docker_mount_cwd_to_workspace": "TERMINAL_DOCKER_MOUNT_CWD_TO_WORKSPACE",
|
|
"docker_network": "TERMINAL_DOCKER_NETWORK",
|
|
"docker_run_as_host_user": "TERMINAL_DOCKER_RUN_AS_HOST_USER",
|
|
"docker_persist_across_processes": "TERMINAL_DOCKER_PERSIST_ACROSS_PROCESSES",
|
|
"docker_shared_container_key": "TERMINAL_DOCKER_SHARED_CONTAINER_KEY",
|
|
"docker_orphan_reaper": "TERMINAL_DOCKER_ORPHAN_REAPER",
|
|
"sandbox_dir": "TERMINAL_SANDBOX_DIR",
|
|
"persistent_shell": "TERMINAL_PERSISTENT_SHELL",
|
|
}
|
|
for _cfg_key, _env_var in _terminal_env_map.items():
|
|
if _cfg_key in _terminal_cfg:
|
|
_val = _terminal_cfg[_cfg_key]
|
|
# Skip cwd placeholders (".", "auto", "cwd") — the gateway resolves them to
|
|
# Path.home() later; only bridge explicit absolute paths from config.yaml.
|
|
if _cfg_key == "cwd" and str(_val) in {".", "auto", "cwd"}:
|
|
continue
|
|
# Expand "~" in local/container cwd so subprocess.Popen never gets a literal
|
|
# "~/" (the kernel rejects it); SSH cwd is interpreted by the remote shell, so
|
|
# keep "~" there. Predicate shared with terminal_tool so the sites can't drift.
|
|
if _cfg_key == "cwd" and isinstance(_val, str):
|
|
if not _is_ssh_remote_tilde_cwd(_terminal_backend, _val.strip()):
|
|
_val = os.path.expanduser(_val)
|
|
if isinstance(_val, (list, dict)):
|
|
os.environ[_env_var] = json.dumps(_val)
|
|
else:
|
|
os.environ[_env_var] = str(_val)
|
|
# Compression config is read from config.yaml by run_agent.py/auxiliary_client.py (no env
|
|
# bridge). Auxiliary model/endpoint overrides: vision, approval, plugin-registered tasks.
|
|
_auxiliary_cfg = _cfg.get("auxiliary", {})
|
|
if _auxiliary_cfg and isinstance(_auxiliary_cfg, dict):
|
|
# Canonical built-in bridged set; plugin tasks are added below via the aux registry.
|
|
_aux_bridged_keys = {"vision", "approval"}
|
|
try:
|
|
from hermes_cli.plugins import get_plugin_auxiliary_tasks
|
|
for _entry in get_plugin_auxiliary_tasks():
|
|
_aux_bridged_keys.add(_entry["key"])
|
|
except Exception:
|
|
# Plugin discovery failure must not break gateway startup;
|
|
# built-in bridging stays intact.
|
|
pass
|
|
|
|
for _task_key in _aux_bridged_keys:
|
|
_task_cfg = _auxiliary_cfg.get(_task_key, {})
|
|
if not isinstance(_task_cfg, dict):
|
|
continue
|
|
_prov = str(_task_cfg.get("provider", "")).strip()
|
|
_model = str(_task_cfg.get("model", "")).strip()
|
|
_base_url = str(_task_cfg.get("base_url", "")).strip()
|
|
_api_key = str(_task_cfg.get("api_key", "")).strip()
|
|
_upper = _task_key.upper()
|
|
if _prov and _prov != "auto":
|
|
os.environ[f"AUXILIARY_{_upper}_PROVIDER"] = _prov
|
|
if _model:
|
|
os.environ[f"AUXILIARY_{_upper}_MODEL"] = _model
|
|
if _base_url:
|
|
os.environ[f"AUXILIARY_{_upper}_BASE_URL"] = _base_url
|
|
if _api_key:
|
|
os.environ[f"AUXILIARY_{_upper}_API_KEY"] = _api_key
|
|
# config.yaml is authoritative and unconditionally wins over .env; a `not in os.environ`
|
|
# guard would let stale .env entries (an old HERMES_MAX_ITERATIONS) shadow current config.
|
|
_agent_cfg = _cfg.get("agent", {})
|
|
if _agent_cfg and isinstance(_agent_cfg, dict):
|
|
if "max_turns" in _agent_cfg:
|
|
_raw_mt = _agent_cfg["max_turns"]
|
|
# Same None-guard as _bridge_max_turns_from_config: str(None)
|
|
# → "None" → resolve_turn_limit maps to unlimited, not default.
|
|
if _raw_mt is not None:
|
|
os.environ["HERMES_MAX_ITERATIONS"] = str(_raw_mt)
|
|
elif "HERMES_MAX_ITERATIONS" in os.environ:
|
|
del os.environ["HERMES_MAX_ITERATIONS"]
|
|
if "gateway_timeout" in _agent_cfg:
|
|
os.environ["HERMES_AGENT_TIMEOUT"] = str(_agent_cfg["gateway_timeout"])
|
|
if "gateway_turn_lease_timeout" in _agent_cfg:
|
|
os.environ["HERMES_TURN_LEASE_TIMEOUT"] = str(
|
|
_agent_cfg["gateway_turn_lease_timeout"]
|
|
)
|
|
if "gateway_timeout_warning" in _agent_cfg:
|
|
os.environ["HERMES_AGENT_TIMEOUT_WARNING"] = str(_agent_cfg["gateway_timeout_warning"])
|
|
if "gateway_notify_interval" in _agent_cfg:
|
|
os.environ["HERMES_AGENT_NOTIFY_INTERVAL"] = str(_agent_cfg["gateway_notify_interval"])
|
|
if "session_stall_timeout" in _agent_cfg:
|
|
os.environ["HERMES_SESSION_STALL_TIMEOUT"] = str(
|
|
_agent_cfg["session_stall_timeout"]
|
|
)
|
|
if "reconnect_attention_after" in _agent_cfg:
|
|
# Internal bridge only — config.yaml (agent.reconnect_attention_after)
|
|
# is the documented, user-facing setting.
|
|
os.environ["HERMES_RECONNECT_ATTENTION_AFTER_SECONDS"] = str(
|
|
_agent_cfg["reconnect_attention_after"]
|
|
)
|
|
if "restart_drain_timeout" in _agent_cfg:
|
|
os.environ["HERMES_RESTART_DRAIN_TIMEOUT"] = str(_agent_cfg["restart_drain_timeout"])
|
|
if "cron_drain_timeout" in _agent_cfg:
|
|
os.environ["HERMES_CRON_DRAIN_TIMEOUT"] = str(_agent_cfg["cron_drain_timeout"])
|
|
if "gateway_auto_continue_freshness" in _agent_cfg:
|
|
os.environ["HERMES_AUTO_CONTINUE_FRESHNESS"] = str(
|
|
_agent_cfg["gateway_auto_continue_freshness"]
|
|
)
|
|
if "gateway_startup_restore_drain_timeout" in _agent_cfg:
|
|
os.environ["HERMES_STARTUP_RESTORE_DRAIN_TIMEOUT"] = str(
|
|
_agent_cfg["gateway_startup_restore_drain_timeout"]
|
|
)
|
|
if "gateway_startup_warmup_timeout" in _agent_cfg:
|
|
os.environ["HERMES_STARTUP_WARMUP_TIMEOUT"] = str(
|
|
_agent_cfg["gateway_startup_warmup_timeout"]
|
|
)
|
|
# config-authoritative knobs for the session-search index; same
|
|
# bridge semantics as the agent settings above.
|
|
_sessions_cfg = _cfg.get("sessions", {})
|
|
if _sessions_cfg and isinstance(_sessions_cfg, dict):
|
|
if "cjk_fts" in _sessions_cfg:
|
|
os.environ["HERMES_CJK_FTS"] = str(_sessions_cfg["cjk_fts"])
|
|
if "search_slow_ms" in _sessions_cfg:
|
|
os.environ["HERMES_SEARCH_SLOW_MS"] = str(
|
|
_sessions_cfg["search_slow_ms"]
|
|
)
|
|
_display_cfg = _cfg.get("display", {})
|
|
if _display_cfg and isinstance(_display_cfg, dict):
|
|
if "busy_input_mode" in _display_cfg:
|
|
os.environ["HERMES_GATEWAY_BUSY_INPUT_MODE"] = str(_display_cfg["busy_input_mode"])
|
|
if "busy_text_mode" in _display_cfg:
|
|
os.environ["HERMES_GATEWAY_BUSY_TEXT_MODE"] = str(_display_cfg["busy_text_mode"])
|
|
if "busy_ack_enabled" in _display_cfg:
|
|
os.environ["HERMES_GATEWAY_BUSY_ACK_ENABLED"] = str(_display_cfg["busy_ack_enabled"])
|
|
# Documented as a service-manager override, so preserve it when already set; other
|
|
# display bridges stay config-authoritative for backwards compatibility.
|
|
if (
|
|
"busy_steer_ack_enabled" in _display_cfg
|
|
and "HERMES_GATEWAY_BUSY_STEER_ACK_ENABLED" not in os.environ
|
|
):
|
|
os.environ["HERMES_GATEWAY_BUSY_STEER_ACK_ENABLED"] = str(
|
|
_display_cfg["busy_steer_ack_enabled"]
|
|
)
|
|
# Timezone: bridge config.yaml → HERMES_TIMEZONE env var.
|
|
_tz_cfg = _cfg.get("timezone", "")
|
|
if _tz_cfg and isinstance(_tz_cfg, str):
|
|
os.environ["HERMES_TIMEZONE"] = _tz_cfg.strip()
|
|
# Security settings
|
|
_security_cfg = _cfg.get("security", {})
|
|
if isinstance(_security_cfg, dict):
|
|
_redact = _security_cfg.get("redact_secrets")
|
|
if _redact is not None:
|
|
os.environ["HERMES_REDACT_SECRETS"] = str(_redact).lower()
|
|
# Media settings (delivery allowlist, recency trust, strict mode) use the shared bridge so
|
|
# standalone entrypoints (`hermes cron run`, gateway-less ticks) apply the SAME policy.
|
|
_gateway_cfg = _cfg.get("gateway", {})
|
|
if isinstance(_gateway_cfg, dict):
|
|
from gateway.media_policy import apply_media_policy_env
|
|
|
|
apply_media_policy_env(_cfg)
|
|
_trust_recent_seconds = _gateway_cfg.get("trust_recent_files_seconds")
|
|
if _trust_recent_seconds is not None:
|
|
os.environ["HERMES_MEDIA_TRUST_RECENT_SECONDS"] = str(_trust_recent_seconds)
|
|
# Bridge gateway.platform_connect_timeout → the env var the connect path and Discord
|
|
# ready-wait read. Unlike the bridges above, it is an escape hatch: WINS if already set.
|
|
if (
|
|
"platform_connect_timeout" in _gateway_cfg
|
|
and not os.environ.get("HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT", "").strip()
|
|
):
|
|
os.environ["HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT"] = str(
|
|
_gateway_cfg["platform_connect_timeout"]
|
|
)
|
|
except Exception as _bridge_err:
|
|
# Surface the failure to stderr so operators see it even though `logger` is not yet
|
|
# initialized at module-import time (logger is defined further down this module).
|
|
print(
|
|
f" Warning: config.yaml → env bridge failed: "
|
|
f"{type(_bridge_err).__name__}: {_bridge_err}",
|
|
file=sys.stderr,
|
|
)
|
|
print(
|
|
" Gateway will fall back to .env values, which may not match "
|
|
"your current config.yaml. Run `hermes doctor` to investigate.",
|
|
file=sys.stderr,
|
|
)
|
|
|
|
# Apply IPv4 preference if configured (before any HTTP clients are created).
|
|
try:
|
|
from hermes_constants import apply_ipv4_preference
|
|
_network_cfg = (_cfg if '_cfg' in dir() else {}).get("network", {})
|
|
if isinstance(_network_cfg, dict) and _network_cfg.get("force_ipv4"):
|
|
apply_ipv4_preference(force=True)
|
|
except Exception as _bootstrap_exc:
|
|
print(f" Warning: IPv4 preference application failed: {_bootstrap_exc}", file=sys.stderr)
|
|
|
|
# Validate config structure early — log warnings so gateway operators see problems
|
|
try:
|
|
from hermes_cli.config import print_config_warnings
|
|
print_config_warnings()
|
|
except Exception as _bootstrap_exc:
|
|
print(f" Warning: config validation failed: {_bootstrap_exc}", file=sys.stderr)
|
|
|
|
# Warn if user has deprecated MESSAGING_CWD / TERMINAL_CWD in .env
|
|
try:
|
|
from hermes_cli.config import warn_deprecated_cwd_env_vars
|
|
warn_deprecated_cwd_env_vars()
|
|
except Exception as _bootstrap_exc:
|
|
print(f" Warning: deprecation check failed: {_bootstrap_exc}", file=sys.stderr)
|
|
|
|
# Gateway runs in quiet mode - suppress debug output and use cwd directly (no temp dirs)
|
|
os.environ["HERMES_QUIET"] = "1"
|
|
|
|
# HERMES_EXEC_ASK is set in start_gateway(), not at import time. Importing this module from CLI
|
|
# tools (e.g. send_message → _gateway_runner_ref) must not flip interactive CLI sessions into ask-
|
|
# mode, or Dangerous Command prompts become silent pending_approval with no Approve/Deny UI.
|
|
|
|
# Terminal cwd for messaging platforms: config.yaml terminal.cwd is canonical (bridged to
|
|
# TERMINAL_CWD above); MESSAGING_CWD is a backward-compat fallback.
|
|
from gateway.cwd_placeholder import CWD_PLACEHOLDERS, resolve_placeholder_terminal_cwd
|
|
|
|
_configured_cwd = os.environ.get("TERMINAL_CWD", "")
|
|
if not _configured_cwd or _configured_cwd in CWD_PLACEHOLDERS:
|
|
_resolved_cwd = resolve_placeholder_terminal_cwd(
|
|
configured_cwd=_configured_cwd,
|
|
terminal_backend=os.environ.get("TERMINAL_ENV", ""),
|
|
messaging_cwd=os.getenv("MESSAGING_CWD"),
|
|
docker_mount_cwd_to_workspace=os.getenv(
|
|
"TERMINAL_DOCKER_MOUNT_CWD_TO_WORKSPACE", "false"
|
|
).lower()
|
|
in {"true", "1", "yes"},
|
|
home_fallback=str(Path.home()),
|
|
)
|
|
if _resolved_cwd is None:
|
|
os.environ.pop("TERMINAL_CWD", None)
|
|
else:
|
|
os.environ["TERMINAL_CWD"] = _resolved_cwd
|
|
|
|
from gateway.config import (
|
|
ChannelOverride,
|
|
Platform,
|
|
GatewayConfig,
|
|
PlatformConfig,
|
|
_getenv,
|
|
load_gateway_config,
|
|
)
|
|
from gateway.session import (
|
|
AsyncSessionStore,
|
|
SessionStore,
|
|
SessionSource,
|
|
SessionContext,
|
|
build_session_key,
|
|
)
|
|
from gateway.delivery import (
|
|
DeliveryRouter,
|
|
resolve_delivery_transport, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.<name>)
|
|
)
|
|
from gateway.turn_lease import (
|
|
SessionTurnLeaseRegistry,
|
|
)
|
|
from gateway.session_state import (
|
|
SessionState,
|
|
legacy_dict_property,
|
|
legacy_lease_token_property,
|
|
)
|
|
from gateway.authz_mixin import GatewayAuthorizationMixin
|
|
from gateway.kanban_watchers import GatewayKanbanWatchersMixin
|
|
from gateway.slash_commands import GatewaySlashCommandsMixin
|
|
from gateway.run_voice import GatewayVoiceMixin
|
|
from gateway.run_adapters import GatewayAdapterLifecycleMixin
|
|
from gateway.run_topics import GatewayTopicThreadsMixin
|
|
from gateway.run_turn import GatewayTurnMixin
|
|
from gateway.run_shutdown import GatewayShutdownMixin
|
|
from gateway.run_busy import GatewayBusySessionMixin
|
|
from gateway.run_config_loaders import GatewayConfigLoadersMixin
|
|
from gateway.run_startup import GatewayStartupMixin
|
|
from gateway.run_watchers import GatewaySessionWatchersMixin
|
|
from gateway.run_notifications import GatewayNotificationsMixin
|
|
from gateway.run_inbound import GatewayInboundMixin
|
|
from gateway.run_goals import GatewayGoalsMixin
|
|
from gateway.run_agent_cache import GatewayAgentCacheMixin
|
|
from gateway.run_turn_runner import TurnRunner # noqa: F401 (re-exported; run.py callers + tests)
|
|
from gateway.platforms.base import (
|
|
BasePlatformAdapter,
|
|
MessageEvent,
|
|
MessageType,
|
|
_reply_anchor_for_event,
|
|
merge_pending_message_event, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.<name>)
|
|
)
|
|
from gateway.shutdown_watchdog import (
|
|
_arm_loop_floor_timer, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.<name>)
|
|
start_loop_liveness_watchdog, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.<name>)
|
|
)
|
|
from gateway.restart import (
|
|
DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT,
|
|
DEFAULT_GATEWAY_POST_INTERRUPT_GRACE_TIMEOUT, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.<name>)
|
|
DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT,
|
|
DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT,
|
|
DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT,
|
|
)
|
|
|
|
|
|
from gateway.whatsapp_identity import (
|
|
canonical_whatsapp_identifier as _canonical_whatsapp_identifier, # noqa: F401
|
|
)
|
|
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
# Ceiling for the shutdown quiesce of the gateway-owned thread pool. Drain has
|
|
# already waited for the agents, so what is left here is short blocking work
|
|
# (a transcript append, a routing save); anything slower is a stuck worker we
|
|
# must not wait on, and the caller clamps this to the watchdog leash anyway.
|
|
_EXECUTOR_QUIESCE_TIMEOUT = 2.0
|
|
|
|
|
|
_OWN_POLICY_OPEN_ENV = {
|
|
Platform.WECOM: ("WECOM_DM_POLICY", "WECOM_GROUP_POLICY", "WECOM_ALLOW_ALL_USERS"),
|
|
Platform.WEIXIN: ("WEIXIN_DM_POLICY", "WEIXIN_GROUP_POLICY", "WEIXIN_ALLOW_ALL_USERS"),
|
|
Platform.YUANBAO: ("YUANBAO_DM_POLICY", "YUANBAO_GROUP_POLICY", "YUANBAO_ALLOW_ALL_USERS"),
|
|
Platform.QQBOT: (None, None, "QQ_ALLOW_ALL_USERS"),
|
|
Platform.WHATSAPP: ("WHATSAPP_DM_POLICY", "WHATSAPP_GROUP_POLICY", "WHATSAPP_ALLOW_ALL_USERS"),
|
|
}
|
|
|
|
|
|
def _own_policy_open_startup_violation(config) -> Optional[str]:
|
|
"""Return a startup-abort reason when open policy lacks allow-all opt-in."""
|
|
for platform, platform_config in getattr(config, "platforms", {}).items():
|
|
if not getattr(platform_config, "enabled", False):
|
|
continue
|
|
open_env = _OWN_POLICY_OPEN_ENV.get(platform)
|
|
if not open_env:
|
|
continue
|
|
dm_env, group_env, allow_all_env = open_env
|
|
extra = getattr(platform_config, "extra", None) or {}
|
|
dm_policy = str(
|
|
extra.get("dm_policy")
|
|
or (_getenv(dm_env, "pairing") if dm_env else "pairing")
|
|
).strip().lower()
|
|
group_policy = str(
|
|
extra.get("group_policy")
|
|
or (_getenv(group_env, "pairing") if group_env else "pairing")
|
|
).strip().lower()
|
|
if dm_policy != "open" and group_policy != "open":
|
|
continue
|
|
gateway_allow_all = _getenv(
|
|
"GATEWAY_ALLOW_ALL_USERS", ""
|
|
).lower() in {"true", "1", "yes"}
|
|
platform_opted_in = gateway_allow_all or (
|
|
allow_all_env
|
|
and _getenv(allow_all_env, "").lower() in {"true", "1", "yes"}
|
|
)
|
|
if platform_opted_in:
|
|
continue
|
|
return f"{platform.value}: open policy without allow-all opt-in"
|
|
return None
|
|
|
|
|
|
# Sentinel placed into _running_agents *before* any await when a session starts processing, so a
|
|
# second message can't slip past the "already running" guard before the agent actually exists.
|
|
_AGENT_PENDING_SENTINEL = object()
|
|
|
|
# Conversation-scoped per-session state registry (legacy contract). The state itself lives in
|
|
# ``SessionState.conversation`` and boundaries clear it via ``ConversationState.clear()`` (new
|
|
# fields are picked up automatically). Retained for (a) plain-dict stores not yet folded into
|
|
# SessionState (``_pending_model_notes``), popped per-key by _clear_conversation_scope, and (b) the
|
|
# public test contract. NOT in this list (different lifecycles): _running_agents/_running_agents_ts/
|
|
# _active_session_leases/_busy_ack_ts/_turn_lease_tokens (turn-scoped, owned by
|
|
# _release_running_agent_state and the dispatch finally); _session_run_generation (monotonic —
|
|
# clearing breaks stale-run detection); _agent_cache (own eviction path _evict_cached_agent);
|
|
# approval/slash-confirm state (cleared via _clear_session_boundary_security_state).
|
|
_CONVERSATION_SCOPED_STATE: tuple = (
|
|
"_session_model_overrides",
|
|
"_pending_one_turn_model_restores",
|
|
"_session_reasoning_overrides",
|
|
"_session_service_tier_overrides",
|
|
"_pending_model_notes",
|
|
"_last_resolved_model",
|
|
"_queued_events",
|
|
# Stall-watchdog "already notified" latch (#72016). Cleared on /new so a
|
|
# fresh conversation can warn again if it later stalls with pending inbound.
|
|
"_session_stall_notified",
|
|
# Sidecar notes staged but never consumed (turn aborted before run_sync) must not leak into a
|
|
# future conversation's first user message — session keys are source-derived and REUSED.
|
|
"_pending_turn_sidecar_notes",
|
|
)
|
|
|
|
from gateway.run_common import _UNSET # noqa: F401 (def-time sentinel shared with run_* mixins)
|
|
|
|
|
|
def _resolve_runtime_agent_kwargs() -> dict:
|
|
"""Resolve provider credentials for gateway-created AIAgent instances.
|
|
|
|
``resolve_runtime_provider()`` falls through to env vars for legacy compatibility, but the
|
|
gateway never consults env vars for behavioral config — config.yaml is authoritative.
|
|
"""
|
|
from hermes_cli.runtime_provider import (
|
|
resolve_runtime_provider,
|
|
format_runtime_provider_error,
|
|
_get_model_config,
|
|
)
|
|
from hermes_cli.auth import AuthError, is_rate_limited_auth_error
|
|
|
|
try:
|
|
runtime = resolve_runtime_provider()
|
|
except AuthError as auth_exc:
|
|
# Distinguish a rate-limit/quota cap (credentials fine, re-auth can't help) from a real auth
|
|
# failure (expired/revoked token): both use the fallback chain; the log must not mislabel.
|
|
if is_rate_limited_auth_error(auth_exc):
|
|
logger.warning("Primary provider rate-limited (429): %s — trying fallback", auth_exc)
|
|
else:
|
|
logger.warning("Primary provider auth failed: %s — trying fallback", auth_exc)
|
|
fb_config = _try_resolve_fallback_provider()
|
|
if fb_config is not None:
|
|
return fb_config
|
|
raise RuntimeError(format_runtime_provider_error(auth_exc)) from auth_exc
|
|
except Exception as exc:
|
|
raise RuntimeError(format_runtime_provider_error(exc)) from exc
|
|
|
|
model_cfg = _get_model_config()
|
|
max_tokens = None
|
|
_env_mt = os.environ.get("HERMES_MAX_TOKENS")
|
|
if _env_mt:
|
|
try:
|
|
max_tokens = int(_env_mt)
|
|
except (ValueError, TypeError):
|
|
max_tokens = None
|
|
elif isinstance(model_cfg, dict):
|
|
mt = model_cfg.get("max_tokens")
|
|
if isinstance(mt, int):
|
|
max_tokens = mt
|
|
# Per-provider output cap (custom_providers max_output_tokens) applies only when the documented
|
|
# global model.max_tokens is unset, so the global key always wins.
|
|
if max_tokens is None:
|
|
_runtime_mot = runtime.get("max_output_tokens")
|
|
if isinstance(_runtime_mot, int) and _runtime_mot > 0:
|
|
max_tokens = _runtime_mot
|
|
|
|
capabilities = runtime.get("capabilities")
|
|
capabilities = (
|
|
{
|
|
key: value
|
|
for key, value in capabilities.items()
|
|
if isinstance(key, str) and isinstance(value, bool)
|
|
}
|
|
if isinstance(capabilities, dict)
|
|
else {}
|
|
)
|
|
|
|
return {
|
|
"api_key": runtime.get("api_key"),
|
|
"base_url": runtime.get("base_url"),
|
|
"provider": runtime.get("provider"),
|
|
"requested_provider": runtime.get("requested_provider"),
|
|
"api_mode": runtime.get("api_mode"),
|
|
"command": runtime.get("command"),
|
|
"args": list(runtime.get("args") or []),
|
|
"credential_pool": runtime.get("credential_pool"),
|
|
"request_overrides": dict(runtime.get("request_overrides") or {}),
|
|
"max_tokens": max_tokens,
|
|
# Per-provider request_overrides (e.g. custom_providers ``extra_body`` with
|
|
# ``chat_template_kwargs``) from resolve_runtime_provider() must reach the per-turn route,
|
|
# else the provider's configured request body never reaches the model on the gateway path.
|
|
"request_overrides": runtime.get("request_overrides"),
|
|
"capabilities": capabilities,
|
|
}
|
|
|
|
|
|
@dataclasses.dataclass(frozen=True)
|
|
class _GatewayModelContext:
|
|
"""Effective gateway model route and context-window resolution."""
|
|
|
|
model: str
|
|
provider: str
|
|
base_url: str
|
|
context_length: int
|
|
context_source: str
|
|
|
|
|
|
def _resolve_gateway_model_context(model: Optional[str] = None) -> _GatewayModelContext:
|
|
"""Resolve the configured gateway route and its effective context window.
|
|
|
|
Shared authority for status/session banners and slash commands. Call it off the event loop:
|
|
credential resolution and model metadata may block.
|
|
"""
|
|
from agent.model_metadata import DEFAULT_FALLBACK_CONTEXT, get_model_context_length
|
|
|
|
resolved_model = model or _resolve_gateway_model()
|
|
config_context_length = None
|
|
provider = None
|
|
base_url = None
|
|
api_key = None
|
|
custom_providers = None
|
|
configured_model = None
|
|
configured_provider = None
|
|
configured_base_url = None
|
|
|
|
try:
|
|
data = _load_gateway_config()
|
|
if data:
|
|
model_cfg = data.get("model", {})
|
|
if isinstance(model_cfg, dict):
|
|
configured_model = model_cfg.get("default") or model_cfg.get("model")
|
|
raw_ctx = model_cfg.get("context_length")
|
|
if raw_ctx is not None:
|
|
with suppress(TypeError, ValueError):
|
|
config_context_length = int(raw_ctx)
|
|
provider = model_cfg.get("provider") or None
|
|
base_url = model_cfg.get("base_url") or None
|
|
configured_provider = provider
|
|
configured_base_url = base_url
|
|
try:
|
|
from hermes_cli.config import get_compatible_custom_providers
|
|
|
|
custom_providers = get_compatible_custom_providers(data)
|
|
except Exception:
|
|
custom_providers = data.get("custom_providers")
|
|
except Exception:
|
|
pass
|
|
|
|
try:
|
|
runtime = _resolve_runtime_agent_kwargs()
|
|
provider = runtime.get("provider") or provider
|
|
base_url = runtime.get("base_url") or base_url
|
|
api_key = runtime.get("api_key")
|
|
except Exception:
|
|
pass
|
|
|
|
if config_context_length is not None:
|
|
try:
|
|
from hermes_cli.route_identity import should_clear_context_pin
|
|
|
|
if should_clear_context_pin(
|
|
configured_model,
|
|
resolved_model,
|
|
configured_base_url,
|
|
base_url,
|
|
configured_provider,
|
|
provider,
|
|
):
|
|
config_context_length = None
|
|
except Exception:
|
|
config_context_length = None
|
|
|
|
if config_context_length is None and custom_providers and base_url:
|
|
try:
|
|
from hermes_cli.config import get_custom_provider_context_length
|
|
|
|
custom_ctx = get_custom_provider_context_length(
|
|
model=resolved_model,
|
|
base_url=base_url,
|
|
custom_providers=custom_providers,
|
|
)
|
|
if custom_ctx:
|
|
config_context_length = custom_ctx
|
|
except Exception:
|
|
pass
|
|
|
|
context_length = get_model_context_length(
|
|
resolved_model,
|
|
base_url=base_url or "",
|
|
api_key=api_key or "",
|
|
config_context_length=config_context_length,
|
|
provider=provider or "",
|
|
custom_providers=custom_providers,
|
|
)
|
|
if config_context_length is not None:
|
|
context_source = "config"
|
|
elif context_length == DEFAULT_FALLBACK_CONTEXT:
|
|
context_source = "default"
|
|
else:
|
|
context_source = "detected"
|
|
|
|
return _GatewayModelContext(
|
|
model=resolved_model,
|
|
provider=provider or "",
|
|
base_url=base_url or "",
|
|
context_length=context_length,
|
|
context_source=context_source,
|
|
)
|
|
|
|
|
|
def _resolve_runtime_agent_kwargs_for_provider(provider: str) -> dict:
|
|
"""Resolve runtime credentials for a specific provider (e.g. from channel override)."""
|
|
from hermes_cli.runtime_provider import (
|
|
resolve_runtime_provider,
|
|
format_runtime_provider_error,
|
|
)
|
|
try:
|
|
runtime = resolve_runtime_provider(requested=provider)
|
|
except Exception as exc:
|
|
raise RuntimeError(format_runtime_provider_error(exc)) from exc
|
|
return {
|
|
"api_key": runtime.get("api_key"),
|
|
"base_url": runtime.get("base_url"),
|
|
"provider": runtime.get("provider"),
|
|
"requested_provider": runtime.get("requested_provider"),
|
|
"api_mode": runtime.get("api_mode"),
|
|
"command": runtime.get("command"),
|
|
"args": list(runtime.get("args") or []),
|
|
"credential_pool": runtime.get("credential_pool"),
|
|
"request_overrides": dict(runtime.get("request_overrides") or {}),
|
|
"capabilities": dict(runtime.get("capabilities") or {}),
|
|
"max_tokens": runtime.get("max_output_tokens"),
|
|
}
|
|
|
|
|
|
def _deep_merge_request_overrides(base: Optional[dict], override: Optional[dict]) -> dict:
|
|
"""Merge request_overrides dicts, deep-merging nested dictionaries."""
|
|
from hermes_cli.config import _deep_merge
|
|
|
|
base_dict = dict(base or {})
|
|
override_dict = dict(override or {})
|
|
if not base_dict:
|
|
return override_dict
|
|
if not override_dict:
|
|
return base_dict
|
|
return _deep_merge(base_dict, override_dict)
|
|
|
|
|
|
def _credential_pool_for_provider(provider: Optional[str]):
|
|
"""Return the live credential pool for a provider id (e.g. ``custom:hyper``)."""
|
|
if not provider or not str(provider).strip():
|
|
return None
|
|
try:
|
|
return _resolve_runtime_agent_kwargs_for_provider(str(provider).strip()).get(
|
|
"credential_pool"
|
|
)
|
|
except Exception:
|
|
logger.debug(
|
|
"Failed to resolve credential pool for provider=%s",
|
|
provider,
|
|
exc_info=True,
|
|
)
|
|
return None
|
|
|
|
|
|
def _try_resolve_fallback_provider() -> dict | None:
|
|
"""Attempt to resolve credentials from the fallback_model/fallback_providers config."""
|
|
from hermes_cli.runtime_provider import resolve_runtime_provider
|
|
try:
|
|
# Canonical loader so managed overlay, ${VAR} expansion and root-model normalization
|
|
# reach the fallback chain (a raw read misses administrator-pinned fallback_providers).
|
|
cfg = _load_gateway_runtime_config()
|
|
fb_list = get_fallback_chain(cfg)
|
|
if not fb_list:
|
|
return None
|
|
for entry in fb_list:
|
|
try:
|
|
from hermes_cli.fallback_config import resolve_entry_api_key
|
|
|
|
runtime = resolve_runtime_provider(
|
|
requested=entry.get("provider"),
|
|
explicit_base_url=entry.get("base_url"),
|
|
explicit_api_key=resolve_entry_api_key(entry),
|
|
)
|
|
# Log the literal config `provider`, not the resolved runtime category: an Ollama
|
|
# fallback resolves via the OpenAI-compatible path and would log as "openrouter".
|
|
logger.info(
|
|
"Fallback provider resolved: %s model=%s",
|
|
entry.get("provider") or runtime.get("provider"),
|
|
entry.get("model"),
|
|
)
|
|
return {
|
|
"api_key": runtime.get("api_key"),
|
|
"base_url": runtime.get("base_url"),
|
|
"provider": runtime.get("provider"),
|
|
"requested_provider": runtime.get("requested_provider"),
|
|
"api_mode": runtime.get("api_mode"),
|
|
"command": runtime.get("command"),
|
|
"args": list(runtime.get("args") or []),
|
|
"credential_pool": runtime.get("credential_pool"),
|
|
"request_overrides": dict(runtime.get("request_overrides") or {}),
|
|
"model": entry.get("model"),
|
|
"request_overrides": runtime.get("request_overrides"),
|
|
}
|
|
except Exception as fb_exc:
|
|
logger.debug("Fallback entry %s failed: %s", entry.get("provider"), fb_exc)
|
|
continue
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def _event_media_type_at(event, index: int) -> str:
|
|
"""Per-attachment MIME at *index*; "" when the adapter set only a message-level type."""
|
|
media_types = getattr(event, "media_types", None) or []
|
|
return media_types[index] if index < len(media_types) else ""
|
|
|
|
|
|
def _event_media_is_image(event, index: int) -> bool:
|
|
"""True if the attachment at *index* is an image.
|
|
|
|
Trust the per-attachment MIME; fall back to message-level ``PHOTO`` only when unknown, else a
|
|
document uploaded alongside an image is base64'd as vision and the provider 400s.
|
|
"""
|
|
mtype = _event_media_type_at(event, index)
|
|
if mtype:
|
|
return mtype.startswith("image/")
|
|
return getattr(event, "message_type", None) == MessageType.PHOTO
|
|
|
|
|
|
def _event_media_is_audio(event, index: int) -> bool:
|
|
"""True if the attachment at *index* is audio (per-attachment MIME first)."""
|
|
mtype = _event_media_type_at(event, index)
|
|
if mtype:
|
|
return mtype.startswith("audio/")
|
|
return getattr(event, "message_type", None) in {MessageType.VOICE, MessageType.AUDIO}
|
|
|
|
|
|
def _event_media_is_stt_input(event, index: int) -> bool:
|
|
"""True when an audio attachment should enter the automatic STT pipeline."""
|
|
message_type = getattr(event, "message_type", None)
|
|
if message_type in {MessageType.AUDIO, MessageType.DOCUMENT}:
|
|
return False
|
|
return (
|
|
message_type == MessageType.VOICE
|
|
or _event_media_type_at(event, index).startswith("audio/")
|
|
)
|
|
|
|
|
|
def _event_media_is_video(event, index: int) -> bool:
|
|
"""True if the attachment at *index* is video (per-attachment MIME first)."""
|
|
mtype = _event_media_type_at(event, index)
|
|
if mtype:
|
|
return mtype.startswith("video/")
|
|
return getattr(event, "message_type", None) == MessageType.VIDEO
|
|
|
|
|
|
def _build_media_placeholder(event) -> str:
|
|
"""Text placeholder for media-only events (later replaced by vision enrichment).
|
|
|
|
Media queued during active processing is dequeued via .text only, so a caption-less event
|
|
would otherwise be lost.
|
|
"""
|
|
parts = []
|
|
media_urls = getattr(event, "media_urls", None) or []
|
|
for i, url in enumerate(media_urls):
|
|
if _event_media_is_image(event, i):
|
|
parts.append(f"[User sent an image: {url}]")
|
|
elif _event_media_is_audio(event, i):
|
|
parts.append(f"[User sent audio: {url}]")
|
|
elif _event_media_is_video(event, i):
|
|
parts.append(f"[User sent a video: {url}]")
|
|
else:
|
|
parts.append(f"[User sent a file: {url}]")
|
|
return "\n".join(parts)
|
|
|
|
|
|
def _build_document_context_note(
|
|
display_name: str,
|
|
agent_path: str,
|
|
mtype: str,
|
|
*,
|
|
content_inlined: bool = True,
|
|
) -> str:
|
|
"""Context note prepended to a user turn when they attach a document.
|
|
|
|
``content_inlined=False`` = adapter cached the file without injecting content, so tell the agent
|
|
to read it. Binary docs (PDF, DOCX, …) must say *extract* the text; "ask the user" made it punt.
|
|
"""
|
|
if mtype.startswith("text/") and content_inlined:
|
|
return (
|
|
f"[The user sent a text document: '{display_name}'. "
|
|
f"Its content has been included below. "
|
|
f"The file is also saved at: {agent_path}]"
|
|
)
|
|
if mtype.startswith("text/"):
|
|
return (
|
|
f"[The user sent a text document: '{display_name}'. It is saved at: {agent_path}. "
|
|
f"Its content is not inlined here. Read the cached file yourself before answering "
|
|
f"when the user's request involves its contents.]"
|
|
)
|
|
return (
|
|
f"[The user sent a document: '{display_name}'. It is saved at: {agent_path}. "
|
|
f"Its text is not inlined here (it's a binary format such as PDF or DOCX). "
|
|
f"To read it, extract the document's text yourself — for example with the "
|
|
f"terminal tool or the ocr-and-documents skill — before answering, instead "
|
|
f"of asking the user to paste the contents.]"
|
|
)
|
|
|
|
|
|
def _format_duration(seconds: float) -> str:
|
|
total = int(round(seconds))
|
|
if total < 0:
|
|
total = 0
|
|
hours, rem = divmod(total, 3600)
|
|
minutes, secs = divmod(rem, 60)
|
|
if hours:
|
|
return f"{hours}:{minutes:02d}:{secs:02d}"
|
|
return f"{minutes}:{secs:02d}"
|
|
|
|
|
|
async def _probe_audio_duration(path: str) -> Optional[str]:
|
|
"""Best-effort duration probe. Returns formatted MM:SS / HH:MM:SS, or None on failure."""
|
|
ext = os.path.splitext(path)[1].lower()
|
|
|
|
if ext == ".wav":
|
|
try:
|
|
def _wav_duration() -> float:
|
|
import wave
|
|
with wave.open(path, "rb") as wf:
|
|
frames = wf.getnframes()
|
|
rate = wf.getframerate() or 1
|
|
return frames / float(rate)
|
|
secs = await asyncio.to_thread(_wav_duration)
|
|
return _format_duration(secs)
|
|
except Exception:
|
|
pass
|
|
|
|
if ext in (".ogg", ".opus", ".oga"):
|
|
try:
|
|
def _ogg_duration() -> float:
|
|
from mutagen.oggopus import OggOpus
|
|
return float(OggOpus(path).info.length)
|
|
secs = await asyncio.to_thread(_ogg_duration)
|
|
return _format_duration(secs)
|
|
except Exception:
|
|
pass
|
|
|
|
try:
|
|
proc = await asyncio.create_subprocess_exec(
|
|
"ffprobe", "-v", "error", "-show_entries", "format=duration",
|
|
"-of", "default=noprint_wrappers=1:nokey=1", path,
|
|
stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE,
|
|
)
|
|
stdout, _ = await asyncio.wait_for(proc.communicate(), timeout=5.0)
|
|
if proc.returncode == 0:
|
|
return _format_duration(float(stdout.decode().strip()))
|
|
except Exception:
|
|
pass
|
|
|
|
return None
|
|
|
|
|
|
def _dequeue_pending_event(adapter, session_key: str) -> MessageEvent | None:
|
|
"""Consume and return the full pending event for a session.
|
|
|
|
Queued follow-ups keep their media metadata so they re-enter the normal image/STT/document
|
|
preprocessing path instead of collapsing to a placeholder string.
|
|
"""
|
|
return adapter.get_pending_message(session_key)
|
|
|
|
|
|
_INTERRUPT_REASON_STOP = "Stop requested"
|
|
_INTERRUPT_REASON_RESET = "Session reset requested"
|
|
_INTERRUPT_REASON_TIMEOUT = "Execution timed out (inactivity)"
|
|
_INTERRUPT_REASON_SSE_DISCONNECT = "SSE client disconnected"
|
|
_INTERRUPT_REASON_GATEWAY_SHUTDOWN = "Gateway shutting down"
|
|
_INTERRUPT_REASON_GATEWAY_RESTART = "Gateway restarting"
|
|
|
|
|
|
def _reap_gateway_turn_processes(
|
|
task_id: str,
|
|
process_baseline,
|
|
*,
|
|
source: str,
|
|
is_still_current: Optional[Callable[[], bool]] = None,
|
|
) -> int:
|
|
"""Reap only background processes created by one abandoned turn.
|
|
|
|
``task_id`` is session-scoped, so a *replacement* turn can spawn its own process mid-reap;
|
|
``is_still_current`` (closure over the captured run_generation) lets the caller bail instead of
|
|
killing it. That turn snapshots its own baseline, so nothing stays unreaped.
|
|
"""
|
|
if not task_id:
|
|
# ProcessSession.task_id defaults to "" for sessionless callers; a blank id would match
|
|
# (and kill) every unrelated empty-task process. Nothing session-scoped to reap.
|
|
return 0
|
|
if is_still_current is not None:
|
|
try:
|
|
if not is_still_current():
|
|
logger.debug(
|
|
"Skipping reap for turn %s (%s): a newer turn already "
|
|
"claimed this session; it owns its own baseline.",
|
|
task_id,
|
|
source,
|
|
)
|
|
return 0
|
|
except Exception:
|
|
logger.debug(
|
|
"is_still_current check failed for turn %s (%s); reaping anyway",
|
|
task_id,
|
|
source,
|
|
exc_info=True,
|
|
)
|
|
|
|
from tools.process_registry import process_registry
|
|
|
|
try:
|
|
killed = process_registry.kill_started_since(
|
|
task_id,
|
|
process_baseline,
|
|
source=source,
|
|
)
|
|
except Exception:
|
|
# Runs on a detached daemon thread (fire-and-forget from interrupt and timeout paths); an
|
|
# uncaught exception would only reach threading.excepthook. Swallow and log normally.
|
|
logger.warning(
|
|
"Failed to reap background processes for turn %s (%s)",
|
|
task_id,
|
|
source,
|
|
exc_info=True,
|
|
)
|
|
return 0
|
|
if killed:
|
|
logger.warning(
|
|
"Reaped %d background process(es) created by abandoned turn %s (%s)",
|
|
killed,
|
|
task_id,
|
|
source,
|
|
)
|
|
return killed
|
|
|
|
|
|
_TURN_STACK_DUMP_FRAME_MARKERS = (
|
|
"run_conversation",
|
|
"run_sync",
|
|
"_run_sync_with_timeout_lifecycle",
|
|
"finalize_turn",
|
|
"end_turn",
|
|
"run_in_session",
|
|
)
|
|
|
|
|
|
def _dump_wedged_turn_stacks(task_id: str) -> None:
|
|
"""Log the stack of every thread that looks like turn work, at reap time.
|
|
|
|
The reaper's hard interrupt frees the wedged worker before a profiler can attach, so dump BEFORE
|
|
interrupting. Best-effort, bounded (turn-machinery threads only, capped output), never raises.
|
|
"""
|
|
try:
|
|
frames = sys._current_frames()
|
|
names = {t.ident: t.name for t in threading.enumerate()}
|
|
dumped = 0
|
|
for ident, frame in frames.items():
|
|
if ident == threading.get_ident():
|
|
continue # the reaper itself
|
|
stack = traceback.format_stack(frame)
|
|
joined = "".join(stack)
|
|
if not any(marker in joined for marker in _TURN_STACK_DUMP_FRAME_MARKERS):
|
|
continue
|
|
dumped += 1
|
|
if dumped > 8:
|
|
logger.error(
|
|
"Wedged-turn stack dump for task %s truncated: more than "
|
|
"8 candidate threads",
|
|
task_id,
|
|
)
|
|
break
|
|
logger.error(
|
|
"Wedged-turn stack dump (task=%s thread=%s ident=%s):\n%s",
|
|
task_id,
|
|
names.get(ident, "?"),
|
|
ident,
|
|
"".join(stack[-25:]),
|
|
)
|
|
if dumped == 0:
|
|
logger.error(
|
|
"Wedged-turn stack dump for task %s: no thread with "
|
|
"turn-machinery frames found (worker may have already exited)",
|
|
task_id,
|
|
)
|
|
except Exception:
|
|
logger.debug("Wedged-turn stack dump failed", exc_info=True)
|
|
|
|
|
|
def _abandon_timed_out_gateway_turn(
|
|
*,
|
|
agent_holder,
|
|
task_id: str,
|
|
process_baseline,
|
|
worker_done: threading.Event,
|
|
timeout_fired: threading.Event,
|
|
cleanup_lock: threading.Lock,
|
|
is_still_current: Optional[Callable[[], bool]] = None,
|
|
) -> bool:
|
|
"""Interrupt one timed-out turn and reap only processes it created."""
|
|
with cleanup_lock:
|
|
if worker_done.is_set() or timeout_fired.is_set():
|
|
return False
|
|
timeout_fired.set()
|
|
|
|
# Capture the wedged worker's stack BEFORE interrupting: the interrupt frees the blocked
|
|
# frame, destroying the only evidence of where the turn was stuck.
|
|
_dump_wedged_turn_stacks(task_id)
|
|
|
|
agent = agent_holder[0] if agent_holder else None
|
|
if agent is not None:
|
|
try:
|
|
request_hard_interrupt(agent, _INTERRUPT_REASON_TIMEOUT)
|
|
except Exception:
|
|
logger.debug("Timed-out agent interrupt failed", exc_info=True)
|
|
|
|
try:
|
|
_reap_gateway_turn_processes(
|
|
task_id,
|
|
process_baseline,
|
|
source="gateway_turn_timeout",
|
|
is_still_current=is_still_current,
|
|
)
|
|
except Exception:
|
|
logger.warning(
|
|
"Failed to reap background processes for timed-out turn %s",
|
|
task_id,
|
|
exc_info=True,
|
|
)
|
|
return True
|
|
|
|
|
|
def _watch_gateway_turn_inactivity(
|
|
*,
|
|
agent_holder,
|
|
task_id: str,
|
|
process_baseline,
|
|
timeout: float,
|
|
worker_done: threading.Event,
|
|
timeout_fired: threading.Event,
|
|
cleanup_lock: threading.Lock,
|
|
poll_interval: float = 5.0,
|
|
is_still_current: Optional[Callable[[], bool]] = None,
|
|
) -> None:
|
|
"""Thread watchdog that remains runnable when gateway asyncio is starved."""
|
|
while not worker_done.wait(max(0.01, poll_interval)):
|
|
agent = agent_holder[0] if agent_holder else None
|
|
if agent is None or not hasattr(agent, "get_activity_summary"):
|
|
continue
|
|
try:
|
|
idle_seconds = float(
|
|
agent.get_activity_summary().get("seconds_since_activity", 0.0)
|
|
)
|
|
except Exception:
|
|
continue
|
|
if idle_seconds < timeout:
|
|
continue
|
|
_abandon_timed_out_gateway_turn(
|
|
agent_holder=agent_holder,
|
|
task_id=task_id,
|
|
process_baseline=process_baseline,
|
|
worker_done=worker_done,
|
|
timeout_fired=timeout_fired,
|
|
cleanup_lock=cleanup_lock,
|
|
is_still_current=is_still_current,
|
|
)
|
|
return
|
|
|
|
|
|
_CONTROL_INTERRUPT_MESSAGES = frozenset(
|
|
{
|
|
_INTERRUPT_REASON_STOP.lower(),
|
|
_INTERRUPT_REASON_RESET.lower(),
|
|
_INTERRUPT_REASON_TIMEOUT.lower(),
|
|
_INTERRUPT_REASON_SSE_DISCONNECT.lower(),
|
|
_INTERRUPT_REASON_GATEWAY_SHUTDOWN.lower(),
|
|
_INTERRUPT_REASON_GATEWAY_RESTART.lower(),
|
|
}
|
|
)
|
|
|
|
|
|
def _is_control_interrupt_message(message: Optional[str]) -> bool:
|
|
"""Return True when an interrupt message is internal control flow."""
|
|
if not message:
|
|
return False
|
|
normalized = " ".join(str(message).strip().split()).lower()
|
|
return normalized in _CONTROL_INTERRUPT_MESSAGES
|
|
|
|
|
|
def _strip_response_attachments_for_direct_send(response: str, adapter) -> str:
|
|
"""Return the visible text portion of a response before direct send().
|
|
|
|
Queued follow-up resends replay only explicit ``MEDIA:`` attachments; bare local paths and image
|
|
URLs stay visible because the post-stream uploader ignores them. No broad ``MEDIA:`` regex after
|
|
``extract_media()`` — it deliberately preserves protected code spans and unvalidated tags.
|
|
"""
|
|
_, cleaned = adapter.extract_media(response)
|
|
cleaned = cleaned.replace("[[audio_as_voice]]", "").strip()
|
|
cleaned = cleaned.replace("[[as_document]]", "").strip()
|
|
return cleaned.strip()
|
|
|
|
|
|
def _skill_slug_from_frontmatter(skill_md: Path) -> tuple[str | None, str | None]:
|
|
"""Derive the /command slug and declared frontmatter name from a SKILL.md.
|
|
|
|
Matches ``scan_skill_commands``: the slug comes from frontmatter ``name:``, NOT the directory
|
|
name. Returns ``(slug, declared_name)`` or ``(None, None)`` if unreadable or lacking ``name:``.
|
|
"""
|
|
try:
|
|
content = skill_md.read_text(encoding="utf-8", errors="replace")
|
|
except Exception:
|
|
return None, None
|
|
content = content.lstrip("\ufeff") # tolerate UTF-8 BOM (Windows editors)
|
|
if not content.startswith("---"):
|
|
return None, None
|
|
end = content.find("\n---", 3)
|
|
if end < 0:
|
|
return None, None
|
|
declared_name: str | None = None
|
|
for line in content[3:end].splitlines():
|
|
line = line.strip()
|
|
if line.startswith("name:"):
|
|
raw = line.split(":", 1)[1].strip()
|
|
# Strip YAML quote wrappers if present
|
|
if len(raw) >= 2 and raw[0] == raw[-1] and raw[0] in {'"', "'"}:
|
|
raw = raw[1:-1]
|
|
declared_name = raw.strip()
|
|
break
|
|
if not declared_name:
|
|
return None, None
|
|
slug = declared_name.lower().replace(" ", "-").replace("_", "-")
|
|
# Mirror _SKILL_INVALID_CHARS and _SKILL_MULTI_HYPHEN from skill_commands
|
|
import re as _re
|
|
slug = _re.sub(r"[^a-z0-9-]", "", slug)
|
|
slug = _re.sub(r"-{2,}", "-", slug).strip("-")
|
|
if not slug:
|
|
return None, declared_name
|
|
return slug, declared_name
|
|
|
|
|
|
def _check_unavailable_skill(command_name: str) -> str | None:
|
|
"""Match a command to a known-but-inactive skill.
|
|
|
|
Returns a hint if the skill exists but is disabled or optional-install only; else None.
|
|
"""
|
|
# Normalize: command uses hyphens, skill names may use hyphens or underscores
|
|
normalized = command_name.lower().replace("_", "-")
|
|
try:
|
|
from tools.skills_tool import _get_disabled_skill_names
|
|
from agent.skill_utils import get_all_skills_dirs, is_excluded_skill_path
|
|
disabled = _get_disabled_skill_names()
|
|
|
|
# Check disabled skills across all dirs (local + external)
|
|
for skills_dir in get_all_skills_dirs():
|
|
if not skills_dir.exists():
|
|
continue
|
|
for skill_md in skills_dir.rglob("SKILL.md"):
|
|
if is_excluded_skill_path(skill_md):
|
|
continue
|
|
slug, declared_name = _skill_slug_from_frontmatter(skill_md)
|
|
if not slug or not declared_name:
|
|
continue
|
|
# disabled is keyed by the declared frontmatter name (what
|
|
# skills.disabled / skills.platform_disabled store).
|
|
if slug == normalized and declared_name in disabled:
|
|
return (
|
|
f"The **{command_name}** skill is installed but disabled.\n"
|
|
f"Enable it with: `hermes skills config`"
|
|
)
|
|
|
|
# Check optional skills (shipped with repo but not installed)
|
|
from hermes_constants import get_optional_skills_dir
|
|
repo_root = Path(__file__).resolve().parent.parent
|
|
optional_dir = get_optional_skills_dir(repo_root / "optional-skills")
|
|
if optional_dir.exists():
|
|
for skill_md in optional_dir.rglob("SKILL.md"):
|
|
if is_excluded_skill_path(skill_md):
|
|
continue
|
|
slug, _declared = _skill_slug_from_frontmatter(skill_md)
|
|
if not slug:
|
|
continue
|
|
if slug == normalized:
|
|
# Build install path: official/<category>/<name>
|
|
rel = skill_md.parent.relative_to(optional_dir)
|
|
parts = list(rel.parts)
|
|
install_path = f"official/{'/'.join(parts)}"
|
|
return (
|
|
f"The **{command_name}** skill is available but not installed.\n"
|
|
f"Install it with: `hermes skills install {install_path}`"
|
|
)
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def _platform_config_key(platform: "Platform") -> str:
|
|
"""Map a Platform enum to its config.yaml key (LOCAL→"cli", rest→enum value)."""
|
|
return "cli" if platform == Platform.LOCAL else platform.value
|
|
|
|
|
|
def _teams_pipeline_plugin_enabled() -> bool:
|
|
"""Return True when the standalone Teams pipeline plugin is enabled."""
|
|
config = _load_gateway_config()
|
|
enabled = cfg_get(config, "plugins", "enabled", default=[])
|
|
if not isinstance(enabled, list):
|
|
return False
|
|
return "teams_pipeline" in enabled or "teams-pipeline" in enabled
|
|
|
|
|
|
def _gateway_config_home() -> Path:
|
|
"""Return the Hermes home that gateway config reads should use."""
|
|
override = get_hermes_home_override()
|
|
if override:
|
|
return Path(override)
|
|
return _hermes_home
|
|
|
|
|
|
def _load_gateway_config(config_path: "Path | None" = None) -> dict:
|
|
"""Load and parse a gateway config.yaml, returning {} on any error (fail-open).
|
|
|
|
Defaults to the active gateway home (``_hermes_home`` monkeypatches apply); multiplexed callers
|
|
may pass a profile path. Managed scope is overlaid here because neither read_raw_config nor
|
|
yaml.safe_load carries the managed merge, and pinned values must be honored.
|
|
"""
|
|
if config_path is None:
|
|
config_path = _gateway_config_home() / 'config.yaml'
|
|
raw: dict = {}
|
|
used_canonical = False
|
|
try:
|
|
from hermes_cli.config import get_config_path, read_raw_config
|
|
# Fast path: reuse the shared cache when _hermes_home agrees with the canonical config
|
|
# location; otherwise fall through to a direct read (monkeypatched _hermes_home in tests).
|
|
if config_path == get_config_path():
|
|
raw = read_raw_config()
|
|
used_canonical = True
|
|
except Exception:
|
|
pass
|
|
|
|
if not used_canonical:
|
|
try:
|
|
if config_path.exists():
|
|
import yaml
|
|
with open(config_path, 'r', encoding='utf-8') as f:
|
|
raw = yaml.safe_load(f) or {}
|
|
except Exception:
|
|
logger.debug("Could not load gateway config from %s", config_path)
|
|
raw = {}
|
|
|
|
# read_raw_config() returns raw YAML WITHOUT the managed merge (that lives in load_config), so
|
|
# the managed overlay is required on both paths for the gateway to honor pinned values.
|
|
try:
|
|
from hermes_cli import managed_scope
|
|
raw = managed_scope.apply_managed_overlay(raw if isinstance(raw, dict) else {})
|
|
except Exception:
|
|
pass
|
|
if not isinstance(raw, dict):
|
|
return {}
|
|
# Canonicalize model-id aliases (model.name / model.model → model.default) and migrate stale
|
|
# root-level provider/base_url into the model section. The gateway bypasses load_config(), so
|
|
# without this replay ``model: {name: <id>}`` resolves to an empty model. Fail-open.
|
|
try:
|
|
from hermes_cli.config import _normalize_root_model_keys
|
|
raw = _normalize_root_model_keys(raw)
|
|
except Exception:
|
|
pass
|
|
return raw
|
|
|
|
|
|
def _checkpoint_agent_kwargs(config: dict | None) -> dict:
|
|
"""Translate gateway checkpoint config into ``AIAgent`` constructor args.
|
|
|
|
Gateway bypasses ``load_config()``, so defaults are here; legacy ``checkpoints: true`` works.
|
|
"""
|
|
cp_cfg = config.get("checkpoints", {}) if isinstance(config, dict) else {}
|
|
if isinstance(cp_cfg, bool):
|
|
cp_cfg = {"enabled": cp_cfg}
|
|
elif not isinstance(cp_cfg, dict):
|
|
cp_cfg = {}
|
|
|
|
from hermes_cli.config import DEFAULT_CONFIG
|
|
defaults = DEFAULT_CONFIG["checkpoints"]
|
|
return {
|
|
"checkpoints_enabled": cp_cfg.get("enabled", defaults["enabled"]),
|
|
"checkpoint_max_snapshots": cp_cfg.get(
|
|
"max_snapshots", defaults["max_snapshots"],
|
|
),
|
|
"checkpoint_max_total_size_mb": cp_cfg.get(
|
|
"max_total_size_mb", defaults["max_total_size_mb"],
|
|
),
|
|
"checkpoint_max_file_size_mb": cp_cfg.get(
|
|
"max_file_size_mb", defaults["max_file_size_mb"],
|
|
),
|
|
}
|
|
|
|
|
|
def _load_gateway_runtime_config() -> dict:
|
|
"""Load gateway config for runtime reads, expanding supported ``${VAR}`` refs.
|
|
|
|
Built on ``_load_gateway_config()``. Expansion failures are deliberately NOT swallowed —
|
|
returning the unexpanded dict would mask the very bug this helper fixes.
|
|
"""
|
|
cfg = _load_gateway_config()
|
|
if not isinstance(cfg, dict) or not cfg:
|
|
return {}
|
|
from hermes_cli.config import _expand_env_vars
|
|
|
|
expanded = _expand_env_vars(cfg)
|
|
return expanded if isinstance(expanded, dict) else {}
|
|
|
|
|
|
def _resolve_gateway_model(config: dict | None = None) -> str:
|
|
"""Read model from config.yaml (single source of truth).
|
|
|
|
Otherwise temporary AIAgent instances (e.g. /compress) use the hardcoded default, which fails
|
|
when the active provider is openai-codex.
|
|
"""
|
|
cfg = config if config is not None else _load_gateway_config()
|
|
model_cfg = cfg.get("model", {})
|
|
if isinstance(model_cfg, str):
|
|
return model_cfg
|
|
elif isinstance(model_cfg, dict):
|
|
return model_cfg.get("default") or model_cfg.get("model") or ""
|
|
return ""
|
|
|
|
|
|
def _channel_override_lookup_keys(
|
|
chat_id: str,
|
|
*,
|
|
thread_id: Optional[str] = None,
|
|
parent_id: Optional[str] = None,
|
|
) -> list[str]:
|
|
"""Ordered, de-duplicated ``channel_overrides`` lookup keys.
|
|
|
|
Matches ``resolve_channel_prompt``: exact thread/channel id first, then parent channel/forum id
|
|
(Discord threads inherit parent overrides).
|
|
"""
|
|
keys: list[str] = []
|
|
seen: set[str] = set()
|
|
for key in (chat_id, thread_id, parent_id):
|
|
if not key:
|
|
continue
|
|
sk = str(key)
|
|
if sk in seen:
|
|
continue
|
|
seen.add(sk)
|
|
keys.append(sk)
|
|
return keys
|
|
|
|
|
|
def _get_channel_override(
|
|
config: GatewayConfig,
|
|
platform: Platform,
|
|
chat_id: str,
|
|
*,
|
|
thread_id: Optional[str] = None,
|
|
parent_id: Optional[str] = None,
|
|
) -> Optional[ChannelOverride]:
|
|
"""Per-channel override for this platform/chat_id, or None.
|
|
|
|
Looks up ``chat_id``, then ``thread_id``, then ``parent_id`` (child channels inherit parent).
|
|
"""
|
|
platforms = getattr(config, "platforms", None)
|
|
if not platforms:
|
|
return None
|
|
platform_config = platforms.get(platform)
|
|
if not platform_config or not platform_config.channel_overrides:
|
|
return None
|
|
overrides = platform_config.channel_overrides
|
|
for key in _channel_override_lookup_keys(
|
|
chat_id, thread_id=thread_id, parent_id=parent_id
|
|
):
|
|
ov = overrides.get(key)
|
|
if ov is not None:
|
|
return ov
|
|
return None
|
|
|
|
|
|
def _resolve_hermes_bin() -> Optional[list[str]]:
|
|
"""Resolve the Hermes update command as argv parts, or ``None``.
|
|
|
|
Tries ``shutil.which("hermes")``, then ``sys.executable -m hermes_cli.main`` (no shim on PATH).
|
|
"""
|
|
import shutil
|
|
|
|
hermes_bin = shutil.which("hermes")
|
|
if hermes_bin:
|
|
return [hermes_bin]
|
|
|
|
try:
|
|
import importlib.util
|
|
|
|
if importlib.util.find_spec("hermes_cli") is not None:
|
|
return [sys.executable, "-m", "hermes_cli.main"]
|
|
except Exception:
|
|
pass
|
|
|
|
return None
|
|
|
|
|
|
def _parse_session_key(session_key: str) -> "dict | None":
|
|
"""Parse a session key (``agent:main:{platform}:{chat_type}:{chat_id}[:{extra}...]``).
|
|
|
|
For group/channel sessions the suffix may be a user_id (per-user isolation), not a thread_id,
|
|
so ``thread_id`` is left out to avoid mis-routing.
|
|
"""
|
|
parts = session_key.split(":")
|
|
if len(parts) >= 5 and parts[0] == "agent" and parts[1] == "main":
|
|
result = {
|
|
"platform": parts[2],
|
|
"chat_type": parts[3],
|
|
"chat_id": parts[4],
|
|
}
|
|
if len(parts) > 5 and parts[3] in {"dm", "thread"}:
|
|
result["thread_id"] = parts[5]
|
|
return result
|
|
return None
|
|
|
|
|
|
def _shorten_command_for_display(command: str, limit: int = 80) -> str:
|
|
"""Collapse a shell command onto one line and cap its length for display."""
|
|
one_line = " ".join((command or "").split())
|
|
if len(one_line) > limit:
|
|
one_line = one_line[: limit - 1] + "…"
|
|
return one_line
|
|
|
|
|
|
def _format_concise_process_notification(
|
|
session_id: str,
|
|
command: str,
|
|
exit_code,
|
|
output: str,
|
|
duration_seconds=None,
|
|
) -> str:
|
|
"""One-line completion message for the ``concise`` display mode.
|
|
|
|
Success is one status line; failure appends a short output tail (full output via process(log)).
|
|
"""
|
|
ok = exit_code in {0, None}
|
|
icon = "✅" if ok else "❌"
|
|
verb = "finished" if ok else f"failed (exit {exit_code})"
|
|
parts = [f"{icon} Background task {verb}"]
|
|
short_cmd = _shorten_command_for_display(command)
|
|
if short_cmd:
|
|
parts.append(f"— `{short_cmd}`")
|
|
if isinstance(duration_seconds, (int, float)) and duration_seconds >= 0:
|
|
secs = int(duration_seconds)
|
|
if secs >= 3600:
|
|
dur = f"{secs // 3600}h {(secs % 3600) // 60}m"
|
|
elif secs >= 60:
|
|
dur = f"{secs // 60}m {secs % 60}s"
|
|
else:
|
|
dur = f"{secs}s"
|
|
parts.append(f"({dur})")
|
|
text = " ".join(parts)
|
|
if not ok and output:
|
|
tail_lines = [ln for ln in output.strip().splitlines() if ln.strip()][-5:]
|
|
tail = "\n".join(tail_lines)
|
|
if len(tail) > 500:
|
|
tail = tail[-500:]
|
|
if tail:
|
|
text += f"\n```\n{tail}\n```"
|
|
return text
|
|
|
|
|
|
def _format_gateway_process_notification(evt: dict) -> "str | None":
|
|
"""Format a watch pattern event from completion_queue into a [IMPORTANT:] message."""
|
|
evt_type = evt.get("type", "completion")
|
|
_sid = evt.get("session_id", "unknown")
|
|
_cmd = evt.get("command", "unknown")
|
|
|
|
if evt_type == "watch_disabled":
|
|
return f"[IMPORTANT: {evt.get('message', '')}]"
|
|
|
|
# Overflow events carry their human-readable summary in `message`, like watch_disabled
|
|
# (shared formatter in tools/process_registry.py).
|
|
if evt_type in ("watch_overflow_tripped", "watch_overflow_released"):
|
|
return f"[IMPORTANT: {evt.get('message', '')}]"
|
|
|
|
if evt_type == "watch_match":
|
|
_pat = evt.get("pattern", "?")
|
|
_out = evt.get("output", "")
|
|
_sup = evt.get("suppressed", 0)
|
|
text = (
|
|
f"[IMPORTANT: Background process {_sid} matched "
|
|
f"watch pattern \"{_pat}\".\n"
|
|
f"Command: {_cmd}\n"
|
|
f"Matched output:\n{_out}"
|
|
)
|
|
if _sup:
|
|
text += f"\n({_sup} earlier matches were suppressed by rate limit)"
|
|
text += "]"
|
|
return text
|
|
|
|
if evt_type == "async_delegation":
|
|
# Reuse the shared rich formatter (self-contained task-source block).
|
|
from tools.process_registry import format_process_notification
|
|
return format_process_notification(evt)
|
|
|
|
return None
|
|
|
|
|
|
def _drain_gateway_watch_events(completion_queue) -> "list[dict]":
|
|
"""Drain gateway-owned watch events without spinning on requeued events.
|
|
|
|
Process completions belong to per-process watchers, async delegation completions to
|
|
``_async_delegation_watcher``; requeueing them inside ``while not queue.empty()`` never
|
|
terminates, so detach the batch first and requeue foreign events afterwards.
|
|
"""
|
|
watch_events: list[dict] = []
|
|
requeue: list[dict] = []
|
|
while not completion_queue.empty():
|
|
try:
|
|
evt = completion_queue.get_nowait()
|
|
except Exception:
|
|
break
|
|
evt_type = evt.get("type", "completion")
|
|
if evt_type in {
|
|
"watch_match",
|
|
"watch_disabled",
|
|
"watch_overflow_tripped",
|
|
"watch_overflow_released",
|
|
}:
|
|
watch_events.append(evt)
|
|
elif evt_type == "async_delegation":
|
|
requeue.append(evt)
|
|
# else: process completion events are handled by the watcher task
|
|
for evt in requeue:
|
|
completion_queue.put(evt)
|
|
return watch_events
|
|
|
|
|
|
# Weak ref to the active GatewayRunner (set in GatewayRunner.__init__), used by tools such as
|
|
# send_message that must route through a live adapter for plugin platforms.
|
|
import weakref as _weakref
|
|
_gateway_runner_ref: _weakref.ref = lambda: None
|
|
|
|
|
|
def _normalize_empty_agent_response(
|
|
agent_result: dict,
|
|
response: str,
|
|
*,
|
|
history_len: int = 0,
|
|
) -> str:
|
|
"""Normalize empty/None agent responses into user-facing messages.
|
|
|
|
Covers ``failed`` plus the case where the agent did work (api_calls > 0) but returned no text,
|
|
and surfaces a retry hint when it never ran (api_calls == 0, not interrupted/failed): the
|
|
post-/stop silent-drop where a stale generation token returns an empty result.
|
|
"""
|
|
if response:
|
|
return response
|
|
|
|
if agent_result.get("failed"):
|
|
# None-safe: the result dict is built with ``'error': holder.get('error')`` and can carry an
|
|
# EXPLICIT None, bypassing dict.get's default and rendering "The request failed: None".
|
|
error_detail = agent_result.get("error") or "unknown error"
|
|
error_str = str(error_detail).lower()
|
|
# Session-persistence failures get a dedicated recovery message: suggesting /reset would
|
|
# destroy the user's context without fixing the storage problem (locks, full disk).
|
|
failure_reason = str(agent_result.get("failure_reason") or "")
|
|
if failure_reason.startswith("session_persistence_failed") or (
|
|
"session storage" in error_str
|
|
):
|
|
if failure_reason.endswith(":disk") or "disk" in error_str:
|
|
return (
|
|
"⚠️ Session storage was temporarily unavailable, so this "
|
|
"turn was stopped to protect your conversation history. "
|
|
"Please check available disk space, then send your "
|
|
"message again."
|
|
)
|
|
return (
|
|
"⚠️ Session storage was temporarily unavailable, so this "
|
|
"turn was stopped to protect your conversation history. "
|
|
"Your message should already be saved — please send it "
|
|
"again in a moment."
|
|
)
|
|
is_context_failure = any(
|
|
p in error_str
|
|
for p in ("context", "token", "too large", "too long", "exceed", "payload")
|
|
) or ("400" in error_str and history_len > 50)
|
|
if is_context_failure:
|
|
return (
|
|
"⚠️ Session too large for the model's context window.\n"
|
|
"Use /compact to compress the conversation, or "
|
|
"/reset to start fresh."
|
|
)
|
|
return (
|
|
f"The request failed: {str(error_detail)[:300]}\n"
|
|
"Try again or use /reset to start a fresh session."
|
|
)
|
|
|
|
api_calls = int(agent_result.get("api_calls", 0) or 0)
|
|
if agent_result.get("interrupted"):
|
|
# Interrupted with api_calls > 0 = deliberately stopped/steered; silence is intentional and
|
|
# queued messages come via the recursive drain in _run_agent. With ZERO api_calls the
|
|
# message was never processed (stale /stop interrupt flag), so surface it.
|
|
if api_calls == 0:
|
|
return (
|
|
"⚠️ Your message was interrupted before processing started "
|
|
"(likely by a recent /stop). Please send it again."
|
|
)
|
|
return response
|
|
if api_calls > 0:
|
|
if _is_gateway_hidden_reasoning_incomplete_turn(agent_result):
|
|
return ""
|
|
if agent_result.get("partial"):
|
|
err = agent_result.get("error", "processing incomplete")
|
|
return f"⚠️ Processing stopped: {str(err)[:200]}. Try again."
|
|
return (
|
|
"⚠️ Processing completed but no response was generated. "
|
|
"This may be a transient error — try sending your message again."
|
|
)
|
|
|
|
# api_calls == 0, not failed, not interrupted: the agent never ran (post-/stop generation race).
|
|
# Without this the gateway silently drops the turn and the user sees no reply.
|
|
if (
|
|
api_calls == 0
|
|
and not agent_result.get("interrupted")
|
|
and not agent_result.get("failed")
|
|
and not agent_result.get("partial")
|
|
):
|
|
return (
|
|
"⚠️ Your message wasn't processed (the previous turn was still "
|
|
"being cleaned up). Please send it again."
|
|
)
|
|
|
|
return response
|
|
|
|
|
|
def _is_gateway_hidden_reasoning_incomplete_turn(agent_result: dict) -> bool:
|
|
"""Detect retry-exhausted turns with hidden reasoning but no visible answer.
|
|
|
|
The loop returns the retry-exhaustion sentinel as BOTH ``final_response`` and ``error``, so a
|
|
non-empty ``final_response`` proves nothing. Hidden only when the sentinel is present and
|
|
``final_response`` is empty or echoes it; any other text is a real answer.
|
|
"""
|
|
if not isinstance(agent_result, dict):
|
|
return False
|
|
if agent_result.get("failed") or agent_result.get("interrupted"):
|
|
return False
|
|
if not agent_result.get("partial"):
|
|
return False
|
|
error_text = str(agent_result.get("error", "") or "").strip()
|
|
if "remained incomplete after" not in error_text.lower():
|
|
return False
|
|
final_response = str(agent_result.get("final_response") or "").strip()
|
|
return not final_response or final_response == error_text
|
|
|
|
|
|
def _should_clear_resume_pending_after_turn(agent_result: dict) -> bool:
|
|
"""True only when a gateway turn really completed successfully.
|
|
|
|
Restart recovery uses ``resume_pending`` as a durable marker; a soft interrupt can look like a
|
|
normal result with an empty final response, and clearing the marker then loses the signal.
|
|
"""
|
|
if not isinstance(agent_result, dict):
|
|
return False
|
|
if agent_result.get("interrupted"):
|
|
return False
|
|
if agent_result.get("failed") or agent_result.get("partial") or agent_result.get("error"):
|
|
return False
|
|
return agent_result.get("completed") is not False
|
|
|
|
|
|
def _preserve_queued_followup_history_offset(
|
|
current_result: dict,
|
|
followup_result: dict,
|
|
) -> dict:
|
|
"""Carry the outer history offset through queued follow-up drains.
|
|
|
|
Each recursive ``_run_agent()`` advances ``history_offset``; uncorrected, the outer persistence
|
|
step sees only the *last* queued turn as "new" and drops earlier ones.
|
|
"""
|
|
if not isinstance(followup_result, dict):
|
|
return followup_result
|
|
if not isinstance(current_result, dict):
|
|
return followup_result
|
|
|
|
current_offset = current_result.get("history_offset")
|
|
followup_offset = followup_result.get("history_offset")
|
|
if not isinstance(current_offset, int):
|
|
return followup_result
|
|
if isinstance(followup_offset, int) and followup_offset <= current_offset:
|
|
return followup_result
|
|
|
|
merged = dict(followup_result)
|
|
merged["history_offset"] = current_offset
|
|
return merged
|
|
|
|
|
|
async def _dispose_unused_adapter(adapter: "BasePlatformAdapter | None") -> None:
|
|
"""Best-effort dispose for an adapter that never made it onto ``self.adapters``.
|
|
|
|
A failed connect leaves the adapter uninstalled, so nothing else calls ``disconnect()``;
|
|
resources opened in ``__init__`` (e.g. SQLite fds) would leak until GC (not prompt for
|
|
asyncio-bound objects) and exhaust the fd ulimit over a long retry loop. ``adapter`` may be
|
|
``None`` (half-constructed / ``_create_adapter`` returned None).
|
|
"""
|
|
if adapter is None:
|
|
return
|
|
try:
|
|
await adapter.disconnect()
|
|
except Exception:
|
|
# Half-constructed adapters (e.g. APIServerAdapter that crashed in aiohttp setup) can raise
|
|
# from disconnect(); that must not abort the watcher loop. ``asyncio.CancelledError`` is a
|
|
# BaseException, so cancellation is not swallowed; dispose failures are best-effort.
|
|
logger.debug(
|
|
"Adapter dispose raised on unowned adapter %r",
|
|
getattr(adapter, "name", type(adapter).__name__),
|
|
exc_info=True,
|
|
)
|
|
|
|
|
|
# Max seconds between platform reconnect retries (primary watcher and
|
|
# secondary-profile reconnects share this policy — tune in one place).
|
|
_RECONNECT_BACKOFF_CAP = 300
|
|
|
|
# Seconds a platform may sit continuously in the reconnect queue before it is flagged
|
|
# NEEDS_ATTENTION. Retrying never stops (transient outages must self-heal); this only makes a
|
|
# permanently-failing loop loud. 0 disables.
|
|
_RECONNECT_ATTENTION_AFTER_SECONDS = _float_env(
|
|
"HERMES_RECONNECT_ATTENTION_AFTER_SECONDS", 7200
|
|
)
|
|
|
|
|
|
def _reconnect_backoff(attempt: int) -> int:
|
|
"""Exponential reconnect backoff: 30s, 60s, 120s, ... capped at 5 min."""
|
|
return min(30 * (2 ** (attempt - 1)), _RECONNECT_BACKOFF_CAP)
|
|
|
|
|
|
def _reconnect_needs_attention(info: dict, now: float) -> bool:
|
|
"""True when a reconnect-queue entry has waited long enough for NEEDS_ATTENTION.
|
|
|
|
``queued_at`` is re-stamped on each (re)entry, so only *continuous* failure escalates.
|
|
"""
|
|
if _RECONNECT_ATTENTION_AFTER_SECONDS <= 0:
|
|
return False # escalation disabled
|
|
queued_at = info.get("queued_at")
|
|
if queued_at is None:
|
|
info["queued_at"] = now
|
|
return False
|
|
return (now - queued_at) >= _RECONNECT_ATTENTION_AFTER_SECONDS
|
|
|
|
|
|
# Sentinel for "no explicit session DB pinned on this runner", so ``_session_db`` can distinguish
|
|
# "resolve from the active profile scope" from a deliberate ``runner._session_db = None`` (disables
|
|
# DB-backed commands, as many test suites do). Mirrors ``gateway.session._DB_UNPINNED``.
|
|
_SESSION_DB_UNPINNED = object()
|
|
|
|
|
|
# Agent-facing sidecar note per auto-reset reason (default: idle).
|
|
_AUTO_RESET_CONTEXT_NOTES = {
|
|
"suspended": "[System note: The user's previous session was stopped and suspended. This is a fresh conversation with no prior context.]",
|
|
"daily": "[System note: The user's session was automatically reset by the daily schedule. This is a fresh conversation with no prior context.]",
|
|
"resume_pending_expired": "[System note: The previous gateway session could not be recovered after a restart (API recovery timed out). This is a fresh conversation — use /resume to restore history if needed.]",
|
|
"idle": "[System note: The user's previous session expired due to inactivity. This is a fresh conversation with no prior context.]",
|
|
}
|
|
|
|
|
|
def _auto_reset_reason_text(reset_reason: str, policy) -> str:
|
|
"""Human-readable cause for the user-facing auto-reset notice."""
|
|
if reset_reason == "suspended":
|
|
return "previous session was stopped or interrupted"
|
|
if reset_reason == "resume_pending_expired":
|
|
return "gateway restart recovery timed out"
|
|
if reset_reason == "daily":
|
|
return f"daily schedule at {policy.at_hour}:00"
|
|
hours = policy.idle_minutes // 60
|
|
mins = policy.idle_minutes % 60
|
|
duration = f"{hours}h" if not mins else f"{hours}h {mins}m" if hours else f"{mins}m"
|
|
return f"inactive for {duration}"
|
|
|
|
|
|
def _write_runtime_status_quiet(**fields: Any) -> None:
|
|
"""Best-effort ``gateway_state.json`` write; status persistence must never abort the caller."""
|
|
try:
|
|
from gateway.status import write_runtime_status
|
|
|
|
write_runtime_status(**fields)
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _command_origin_for_source(source: Any) -> Optional[dict]:
|
|
"""Delivery origin for a shared CLI/gateway command so its job replies to this chat/thread."""
|
|
try:
|
|
platform = getattr(source.platform, "value", None) or str(getattr(source, "platform", "") or "")
|
|
chat_id = getattr(source, "chat_id", None)
|
|
if platform and chat_id:
|
|
return {
|
|
"platform": platform,
|
|
"chat_id": str(chat_id),
|
|
"chat_name": getattr(source, "chat_name", None),
|
|
"thread_id": getattr(source, "thread_id", None),
|
|
}
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def _builtin_adapter_import(module: str, adapter_name: str, requirement: str):
|
|
"""Lazy-import ``(adapter_cls, requirements_ok)`` from ``gateway.platforms.<module>``."""
|
|
import importlib
|
|
|
|
mod = importlib.import_module(f"gateway.platforms.{module}")
|
|
return getattr(mod, adapter_name), getattr(mod, requirement)
|
|
|
|
|
|
# Built-in (non-plugin) adapters: platform -> (module, adapter class, requirements probe
|
|
# name, warning when the probe fails). Signal additionally validates its config below.
|
|
_BUILTIN_ADAPTERS: dict[Platform, tuple[str, str, str, str]] = {
|
|
Platform.WHATSAPP_CLOUD: ("whatsapp_cloud", "WhatsAppCloudAdapter", "check_whatsapp_cloud_requirements",
|
|
"WhatsApp Cloud: aiohttp/httpx missing — reinstall hermes-agent"),
|
|
Platform.SIGNAL: ("signal", "SignalAdapter", "check_signal_requirements",
|
|
"Signal: runtime requirements not met"),
|
|
Platform.WEIXIN: ("weixin", "WeixinAdapter", "check_weixin_requirements",
|
|
"Weixin: aiohttp/cryptography not installed"),
|
|
Platform.API_SERVER: ("api_server", "APIServerAdapter", "check_api_server_requirements",
|
|
"API Server: aiohttp not installed"),
|
|
Platform.WEBHOOK: ("webhook", "WebhookAdapter", "check_webhook_requirements",
|
|
"Webhook: aiohttp not installed"),
|
|
Platform.MSGRAPH_WEBHOOK: ("msgraph_webhook", "MSGraphWebhookAdapter", "check_msgraph_webhook_requirements",
|
|
"MSGraph webhook: aiohttp not installed"),
|
|
Platform.BLUEBUBBLES: ("bluebubbles", "BlueBubblesAdapter", "check_bluebubbles_requirements",
|
|
"BlueBubbles: aiohttp/httpx missing or BLUEBUBBLES_SERVER_URL/BLUEBUBBLES_PASSWORD not configured"),
|
|
Platform.QQBOT: ("qqbot", "QQAdapter", "check_qq_requirements",
|
|
"QQBot: aiohttp/httpx missing or QQ_APP_ID/QQ_CLIENT_SECRET not configured"),
|
|
Platform.YUANBAO: ("yuanbao", "YuanbaoAdapter", "WEBSOCKETS_AVAILABLE",
|
|
"Yuanbao: websockets not installed. Run: pip install websockets"),
|
|
}
|
|
|
|
|
|
def _instantiate_builtin_adapter(platform: Platform, config: Any) -> Optional[BasePlatformAdapter]:
|
|
"""Instantiate a core (non-plugin) adapter, or None when its requirements are unmet/unknown."""
|
|
spec = _BUILTIN_ADAPTERS.get(platform)
|
|
if spec is None:
|
|
return None
|
|
module, adapter_name, requirement, warning = spec
|
|
adapter_cls, requirements_ok = _builtin_adapter_import(module, adapter_name, requirement)
|
|
if not (requirements_ok() if callable(requirements_ok) else requirements_ok):
|
|
logger.warning(warning)
|
|
return None
|
|
if platform == Platform.SIGNAL:
|
|
from gateway.platforms.signal import validate_signal_config
|
|
|
|
if not validate_signal_config(config):
|
|
logger.warning("Signal: SIGNAL_HTTP_URL or SIGNAL_ACCOUNT not configured")
|
|
return None
|
|
return adapter_cls(config)
|
|
|
|
|
|
class GatewayRunner(
|
|
GatewayAuthorizationMixin,
|
|
GatewayKanbanWatchersMixin,
|
|
GatewaySlashCommandsMixin,
|
|
GatewayVoiceMixin,
|
|
GatewayAdapterLifecycleMixin,
|
|
GatewayTopicThreadsMixin,
|
|
GatewayTurnMixin,
|
|
GatewayShutdownMixin,
|
|
GatewayBusySessionMixin,
|
|
GatewayConfigLoadersMixin,
|
|
GatewayStartupMixin,
|
|
GatewaySessionWatchersMixin,
|
|
GatewayNotificationsMixin,
|
|
GatewayInboundMixin,
|
|
GatewayGoalsMixin,
|
|
GatewayAgentCacheMixin,
|
|
):
|
|
"""Main gateway controller: manages adapter lifecycles, routes messages to/from the agent."""
|
|
|
|
# Class-level defaults so partial construction in tests doesn't
|
|
# blow up on attribute access.
|
|
_busy_input_mode: str = "interrupt"
|
|
_busy_text_mode: str = "interrupt"
|
|
_restart_drain_timeout: float = DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT
|
|
_restart_after_turn_timeout: float = DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT
|
|
_cron_drain_timeout: float = DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT
|
|
_signal_interrupt_grace_timeout: float = (
|
|
DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT
|
|
)
|
|
_exit_code: Optional[int] = None
|
|
_draining: bool = False
|
|
_external_drain_active: bool = False
|
|
_restart_requested: bool = False
|
|
_restart_task_started: bool = False
|
|
_restart_detached: bool = False
|
|
_restart_via_service: bool = False
|
|
_detached_restart_helper_started: bool = False
|
|
_restart_command_source: Optional[SessionSource] = None
|
|
_stop_task: Optional[asyncio.Task] = None
|
|
_restart_task: Optional[asyncio.Task] = None
|
|
_profile_failed_platforms: Optional[Dict[str, Dict[Platform, asyncio.Task]]] = None
|
|
_systemd_watchdog: Optional[Any] = None
|
|
_startup_restore_in_progress: bool = False
|
|
_startup_warmup_task: Optional[asyncio.Task] = None
|
|
|
|
# ------------------------------------------------------------------
|
|
# Legacy per-session dict adapters: all per-session state lives in ``self._sessions``
|
|
# (Dict[str, SessionState]); these properties expose the old dict attrs as LIVE MutableMapping
|
|
# views so ``runner._running_agents`` etc. keep working. New code: ``self._session_state(key)``.
|
|
# ------------------------------------------------------------------
|
|
_running_agents = legacy_dict_property("_running_agents")
|
|
_running_agents_ts = legacy_dict_property("_running_agents_ts")
|
|
_active_session_leases = legacy_dict_property("_active_session_leases")
|
|
_busy_ack_ts = legacy_dict_property("_busy_ack_ts")
|
|
_turn_lease_tokens = legacy_lease_token_property()
|
|
_session_run_generation = legacy_dict_property("_session_run_generation")
|
|
_session_model_overrides = legacy_dict_property("_session_model_overrides")
|
|
_pending_one_turn_model_restores = legacy_dict_property(
|
|
"_pending_one_turn_model_restores"
|
|
)
|
|
_session_reasoning_overrides = legacy_dict_property("_session_reasoning_overrides")
|
|
_session_service_tier_overrides = legacy_dict_property(
|
|
"_session_service_tier_overrides"
|
|
)
|
|
_last_resolved_model = legacy_dict_property("_last_resolved_model")
|
|
_queued_events = legacy_dict_property("_queued_events")
|
|
_pending_turn_sidecar_notes = legacy_dict_property("_pending_turn_sidecar_notes")
|
|
_pending_messages = legacy_dict_property("_pending_messages")
|
|
_pending_native_image_paths_by_session = legacy_dict_property(
|
|
"_pending_native_image_paths_by_session"
|
|
)
|
|
_session_ephemeral_pin = legacy_dict_property("_session_ephemeral_pin")
|
|
_session_vc_last = legacy_dict_property("_session_vc_last")
|
|
_pending_approvals = legacy_dict_property("_pending_approvals")
|
|
_update_prompt_pending = legacy_dict_property("_update_prompt_pending")
|
|
|
|
# -- SessionState accessors -----------------------------------------
|
|
def _sessions_map(self) -> Dict[str, "SessionState"]:
|
|
"""The per-session state map; lazily created so bare test runners
|
|
built via ``object.__new__`` work without ``__init__``."""
|
|
sessions = self.__dict__.get("_sessions")
|
|
if sessions is None:
|
|
sessions = {}
|
|
self.__dict__["_sessions"] = sessions
|
|
return sessions
|
|
|
|
def _session_state(self, session_key: str) -> "SessionState":
|
|
"""Get-or-create the :class:`SessionState` for ``session_key``."""
|
|
sessions = self._sessions_map()
|
|
state = sessions.get(session_key)
|
|
if state is None:
|
|
state = SessionState()
|
|
sessions[session_key] = state
|
|
return state
|
|
|
|
def _peek_session_state(self, session_key: str) -> Optional["SessionState"]:
|
|
"""Return the SessionState for ``session_key`` without creating one."""
|
|
sessions = self.__dict__.get("_sessions")
|
|
if not sessions:
|
|
return None
|
|
return sessions.get(session_key)
|
|
|
|
def _is_session_running(self, session_key: str) -> bool:
|
|
"""True when the session holds a running-turn slot (agent or sentinel)."""
|
|
state = self._peek_session_state(session_key)
|
|
return state is not None and state.turn.agent is not None
|
|
|
|
def _running_agent_items(self) -> List[tuple]:
|
|
"""(session_key, agent) pairs for sessions with a running turn
|
|
(including pending sentinels), matching the old ``_running_agents``
|
|
dict contents."""
|
|
return [
|
|
(key, state.turn.agent)
|
|
for key, state in self._sessions_map().items()
|
|
if state.turn.agent is not None
|
|
]
|
|
# Loop-liveness heartbeat / watchdog handles. Class-level defaults so partial construction in
|
|
# tests doesn't blow up on access; real values are set in __init__ / start() / stop().
|
|
_loop_heartbeat_task: Optional["asyncio.Task"] = None
|
|
_loop_floor_timer_handle: Optional[Any] = None
|
|
_loop_liveness_watchdog: Optional[Any] = None
|
|
_gateway_started_at: float = 0.0
|
|
_shutdown_watchdog_done: Optional["threading.Event"] = None
|
|
_platform_lock_takeover_on_start: bool = False
|
|
_reconnect_watcher_task: Optional["asyncio.Task"] = None
|
|
|
|
def __init__(self, config: Optional[GatewayConfig] = None):
|
|
global _gateway_runner_ref
|
|
# With multiplex_profiles on, load under the default profile secret scope so bot tokens in its
|
|
# .env resolve as secondary profiles' do; explicit config= injection (tests) is left untouched.
|
|
self.config = config if config is not None else load_gateway_config_for_runner()
|
|
# Mark the process as a profile multiplexer when configured. This flips
|
|
# agent.secret_scope.get_secret() to fail-closed on any unscoped credential read, so a
|
|
# missed migration crashes loudly instead of leaking a cross-profile value.
|
|
try:
|
|
from agent.secret_scope import set_multiplex_active
|
|
set_multiplex_active(bool(getattr(self.config, "multiplex_profiles", False)))
|
|
except Exception:
|
|
logger.debug("could not set multiplex-active flag", exc_info=True)
|
|
self.adapters: Dict[Platform, BasePlatformAdapter] = {}
|
|
# Non-None means SessionDB init failed — the gateway broadcasts a one-time warning to the home
|
|
# channel(s) after connecting so the user learns persistence is broken before /resume fails.
|
|
self._session_db_init_error: Optional[str] = None
|
|
# Multi-profile multiplexing: adapters for NON-default profiles live here, keyed by profile
|
|
# name then Platform. self.adapters stays the default/active profile's map so existing
|
|
# self.adapters[...] sites are untouched when multiplexing is off (this dict is then empty).
|
|
self._profile_adapters: Dict[str, Dict[Platform, BasePlatformAdapter]] = {}
|
|
self._warn_if_docker_media_delivery_is_risky()
|
|
_gateway_runner_ref = _weakref.ref(self)
|
|
|
|
self._init_runtime_settings()
|
|
self._init_session_store()
|
|
self._init_lifecycle_state()
|
|
self._init_runtime_caches()
|
|
self._init_startup_checks()
|
|
self._init_session_db()
|
|
self._init_registries_and_clocks()
|
|
|
|
def _init_runtime_settings(self) -> None:
|
|
"""Load ephemeral per-call config (prefill, reasoning, busy modes, timeouts, routing)."""
|
|
# Load ephemeral config from config.yaml / env vars.
|
|
# Both are injected at API-call time only and never persisted.
|
|
self._prefill_messages = self._load_prefill_messages()
|
|
self._reasoning_config = self._load_reasoning_config()
|
|
self._service_tier = self._load_service_tier()
|
|
self._show_reasoning = self._load_show_reasoning()
|
|
self._busy_input_mode = self._load_busy_input_mode()
|
|
self._busy_text_mode = self._load_busy_text_mode()
|
|
# Secondary-profile busy modes, snapshotted at multiplex startup; busy-message handlers consult
|
|
# them by routed source without rereading config or mutating process-global environment.
|
|
self._busy_input_modes_by_profile: Dict[str, str] = {}
|
|
self._busy_text_modes_by_profile: Dict[str, str] = {}
|
|
self._restart_drain_timeout = self._load_restart_drain_timeout()
|
|
self._restart_after_turn_timeout = self._load_restart_after_turn_timeout()
|
|
self._cron_drain_timeout = self._load_cron_drain_timeout()
|
|
self._signal_interrupt_grace_timeout = (
|
|
self._load_signal_interrupt_grace_timeout()
|
|
)
|
|
self._provider_routing = self._load_provider_routing()
|
|
self._fallback_model = self._load_fallback_model()
|
|
|
|
def _init_session_store(self) -> None:
|
|
"""Build the SessionStore (with process-registry reset guard), its async facade and the router."""
|
|
# Wire process registry into session store for reset protection. A background process older
|
|
# than session_reset.bg_process_max_age_hours (default 24h) is stale and no longer blocks
|
|
# idle/daily reset. The process is NOT killed, only ignored by the reset guard.
|
|
from tools.process_registry import process_registry
|
|
_bg_max_age_hours = getattr(
|
|
self.config.default_reset_policy, "bg_process_max_age_hours", 24
|
|
)
|
|
_bg_max_age_seconds = (
|
|
_bg_max_age_hours * 3600 if _bg_max_age_hours and _bg_max_age_hours > 0 else None
|
|
)
|
|
self.session_store = SessionStore(
|
|
self.config.sessions_dir, self.config,
|
|
has_active_processes_fn=lambda key: process_registry.has_active_for_session(
|
|
key, max_active_age=_bg_max_age_seconds,
|
|
),
|
|
)
|
|
# One enforced loop-side boundary for the synchronous SessionStore: sync helpers keep using
|
|
# ``session_store`` directly; async gateway handlers call this facade and await every op.
|
|
self._async_session_store = AsyncSessionStore(self.session_store)
|
|
self.delivery_router = DeliveryRouter(self.config)
|
|
|
|
def _init_lifecycle_state(self) -> None:
|
|
"""Initialise run/exit/restart flags, per-session state, and completion-delivery bookkeeping."""
|
|
self._running = False
|
|
self._gateway_loop: Optional[asyncio.AbstractEventLoop] = None
|
|
self._shutdown_event = asyncio.Event()
|
|
self._exit_cleanly = False
|
|
self._exit_with_failure = False
|
|
self._exit_reason: Optional[str] = None
|
|
self._exit_code: Optional[int] = None
|
|
self._draining = False
|
|
self._profile_failed_platforms: Dict[str, Dict[Platform, asyncio.Task]] = {}
|
|
self._systemd_watchdog = None
|
|
# External (NAS-driven) drain state, distinct from the shutdown ``_draining`` flag: set by
|
|
# ``_drain_control_watcher`` when ``.drain_request.json`` exists — NEW turns refused, but the
|
|
# process stays up and removing the marker reverts to ``running``. ``_draining`` is one-way.
|
|
self._external_drain_active = False
|
|
self._restart_requested = False
|
|
# Set by shutdown_signal_handler when SIGTERM/SIGINT arrived WITHOUT a planned-stop/takeover
|
|
# marker (container SIGTERM, OOM-killer, bare `kill`); _stop_impl must NOT persist
|
|
# gateway_state=stopped for an unexpected signal, or container_boot won't auto-start next boot.
|
|
self._signal_initiated_shutdown = False
|
|
self._restart_task_started = False
|
|
self._restart_detached = False
|
|
self._restart_via_service = False
|
|
self._detached_restart_helper_started = False
|
|
self._restart_command_source: Optional[SessionSource] = None
|
|
# Monotonic-ish wall clock of when this GatewayRunner was constructed. Used by the /restart
|
|
# redelivery guard to bound the window where a missing dedup marker means a stale redelivery.
|
|
self._startup_time: float = time.time()
|
|
# True when this process booted from a chat-originated /restart (.restart_notify.json existed
|
|
# on boot). One-shot signal consumed by _is_stale_restart_redelivery so the marker-missing
|
|
# fallback suppresses a /restart only when we KNOW we just restarted — never on a fresh boot.
|
|
self._booted_from_restart: bool = False
|
|
self._stop_task: Optional[asyncio.Task] = None
|
|
self._restart_task: Optional[asyncio.Task] = None
|
|
self._executor_lock = threading.Lock()
|
|
self._executor: Optional[concurrent.futures.ThreadPoolExecutor] = None
|
|
# Set on gateway stop so the recreate-on-shutdown path can't resurrect
|
|
# the pool during a real shutdown.
|
|
self._executor_closing = False
|
|
# ALL per-session state (turn / conversation / persistent scopes) lives in one container —
|
|
# see gateway/session_state.py. Access via self._session_state(key) (get-or-create) or
|
|
# self._peek_session_state(key) (read-only).
|
|
self._sessions: Dict[str, SessionState] = {}
|
|
# Per-SESSION_ID turn lease: serializes the [load history → run → flush] region when two
|
|
# ROUTING KEYS resolve to one session_id (switch_session's many-to-one mapping). The
|
|
# routing-key guards above cannot see that overlap.
|
|
self._turn_leases = SessionTurnLeaseRegistry()
|
|
# Turn-lease tokens live on SessionState.turn.lease_token/.lease_generation; the generation
|
|
# check means a stale unwind can never free a newer turn's lease. pending_command_text is
|
|
# runner-level queued interrupt text (distinct from adapter-level _pending_messages).
|
|
# last_resolved_model backs a config read that transiently returns "" (else model="" → every
|
|
# call HTTP 400); "*" is the process-wide fallback. queued_events is /queue overflow, promoted
|
|
# one per drain FIFO; cleared on /new and /reset, kept across /model. The run-generation
|
|
# counter is monotonic and NEVER reset. Stall-notified keys clear when pending clears /
|
|
# activity resumes / conversation boundary (gateway.session_stall).
|
|
self._session_stall_notified: Dict[str, bool] = {}
|
|
# Startup restore gate: while restart-interrupted sessions are being auto-resumed, real
|
|
# inbound messages are queued instead of competing with the synthetic resume turns for the
|
|
# same session. The queued events drain only after all startup resume tasks have finished.
|
|
self._startup_restore_in_progress = False
|
|
# Set by start_gateway() only for an explicit ``--replace`` launch.
|
|
# _connect_initial_adapter_with_timeout scopes it to each adapter's
|
|
# cold-start connect and removes it before any reconnect can run.
|
|
self._platform_lock_takeover_on_start = False
|
|
self._startup_restore_queue: List[MessageEvent] = []
|
|
self._startup_restore_tasks: List[asyncio.Task] = []
|
|
# LRU cache of live SessionSources keyed by session_key. Used by fallback routing paths
|
|
# (shutdown notifications, synthetic background-process events) when the persisted origin is
|
|
# missing and _parse_session_key can't recover thread_id. Capped so it cannot grow unbounded.
|
|
self._session_sources: "OrderedDict[str, SessionSource]" = OrderedDict()
|
|
self._session_sources_max = 512
|
|
# Completion delivery is intentionally lifecycle-scoped: it closes duplicate queue/watcher
|
|
# races inside one gateway without pretending adapter send + persistence write are exactly-once
|
|
# across a crash. Durable async-delegation replay state stays owned by tools.async_delegation.
|
|
self._completion_delivery_lock = threading.Lock()
|
|
self._completion_deliveries_inflight: set[tuple[str, str, object]] = set()
|
|
self._completion_deliveries_delivered: "OrderedDict[tuple[str, str, object], None]" = OrderedDict()
|
|
self._completion_delivery_retention = 2048
|
|
# Agent-triggered terminal completions from one conversation often land in the same scheduler
|
|
# tick; hold them briefly so the agent gets one synthetic turn instead of one per process.
|
|
self._completion_notification_batches: dict[tuple[str, ...], list[tuple[str, dict, asyncio.Future]]] = {}
|
|
self._completion_notification_batch_tasks: dict[tuple[str, ...], asyncio.Task] = {}
|
|
self._completion_notification_batch_flush_tasks: set[asyncio.Task] = set()
|
|
self._completion_notification_batch_window = 0.1
|
|
self._completion_notification_batches_stopping = False
|
|
|
|
def _init_runtime_caches(self) -> None:
|
|
"""Agent cache, profile identity, Teams runtime, failed-platform tracking, slash-confirm counter."""
|
|
# Cache AIAgent instances per session to preserve prompt caching (a fresh agent per message
|
|
# rebuilds the system prompt and breaks the prefix cache, ~10x cost on Anthropic). Value:
|
|
# (AIAgent, config_signature_str). OrderedDict for LRU eviction in _enforce_agent_cache_cap();
|
|
# hard cap _AGENT_CACHE_MAX_SIZE, idle TTL from _session_expiry_watcher().
|
|
import threading as _threading
|
|
self._agent_cache: "OrderedDict[str, tuple]" = OrderedDict()
|
|
self._agent_cache_lock = _threading.Lock()
|
|
|
|
# Conversation-scoped per-session state (/model, /model --once, /reasoning, /fast overrides;
|
|
# per-turn sidecar notes; ephemeral context pin; last-delivered voice-channel context) lives
|
|
# on SessionState.conversation — see gateway/session_state.py.
|
|
self._kanban_notifier_profile = self._active_profile_name()
|
|
# Launch-time identity of the profile that owns ``self.adapters``; ``_authorization_adapter``
|
|
# compares against this rather than the per-turn ``_active_profile_name()``.
|
|
self._primary_profile_name = self._kanban_notifier_profile
|
|
# Teams meeting pipeline runtime (bound later when msgraph_webhook adapter exists).
|
|
self._teams_pipeline_runtime = None
|
|
self._teams_pipeline_runtime_error: Optional[str] = None
|
|
# Pending exec approvals live on SessionState.persistent.approvals.
|
|
|
|
# Track platforms that failed to connect for background reconnection.
|
|
# Key: Platform enum, Value: {"config": platform_config, "attempts": int, "next_retry": float}
|
|
self._failed_platforms: Dict[Platform, Dict[str, Any]] = {}
|
|
|
|
# Strong refs to detached fatal-error handler tasks (see
|
|
# _handle_adapter_fatal_error) so the event loop can't GC them mid-run.
|
|
self._fatal_handler_tasks: set = set()
|
|
|
|
# Pending /update prompt flags live on
|
|
# SessionState.persistent.update_prompt_pending.
|
|
|
|
# Slash-confirm state lives in tools.slash_confirm (module-level), so platform adapters can
|
|
# resolve callbacks without a backref to this runner. Keep a local counter for confirm_id
|
|
# generation so IDs stay compact (button callback_data has a 64-byte cap on some platforms).
|
|
import itertools as _itertools
|
|
self._slash_confirm_counter = _itertools.count(1)
|
|
|
|
def _init_startup_checks(self) -> None:
|
|
"""Ensure tirith is installed and warn when manual approvals have no automated assessor."""
|
|
# Persistent Honcho managers keyed by gateway session key: preserves write_frequency="session"
|
|
# semantics across short-lived per-message AIAgent instances.
|
|
|
|
# Ensure tirith security scanner is available (downloads if needed)
|
|
try:
|
|
from tools.tirith_security import ensure_installed
|
|
ensure_installed(log_failures=False)
|
|
except Exception:
|
|
pass # Non-fatal — fail-open at scan time if unavailable
|
|
|
|
# Startup heads-up: manual approval mode with no automated risk assessor (tirith disabled AND
|
|
# no auxiliary.approval model) can only gate dangerous commands via live in-chat approval, so
|
|
# they fail closed on unattended gateways — surface it so operators knowingly enable one.
|
|
try:
|
|
from hermes_cli.config import load_config as _load_full_config
|
|
_appr_cfg = _load_full_config()
|
|
_appr_mode = str(
|
|
cfg_get(_appr_cfg, "approvals", "mode", default="manual") or "manual"
|
|
).strip().lower()
|
|
_tirith_on = bool(cfg_get(_appr_cfg, "security", "tirith_enabled", default=True))
|
|
_aux_approval = cfg_get(_appr_cfg, "auxiliary", "approval", default=None)
|
|
if _appr_mode == "manual" and not _tirith_on and not _aux_approval:
|
|
logger.warning(
|
|
"Gateway approvals.mode=manual with no automated risk "
|
|
"assessor (security.tirith_enabled is false and "
|
|
"auxiliary.approval is unset): dangerous commands and "
|
|
"execute_code scripts will BLOCK until a human approves "
|
|
"them in chat. Enable security.tirith_enabled or configure "
|
|
"auxiliary.approval for unattended operation."
|
|
)
|
|
except Exception:
|
|
logger.debug("approvals.mode startup check skipped", exc_info=True)
|
|
|
|
def _init_session_db(self) -> None:
|
|
"""Open the session DB for the active scope and run opportunistic state.db / checkpoint maintenance."""
|
|
# Session DB for session_search: a property caches one AsyncSessionDB per path (not a handle
|
|
# bound here, which would pin the root home — /resume, /title, /history and search run inside
|
|
# _profile_runtime_scope under multiplex); priming here keeps startup diagnostics at init.
|
|
self._session_db_pinned: Any = _SESSION_DB_UNPINNED
|
|
self._session_db_handles: Dict[Path, Any] = {}
|
|
self._session_db_handles_lock = threading.Lock()
|
|
from gateway.session_db_recovery import RecoverableHandleCache
|
|
|
|
self._session_db_handle_cache = RecoverableHandleCache(
|
|
handles=self._session_db_handles,
|
|
lock=self._session_db_handles_lock,
|
|
)
|
|
try:
|
|
self._open_session_db_for_active_scope(raise_on_error=True)
|
|
except Exception as e:
|
|
# WARNING (not DEBUG) so it lands in errors.log, matching cli.py; otherwise an NFS-mounted
|
|
# HERMES_HOME silently loses /resume, /title, /history, /branch and session search.
|
|
logger.warning("SQLite session store not available: %s", e)
|
|
# Surface the failure on the user's home channel(s) once connected; otherwise state.db
|
|
# corruption or NFS/SMB lock failures silently degrade the gateway (nothing persists).
|
|
self._session_db_init_error = str(e)
|
|
|
|
# Opportunistic state.db maintenance: prune ended sessions past sessions.retention_days +
|
|
# optional VACUUM, at most once per sessions.min_interval_hours (last-run in state_meta).
|
|
# A few blocking seconds per day is fine for a long-lived gateway; failures log, never raise.
|
|
if self._session_db is not None:
|
|
try:
|
|
from hermes_cli.config import load_config as _load_full_config
|
|
_sess_cfg = (_load_full_config().get("sessions") or {})
|
|
# Non-destructive stale-session archive, independent of prune.
|
|
if _sess_cfg.get("auto_archive", False):
|
|
self._session_db._db.maybe_auto_archive(
|
|
idle_days=float(_sess_cfg.get("auto_archive_days", 3)),
|
|
min_interval_hours=int(_sess_cfg.get("min_interval_hours", 24)),
|
|
)
|
|
if _sess_cfg.get("auto_prune", False):
|
|
# Construction-time, before the loop serves traffic; sync DB is fine.
|
|
self._session_db._db.maybe_auto_prune_and_vacuum(
|
|
retention_days=int(_sess_cfg.get("retention_days", 90)),
|
|
min_interval_hours=int(_sess_cfg.get("min_interval_hours", 24)),
|
|
min_vacuum_interval_days=int(
|
|
_sess_cfg.get("min_vacuum_interval_days", 30)
|
|
),
|
|
vacuum=bool(_sess_cfg.get("vacuum_after_prune", True)),
|
|
sessions_dir=self.config.sessions_dir,
|
|
)
|
|
except Exception as exc:
|
|
logger.debug("state.db auto-maintenance skipped: %s", exc)
|
|
|
|
# Opportunistic shadow-repo cleanup of stale checkpoint repos under ~/.hermes/checkpoints/;
|
|
# opt-in via checkpoints.auto_prune, idempotent via .last_prune marker.
|
|
try:
|
|
from hermes_cli.config import load_config as _load_full_config
|
|
_ckpt_cfg = (_load_full_config().get("checkpoints") or {})
|
|
if _ckpt_cfg.get("auto_prune", False):
|
|
from tools.checkpoint_manager import maybe_auto_prune_checkpoints
|
|
# delete_orphans is never honoured here: this sweep runs unattended and a missing
|
|
# workdir at startup is ambiguous (deleted project vs. unmounted volume/share/VPN).
|
|
# Orphan cleanup happens only via the explicit `hermes checkpoints prune` command.
|
|
maybe_auto_prune_checkpoints(
|
|
retention_days=int(_ckpt_cfg.get("retention_days", 7)),
|
|
min_interval_hours=int(_ckpt_cfg.get("min_interval_hours", 24)),
|
|
delete_orphans=False,
|
|
max_total_size_mb=int(_ckpt_cfg.get("max_total_size_mb", 500)),
|
|
)
|
|
except Exception as exc:
|
|
logger.debug("checkpoint auto-maintenance skipped: %s", exc)
|
|
|
|
def _init_registries_and_clocks(self) -> None:
|
|
"""Pairing stores, hook registry, voice modes, background-task set, liveness and idle clocks."""
|
|
# DM pairing store for code-based user authorization. ``pairing_store`` is the global/default
|
|
# store (``hermes pairing`` CLI, callers without profile context); ``pairing_stores`` is the
|
|
# per-profile map ``authz_mixin._is_user_authorized`` routes through (one whitelist/profile).
|
|
from gateway.pairing import PairingStore
|
|
self.pairing_store = PairingStore()
|
|
self.pairing_stores: Dict[str, "PairingStore"] = {}
|
|
|
|
# Event hook system
|
|
from gateway.hooks import HookRegistry
|
|
self.hooks = HookRegistry()
|
|
|
|
# Per-chat voice reply mode: "off" | "voice_only" | "all"
|
|
self._voice_mode: Dict[str, str] = self._load_voice_modes()
|
|
# Recent voice transcripts per (guild,user): the voice capture / STT pipeline can emit the
|
|
# same utterance twice, which would otherwise produce a second delayed reply.
|
|
self._recent_voice_transcripts: Dict[tuple[int, int], List[tuple[float, str]]] = {}
|
|
|
|
# Track background tasks to prevent garbage collection mid-execution
|
|
self._background_tasks: set = set()
|
|
|
|
# Event-loop liveness heartbeat: rewritten every 30s while the loop dispatches; supervisors
|
|
# use the file mtime / updated_at to tell "process alive" from "loop frozen".
|
|
self._gateway_started_at: float = time.time()
|
|
self._loop_heartbeat_task: Optional[asyncio.Task] = None
|
|
self._loop_floor_timer_handle = None
|
|
self._loop_liveness_watchdog = None
|
|
|
|
# scale-to-zero: gateway-scoped "last inbound seen" clock (only a per-agent _last_activity_ts
|
|
# exists otherwise). Stamped in _handle_message (the single inbound chokepoint), seeded to
|
|
# "now" so a fresh gateway isn't idle from epoch; read by the scale-to-zero watcher.
|
|
self._last_inbound_at: float = time.time()
|
|
# Set after a wake (re-arm cooldown, 0.F) so we don't immediately re-go
|
|
# dormant before the drained backlog has a chance to update the clock.
|
|
self._scale_to_zero_cooldown_until: float = 0.0
|
|
# One-shot: log the "platform owns the suspend" notice once, not per tick.
|
|
self._scale_to_zero_no_suspend_logged: bool = False
|
|
|
|
def _open_session_db_for_active_scope(self, raise_on_error: bool = False) -> Any:
|
|
"""Return the AsyncSessionDB for the profile scope active on this task.
|
|
|
|
``SessionDB()`` resolves its path at call time via the context-local HERMES_HOME override
|
|
from ``_profile_runtime_scope``, so resolving per access (not in ``__init__``) lets a
|
|
multiplexed profile read its own store. One ``AsyncSessionDB`` is cached per path (stable
|
|
identity; profiles never share a handle). Construction failure enters bounded backoff;
|
|
``raise_on_error=True`` (priming) propagates it after recording state for ``__init__``.
|
|
"""
|
|
from hermes_state import AsyncSessionDB, _default_db_path, get_shared_session_db
|
|
from gateway.session_db_recovery import RecoverableHandleCache
|
|
|
|
path = Path(_default_db_path())
|
|
cache = getattr(self, "_session_db_handle_cache", None)
|
|
if cache is None:
|
|
# Compatibility for lightweight test runners built with
|
|
# object.__new__ rather than GatewayRunner.__init__.
|
|
cache = RecoverableHandleCache(
|
|
handles=self._session_db_handles,
|
|
lock=self._session_db_handles_lock,
|
|
)
|
|
self._session_db_handle_cache = cache
|
|
|
|
def _open():
|
|
# Borrow the SessionStore's handle for this path rather than opening a second one (both
|
|
# caches resolve the SAME path; otherwise two writer connections + two read pools per
|
|
# state.db, doubled per profile). The store owns and sweeps the handle at shutdown; this
|
|
# cache holds only the async wrapper. It cannot go stale: the store drops handles only in
|
|
# close_all_db_handles(), and while its open fails there is nothing to borrow or cache.
|
|
store = getattr(self, "session_store", None)
|
|
borrowed = getattr(store, "_db", None) if store is not None else None
|
|
if borrowed is not None:
|
|
wrapper = AsyncSessionDB(borrowed)
|
|
# close_all_session_db_handles() must not close what the store
|
|
# owns; the store's own sweep already does, and it runs first.
|
|
wrapper.__dict__["_hermes_borrowed_handle"] = True
|
|
return wrapper
|
|
if store is not None:
|
|
# Store handle unavailable (failed open/backoff): opening our own would resurrect the
|
|
# very duplicate this borrows away, so report the store's own unavailability.
|
|
raise RuntimeError("SessionStore SQLite handle unavailable")
|
|
try:
|
|
return AsyncSessionDB(get_shared_session_db())
|
|
except Exception as exc:
|
|
logger.warning("SQLite session store not available: %s", exc)
|
|
raise
|
|
|
|
def _recovered() -> None:
|
|
self._session_db_init_error = None
|
|
logger.info("SQLite session store recovered")
|
|
|
|
return cache.get(
|
|
path,
|
|
_open,
|
|
raise_on_error=raise_on_error,
|
|
on_recovered=_recovered,
|
|
)
|
|
|
|
@property
|
|
def _session_db(self) -> Any:
|
|
"""The AsyncSessionDB for the active profile scope, or a pinned override.
|
|
|
|
Assigning ``runner._session_db`` pins that value for every later read (tests install fakes
|
|
or ``None`` this way); unpinned, each read resolves the active profile scope's own store.
|
|
"""
|
|
if self._session_db_pinned is not _SESSION_DB_UNPINNED:
|
|
return self._session_db_pinned
|
|
return self._open_session_db_for_active_scope()
|
|
|
|
@_session_db.setter
|
|
def _session_db(self, value) -> None:
|
|
self._session_db_pinned = value
|
|
|
|
def close_all_session_db_handles(self) -> None:
|
|
"""Close every per-profile AsyncSessionDB this runner opened.
|
|
|
|
Handles are drained under the lock and closed outside it; a pinned handle is the pinner's to
|
|
close. Wrappers BORROWED from ``session_store`` are drained but not closed: the store's own
|
|
sweep, which runs first in the shutdown sequence, closes that connection.
|
|
"""
|
|
def _close(db) -> None:
|
|
if getattr(db, "__dict__", {}).get("_hermes_borrowed_handle"):
|
|
return
|
|
inner = getattr(db, "_db", db)
|
|
if inner is None or not hasattr(inner, "close"):
|
|
return
|
|
from hermes_state import release_or_close
|
|
try:
|
|
release_or_close(inner)
|
|
except Exception as exc:
|
|
logger.debug("SessionDB close error during handle sweep: %s", exc)
|
|
|
|
self._session_db_handle_cache.close_all(_close)
|
|
|
|
def _wire_teams_pipeline_runtime(self) -> None:
|
|
"""Bind the Teams meeting pipeline runtime to Graph webhook ingress.
|
|
|
|
No-op when the msgraph_webhook adapter isn't running or the teams_pipeline plugin is off.
|
|
"""
|
|
if Platform.MSGRAPH_WEBHOOK not in self.adapters:
|
|
return
|
|
if not _teams_pipeline_plugin_enabled():
|
|
logger.debug("Teams pipeline plugin is disabled; skipping runtime wiring")
|
|
return
|
|
try:
|
|
from plugins.teams_pipeline.runtime import bind_gateway_runtime
|
|
except Exception as exc:
|
|
logger.warning("Teams pipeline runtime import failed: %s", exc)
|
|
return
|
|
try:
|
|
bound = bind_gateway_runtime(self)
|
|
except Exception as exc:
|
|
logger.warning("Teams pipeline runtime wiring failed: %s", exc)
|
|
return
|
|
if bound:
|
|
logger.info("Teams pipeline runtime bound to msgraph webhook ingress")
|
|
elif self._teams_pipeline_runtime_error:
|
|
logger.warning(
|
|
"Teams pipeline runtime unavailable: %s",
|
|
self._teams_pipeline_runtime_error,
|
|
)
|
|
|
|
def _warn_if_docker_media_delivery_is_risky(self) -> None:
|
|
"""Warn when Docker-backed gateways lack an explicit export mount.
|
|
|
|
MEDIA delivery runs in the gateway process, so model-emitted paths like `/output/report.txt`
|
|
must be host-readable — users need an export mount such as `host-dir:/output`.
|
|
"""
|
|
if os.getenv("TERMINAL_ENV", "").strip().lower() != "docker":
|
|
return
|
|
|
|
connected = self.config.get_connected_platforms()
|
|
messaging_platforms = [p for p in connected if p not in {Platform.LOCAL, Platform.API_SERVER, Platform.WEBHOOK}]
|
|
if not messaging_platforms:
|
|
return
|
|
|
|
raw_volumes = os.getenv("TERMINAL_DOCKER_VOLUMES", "").strip()
|
|
volumes: List[str] = []
|
|
if raw_volumes:
|
|
try:
|
|
parsed = json.loads(raw_volumes)
|
|
if isinstance(parsed, list):
|
|
volumes = [str(v) for v in parsed if isinstance(v, str)]
|
|
except Exception:
|
|
logger.debug("Could not parse TERMINAL_DOCKER_VOLUMES for gateway media warning", exc_info=True)
|
|
|
|
has_explicit_output_mount = False
|
|
for spec in volumes:
|
|
match = _DOCKER_VOLUME_SPEC_RE.match(spec)
|
|
if not match:
|
|
continue
|
|
container_path = match.group("container")
|
|
if container_path in _DOCKER_MEDIA_OUTPUT_CONTAINER_PATHS:
|
|
has_explicit_output_mount = True
|
|
break
|
|
|
|
if has_explicit_output_mount:
|
|
return
|
|
|
|
logger.warning(
|
|
"Docker backend is enabled for the messaging gateway but no explicit host-visible "
|
|
"output mount (for example '/home/user/.hermes/cache/documents:/output') is configured. "
|
|
"This is fine if the model already emits host-visible paths, but MEDIA file delivery can fail "
|
|
"for container-local paths like '/workspace/...' or '/output/...'."
|
|
)
|
|
|
|
# -- Voice mode persistence ------------------------------------------
|
|
|
|
_VOICE_MODE_PATH = _hermes_home / "gateway_voice_mode.json"
|
|
|
|
|
|
@property
|
|
def should_exit_cleanly(self) -> bool:
|
|
return self._exit_cleanly
|
|
|
|
@property
|
|
def should_exit_with_failure(self) -> bool:
|
|
return self._exit_with_failure
|
|
|
|
@property
|
|
def exit_reason(self) -> Optional[str]:
|
|
return self._exit_reason
|
|
|
|
@property
|
|
def exit_code(self) -> Optional[int]:
|
|
return self._exit_code
|
|
|
|
def _session_key_for_source(self, source: SessionSource) -> str:
|
|
"""Resolve the current session key for a source, honoring gateway config when available."""
|
|
if hasattr(self, "session_store") and self.session_store is not None:
|
|
try:
|
|
session_key = self.session_store._generate_session_key(source)
|
|
if isinstance(session_key, str) and session_key:
|
|
return session_key
|
|
except Exception:
|
|
pass
|
|
config = getattr(self, "config", None)
|
|
# Mirror SessionStore._resolve_profile_for_key so this fallback yields the primary path's
|
|
# namespace: None (legacy agent:main) unless multiplexing is on, then the active profile.
|
|
_profile = None
|
|
if getattr(config, "multiplex_profiles", False):
|
|
if source.profile:
|
|
_profile = source.profile
|
|
else:
|
|
try:
|
|
from hermes_cli.profiles import get_active_profile_name
|
|
_profile = get_active_profile_name() or "default"
|
|
except Exception:
|
|
_profile = None
|
|
return build_session_key(
|
|
source,
|
|
group_sessions_per_user=getattr(config, "group_sessions_per_user", True),
|
|
thread_sessions_per_user=getattr(config, "thread_sessions_per_user", False),
|
|
profile=_profile,
|
|
)
|
|
|
|
|
|
# Telegram's General (pinned top) topic in forum-enabled private chats: clients variously omit
|
|
# message_thread_id or send "1" for it. Treat both as "root" for lobby/lane purposes.
|
|
_TELEGRAM_GENERAL_TOPIC_IDS = frozenset({"", "1"})
|
|
|
|
|
|
_TELEGRAM_LOBBY_REMINDER_COOLDOWN_S = 30.0
|
|
|
|
|
|
def _normalize_source_for_session_key(
|
|
self,
|
|
source: SessionSource,
|
|
) -> SessionSource:
|
|
"""Apply Telegram DM topic recovery to a source for session-key purposes.
|
|
|
|
``_handle_message_with_agent`` rewrites ``source.thread_id`` via
|
|
``_recover_telegram_topic_thread_id`` *before* deriving the session key, so handlers like
|
|
``/model`` keying off the raw ``event.source`` would store an override under a different key
|
|
than the next turn reads (silently dropped on forum topics / after splits). Returns a
|
|
recovery-normalized copy when a rewrite applies, else the original; always derive the override
|
|
storage key from the result so storage and read match.
|
|
"""
|
|
try:
|
|
recovered = self._recover_telegram_topic_thread_id(source)
|
|
except Exception:
|
|
return source
|
|
if recovered is None:
|
|
return source
|
|
return dataclasses.replace(source, thread_id=recovered)
|
|
|
|
def _resolve_session_key_or_none(self, source, session_key: Optional[str]) -> Optional[str]:
|
|
"""``session_key`` if given, else the key for ``source`` (None when it cannot be derived)."""
|
|
if session_key or source is None:
|
|
return session_key
|
|
try:
|
|
return self._session_key_for_source(source)
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def _running_agent_count(self) -> int:
|
|
return len(self._running_agents)
|
|
|
|
|
|
# ── scale-to-zero idle detection / dormant-quiesce (Phase 0) ──────────────
|
|
# The gateway-side BEHAVIOUR that consumes the relay scale-to-zero primitives (gateway-gateway
|
|
# Phase 5). Pure logic lives in gateway/scale_to_zero.py; the methods here bind it to the live
|
|
# runner/transport.
|
|
|
|
|
|
def _status_action_label(self) -> str:
|
|
return "restart" if self._restart_requested else "shutdown"
|
|
|
|
def _status_action_gerund(self) -> str:
|
|
return "restarting" if self._restart_requested else "shutting down"
|
|
|
|
|
|
# -------- /queue FIFO helpers --------------------------------------
|
|
# /queue yields one full agent turn per invocation, FIFO, no merging. _pending_messages is a
|
|
# single "next-up" slot (shared with photo-burst follow-ups) holding the head; an overflow list
|
|
# holds the tail. Promotion after each run's drain refills the slot. Cleared on /new and /reset.
|
|
|
|
|
|
def _update_runtime_status(self, gateway_state: Optional[str] = None, exit_reason: Optional[str] = None) -> None:
|
|
_write_runtime_status_quiet(
|
|
gateway_state=gateway_state,
|
|
exit_reason=exit_reason,
|
|
restart_requested=self._restart_requested,
|
|
active_agents=self._active_work_count(),
|
|
)
|
|
|
|
def _persist_active_agents(self) -> None:
|
|
"""Persist the live in-flight agent count to ``gateway_state.json``.
|
|
|
|
Called at every turn boundary so the dashboard ``/api/status`` readout is near-real-time
|
|
(otherwise the file only moves on lifecycle transitions). Passes ONLY ``active_agents`` —
|
|
other fields stay ``_UNSET`` so the read-merge-write preserves lifecycle state; passing
|
|
``gateway_state=None`` would clobber it. Best-effort: a failed write must never disrupt a turn.
|
|
"""
|
|
_write_runtime_status_quiet(active_agents=self._active_work_count())
|
|
|
|
|
|
def _running_agent_ids(self) -> set:
|
|
"""``id()`` of every agent mid-turn — identity-keyed so the lookup is O(1) and independent of
|
|
``AIAgent.__eq__`` (MagicMock overrides it in tests)."""
|
|
return {
|
|
id(a)
|
|
for _, a in self._running_agent_items()
|
|
if a is not None and a is not _AGENT_PENDING_SENTINEL
|
|
}
|
|
|
|
def _snapshot_running_agents(self) -> Dict[str, Any]:
|
|
return {
|
|
session_key: agent
|
|
for session_key, agent in self._running_agent_items()
|
|
if agent is not _AGENT_PENDING_SENTINEL
|
|
}
|
|
|
|
|
|
# Hard cap on per-session pending follow-ups for busy_input_mode=queue (and the draining/steer-
|
|
# fallback/subagent-demotion paths that share this entry point). Without a cap, a stuck agent +
|
|
# a rapid-fire user could grow the overflow list unboundedly.
|
|
_BUSY_QUEUE_MAX_PENDING = 32
|
|
|
|
|
|
@dataclasses.dataclass
|
|
class _BusySteerOutcome:
|
|
effective_mode: str
|
|
demoted_for_subagents: bool
|
|
demoted_for_compression: bool
|
|
steered: bool
|
|
redirected: bool
|
|
|
|
|
|
# Bound for off-loop agent-resource cleanup from event-loop coroutines (expiry sweep, cache-hygiene
|
|
# re-eviction). _cleanup_agent_resources is synchronous and can block long (subprocess teardown,
|
|
# memory-provider network/SQLite IO); inline it wedges the loop, so it runs in a worker thread.
|
|
_CLEANUP_TIMEOUT_S = 30.0
|
|
|
|
|
|
# Budget for one finalize_session() dispatch (plugin on_session_finalize hooks + Relay close):
|
|
# enough for a normal trace-export flush, small enough a wedged plugin can't eat the stop window.
|
|
_FINALIZE_TIMEOUT_S = 10.0
|
|
|
|
|
|
_STUCK_LOOP_THRESHOLD = 3 # restarts while active before auto-suspend
|
|
_STUCK_LOOP_FILE = ".restart_failure_counts"
|
|
|
|
|
|
# Reasons set by _stop_impl() on force-interrupt; "restart_interrupted" by suspend_recently_active()
|
|
# on crash recovery (no .clean_shutdown marker). All mean "killed mid-turn" -> startup auto-resume.
|
|
_AUTO_RESUME_REASONS = frozenset(
|
|
{"restart_timeout", "shutdown_timeout", "restart_interrupted"}
|
|
)
|
|
|
|
|
|
_MAX_SUPERVISED_RESTARTS = 5
|
|
# A task that ran at least this long before crashing is HEALTHY: an isolated crash, not a
|
|
# crash-loop; the consecutive-restart counter resets so a long-lived daemon isn't abandoned.
|
|
_SUPERVISED_HEALTHY_SECS = 300
|
|
|
|
|
|
def _active_profile_name(self) -> str:
|
|
"""Return the profile name this gateway represents."""
|
|
try:
|
|
from hermes_cli.profiles import get_active_profile_name
|
|
return get_active_profile_name() or "default"
|
|
except Exception:
|
|
return "default"
|
|
|
|
# ── Kanban board watchers ───────────────────────────────────────────
|
|
# Loops + helpers live in GatewayKanbanWatchersMixin (gateway/kanban_watchers.py).
|
|
|
|
#: Slow respawn tier interval, used once the reconnect watcher has exhausted its supervised
|
|
#: restart budget. Long on purpose: the budget is spent when the watcher is crashing on contact,
|
|
#: so the useful cadence is "check back later"; a tight loop would be worse than the outage.
|
|
_RECONNECT_WATCHER_SLOW_RETRY_SECS = 300
|
|
|
|
#: Slow-tier respawns to attempt while work is still queued. Bounded: if half an hour of
|
|
#: five-minute retries cannot keep a watcher alive, the fault is not transient — fail loudly.
|
|
_MAX_SLOW_WATCHER_RESPAWNS = 6
|
|
|
|
|
|
# Reconnect is scoped to the profile's own config and secret mapping;
|
|
# never rebuild a secondary adapter with the default profile's credentials.
|
|
|
|
|
|
def _is_user_authorized_for_source(
|
|
self,
|
|
source: SessionSource,
|
|
*,
|
|
allow_adapter_delegation: bool = True,
|
|
) -> bool:
|
|
"""Authorize under the live transport's profile, not the routed runtime.
|
|
|
|
The routed runtime profile need not copy the shared bot token or allowlist. The primary
|
|
handlers stamp the transport home as an in-process attribute; read it for authorization
|
|
only, then restore the routed scope for the rest of the turn.
|
|
"""
|
|
def _check() -> bool:
|
|
# Preserve the historical one-argument seam used by plugins/tests;
|
|
# only pass the keyword for the explicit delegation-disabled path.
|
|
if allow_adapter_delegation:
|
|
return self._is_user_authorized(source)
|
|
return self._is_user_authorized(
|
|
source,
|
|
allow_adapter_delegation=False,
|
|
)
|
|
|
|
authorization_home = getattr(source, "_authorization_profile_home", None)
|
|
if authorization_home is not None:
|
|
with _profile_runtime_scope(Path(authorization_home)):
|
|
return _check()
|
|
return _check()
|
|
|
|
|
|
# ------------------------------------------------------------------
|
|
# Mid-run (busy-session) slash command dispatch — "Guard 2".
|
|
# Each command's mid-run behavior is declared on its CommandDef (busy_policy / busy_handler
|
|
# in hermes_cli/commands.py) and resolved through a single handler table.
|
|
# ------------------------------------------------------------------
|
|
|
|
# Command-specific mid-run reject texts (busy_policy == "reject" with a busy_handler naming an
|
|
# entry here); all other rejected commands get the generic text in _dispatch_busy_slash_command.
|
|
_BUSY_REJECT_TEXT: Dict[str, str] = {
|
|
"model": "Agent is running — wait or /stop first, then switch models.",
|
|
"codex-runtime": ("Agent is running — wait or /stop first, then "
|
|
"change runtime."),
|
|
"moa": "Agent is running — wait or /stop first, then run /moa.",
|
|
}
|
|
|
|
|
|
def _cache_session_source(self, session_key: str, source) -> None:
|
|
if not session_key or source is None:
|
|
return
|
|
cached_sources = getattr(self, "_session_sources", None)
|
|
if cached_sources is None:
|
|
cached_sources = OrderedDict()
|
|
self._session_sources = cached_sources
|
|
try:
|
|
cached_sources[session_key] = dataclasses.replace(source)
|
|
except Exception:
|
|
logger.debug("Failed to cache live session source for %s", session_key, exc_info=True)
|
|
return
|
|
# LRU: mark as most-recently-used and trim to max size.
|
|
try:
|
|
cached_sources.move_to_end(session_key)
|
|
max_size = getattr(self, "_session_sources_max", 512)
|
|
while len(cached_sources) > max_size:
|
|
cached_sources.popitem(last=False)
|
|
except Exception:
|
|
pass
|
|
|
|
@property
|
|
def async_session_store(self) -> AsyncSessionStore:
|
|
"""Return the single async facade for this runner's SessionStore."""
|
|
facade = getattr(self, "_async_session_store", None)
|
|
if facade is None or facade._store is not self.session_store:
|
|
facade = AsyncSessionStore(self.session_store)
|
|
self._async_session_store = facade
|
|
return facade
|
|
|
|
|
|
def _get_cached_session_source(self, session_key: str):
|
|
if not session_key:
|
|
return None
|
|
cached_sources = getattr(self, "_session_sources", None)
|
|
if not cached_sources:
|
|
return None
|
|
source = cached_sources.get(session_key)
|
|
if source is not None:
|
|
with suppress(Exception):
|
|
cached_sources.move_to_end(session_key)
|
|
return source
|
|
|
|
@dataclasses.dataclass
|
|
class _HygieneSettings:
|
|
"""Resolved session-hygiene configuration for one inbound turn."""
|
|
|
|
model: str
|
|
threshold_pct: float
|
|
compression_enabled: bool
|
|
hard_msg_limit: int
|
|
timeout_seconds: float
|
|
total_ceiling_seconds: float
|
|
max_turn_hold_seconds: float
|
|
failure_cooldown_seconds: float
|
|
config_context_length: Optional[int]
|
|
provider: Optional[str]
|
|
base_url: Optional[str]
|
|
api_key: Optional[str]
|
|
data: Any
|
|
|
|
@dataclasses.dataclass
|
|
class _HygieneAttempt:
|
|
"""One detached hygiene compression attempt (agent, worker future, commit fence).
|
|
|
|
``cleanup_deferred`` is shared mutable state: the wait handlers set it on their raise
|
|
paths and the owning ``finally`` reads it to decide whether to clean the agent up now.
|
|
"""
|
|
|
|
agent: Any
|
|
meta: Any
|
|
commit_fence: Any = None
|
|
future: Any = None
|
|
wait_started: float = 0.0
|
|
cleanup_deferred: bool = False
|
|
history: Any = None
|
|
|
|
|
|
_TELEGRAM_CAPABILITY_HINT_COOLDOWN_S = 300.0
|
|
|
|
|
|
# Slash-command confirmation primitive (generic): for slash commands with an expensive side
|
|
# effect worth explicit confirmation (currently /reload-mcp, which invalidates the prompt
|
|
# cache). Two delivery paths: adapters overriding ``send_slash_confirm`` render inline buttons
|
|
# and route the click back via ``tools.slash_confirm.resolve(session_key, confirm_id, choice)``;
|
|
# others get a text prompt answered with /approve, /always, or /cancel, matched in
|
|
# ``_handle_message`` against ``tools.slash_confirm.get_pending()``.
|
|
|
|
|
|
def _thread_metadata_for_source(
|
|
self,
|
|
source,
|
|
reply_to_message_id: Optional[str] = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Build the metadata dict platforms need for thread-aware replies."""
|
|
metadata = self._thread_metadata_for_target(
|
|
getattr(source, "platform", None),
|
|
getattr(source, "chat_id", None),
|
|
getattr(source, "thread_id", None),
|
|
chat_type=getattr(source, "chat_type", None),
|
|
reply_to_message_id=reply_to_message_id or getattr(source, "message_id", None),
|
|
)
|
|
if getattr(source, "platform", None) == Platform.SLACK:
|
|
# Per-turn egress identity. Slack's chat.startStream needs recipient_user_id/team_id,
|
|
# which the relay connector fills from metadata.user_id/scope_id; the relay adapter's
|
|
# _with_scope fallback reads both from per-chat caches keyed only by chat_id — mutable
|
|
# state a CONCURRENT turn overwrites (U1's stream opened with U2 as recipient). Stamp
|
|
# this turn's authentic values; _with_scope only fills absent keys, so the cache is
|
|
# reduced to a restart/synthetic-send fallback.
|
|
team_id = getattr(source, "scope_id", None)
|
|
user_id = getattr(source, "user_id", None)
|
|
if team_id or user_id:
|
|
metadata = dict(metadata or {})
|
|
if team_id:
|
|
metadata["slack_team_id"] = str(team_id)
|
|
metadata.setdefault("scope_id", str(team_id))
|
|
if user_id:
|
|
metadata.setdefault("user_id", str(user_id))
|
|
# Routed profile for shared state.db namespaces (#76423): the Telegram
|
|
# prune path needs it because under profile_routes the transport
|
|
# adapter's stamp is not the profile that wrote the binding.
|
|
profile = str(getattr(source, "profile", None) or "").strip()
|
|
if profile and metadata is not None:
|
|
metadata = dict(metadata)
|
|
metadata["hermes_profile"] = profile
|
|
return metadata
|
|
|
|
def _thread_metadata_for_target(
|
|
self,
|
|
platform: Optional[Platform],
|
|
chat_id: Optional[str],
|
|
thread_id: Optional[str],
|
|
*,
|
|
chat_type: Optional[str] = None,
|
|
reply_to_message_id: Optional[str] = None,
|
|
adapter: Optional[Any] = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Build thread metadata for synthetic sends that only have routing state."""
|
|
if thread_id is None:
|
|
return None
|
|
metadata: Dict[str, Any] = {"thread_id": thread_id}
|
|
if self._is_telegram_dm_topic_target(
|
|
platform,
|
|
chat_id,
|
|
thread_id,
|
|
chat_type=chat_type,
|
|
adapter=adapter,
|
|
):
|
|
metadata["telegram_dm_topic_reply_fallback"] = True
|
|
# Telegram DM topic lanes need direct_messages_topic_id in metadata so synthetic/queued
|
|
# messages (goal continuations, status notices) reach the topic without a reply anchor.
|
|
tid = str(thread_id)
|
|
if tid and tid not in {"", "1"}:
|
|
metadata["direct_messages_topic_id"] = tid
|
|
if reply_to_message_id is not None:
|
|
metadata["telegram_reply_to_message_id"] = str(reply_to_message_id)
|
|
if platform == Platform.SLACK and reply_to_message_id is not None:
|
|
# Slack's reply_in_thread=false path uses message_id to distinguish
|
|
# real existing threads from synthetic top-level session keys.
|
|
metadata["message_id"] = str(reply_to_message_id)
|
|
return metadata
|
|
|
|
@staticmethod
|
|
def _is_telegram_dm_topic_target(
|
|
platform: Optional[Platform],
|
|
chat_id: Optional[str],
|
|
thread_id: Optional[str],
|
|
*,
|
|
chat_type: Optional[str] = None,
|
|
adapter: Optional[Any] = None,
|
|
) -> bool:
|
|
"""Return True when a target is a Telegram private DM topic lane."""
|
|
if platform != Platform.TELEGRAM or thread_id is None:
|
|
return False
|
|
if chat_type == "dm":
|
|
return True
|
|
# Inspect operator-declared DM topics via the adapter's lookup. Resolve the method on the
|
|
# CLASS, not the instance: getattr() on a MagicMock auto-creates a callable child for any
|
|
# attribute, so an instance-level lookup would report a DM topic for every test double.
|
|
# Only a dict-shaped return counts as operator-declared — a bare MagicMock must not.
|
|
if adapter is not None and chat_id:
|
|
get_dm_topic_info = getattr(type(adapter), "_get_dm_topic_info", None)
|
|
if callable(get_dm_topic_info):
|
|
try:
|
|
topic_info = get_dm_topic_info(adapter, str(chat_id), str(thread_id))
|
|
except Exception:
|
|
logger.debug("Failed to inspect Telegram DM topic metadata", exc_info=True)
|
|
else:
|
|
return isinstance(topic_info, dict)
|
|
return False
|
|
|
|
@staticmethod
|
|
def _reply_anchor_for_event(event: MessageEvent) -> Optional[str]:
|
|
"""Return the platform-specific reply anchor for GatewayRunner sends."""
|
|
return _reply_anchor_for_event(event)
|
|
|
|
# ------------------------------------------------------------------
|
|
# /approve & /deny — explicit dangerous-command approval
|
|
# ------------------------------------------------------------------
|
|
|
|
_APPROVAL_TIMEOUT_SECONDS = 300 # 5 minutes
|
|
|
|
# Built-in messaging platforms where the ``/update`` command is allowed. ACP, API server, and
|
|
# webhooks are programmatic interfaces that should not trigger system updates. Plugin-migrated
|
|
# platforms are NOT listed here — they declare ``allow_update_command=True`` on their
|
|
# ``PlatformEntry`` and are honored via the registry fallback in ``_handle_update_command``.
|
|
_UPDATE_ALLOWED_PLATFORMS = frozenset({
|
|
Platform.TELEGRAM, Platform.SLACK, Platform.WHATSAPP,
|
|
Platform.SIGNAL, Platform.MATRIX,
|
|
Platform.EMAIL, Platform.SMS, Platform.DINGTALK,
|
|
Platform.FEISHU, Platform.WECOM, Platform.WECOM_CALLBACK, Platform.WEIXIN, Platform.BLUEBUBBLES, Platform.QQBOT, Platform.LOCAL,
|
|
})
|
|
|
|
|
|
def _set_session_env(self, context: SessionContext) -> list:
|
|
"""Set session context variables for the current async task.
|
|
|
|
Uses ``contextvars`` rather than ``os.environ`` so concurrent gateway messages cannot
|
|
overwrite each other's state. Returns reset tokens for ``_clear_session_env`` in a
|
|
``finally`` block.
|
|
"""
|
|
from gateway.session_context import set_session_vars
|
|
# Propagate the adapter's async-delivery capability so async tools (terminal
|
|
# notify_on_complete / watch_patterns, delegate_task background=True) know whether this
|
|
# channel can wake a later turn. Default True keeps CLI/unknown paths working; stateless
|
|
# adapters (api_server) declare False. getattr so bare test runners without self.adapters
|
|
# simply default to supported.
|
|
_adapters = getattr(self, "adapters", None) or {}
|
|
_adapter = _adapters.get(context.source.platform)
|
|
_async_delivery = getattr(_adapter, "supports_async_delivery", True)
|
|
return set_session_vars(
|
|
platform=context.source.platform.value,
|
|
chat_id=context.source.chat_id,
|
|
chat_type=(
|
|
str(context.source.chat_type) if context.source.chat_type else ""
|
|
),
|
|
chat_name=context.source.chat_name or "",
|
|
thread_id=str(context.source.thread_id) if context.source.thread_id else "",
|
|
user_id=str(context.source.user_id) if context.source.user_id else "",
|
|
user_id_alt=str(context.source.user_id_alt) if context.source.user_id_alt else "",
|
|
user_name=str(context.source.user_name) if context.source.user_name else "",
|
|
scope_id=str(getattr(context.source, "scope_id", "") or ""),
|
|
session_key=context.session_key,
|
|
message_id=str(context.source.message_id) if context.source.message_id else "",
|
|
profile=getattr(context.source, "profile", "") or "",
|
|
async_delivery=_async_delivery,
|
|
cron_session="",
|
|
)
|
|
|
|
def _clear_session_env(self, tokens: list) -> None:
|
|
"""Restore session context variables to their pre-handler values."""
|
|
from gateway.session_context import clear_session_vars
|
|
clear_session_vars(tokens)
|
|
|
|
async def _run_in_executor_with_context(self, func, *args):
|
|
"""Run blocking work in the thread pool while preserving session contextvars."""
|
|
loop = asyncio.get_running_loop()
|
|
ctx = copy_context()
|
|
return await loop.run_in_executor(
|
|
self._get_executor(),
|
|
ctx.run,
|
|
func,
|
|
*args,
|
|
)
|
|
|
|
def _get_executor(self) -> concurrent.futures.ThreadPoolExecutor:
|
|
"""Return the gateway-owned executor for blocking agent work."""
|
|
lock = getattr(self, "_executor_lock", None)
|
|
if lock is None:
|
|
lock = threading.Lock()
|
|
self._executor_lock = lock
|
|
|
|
with lock:
|
|
if getattr(self, "_executor_closing", False):
|
|
raise RuntimeError("Gateway is shutting down; executor unavailable")
|
|
executor = getattr(self, "_executor", None)
|
|
if executor is None or getattr(executor, "_shutdown", False):
|
|
executor = concurrent.futures.ThreadPoolExecutor(
|
|
max_workers=10,
|
|
thread_name_prefix="hermes-gateway",
|
|
)
|
|
self._executor = executor
|
|
return executor
|
|
|
|
def _shutdown_executor(self, drain_timeout: float = 0.0) -> int:
|
|
"""Stop the gateway-owned executor without touching the loop default.
|
|
|
|
Returns the number of worker threads still running when this returns.
|
|
With the default ``drain_timeout`` of 0 this is the historical
|
|
fire-and-forget teardown; shutdown passes a bounded budget so blocking
|
|
DB work cannot outlive ``SessionDB.close()`` (see ``_stop_impl``).
|
|
|
|
``cancel_futures`` only drops work that has not started yet, and a
|
|
cancelled ``run_in_executor`` awaitable does not stop the thread behind
|
|
it, so the running futures have to be waited on explicitly.
|
|
"""
|
|
lock = getattr(self, "_executor_lock", None)
|
|
if lock is None:
|
|
return 0
|
|
|
|
with lock:
|
|
self._executor_closing = True
|
|
executor = getattr(self, "_executor", None)
|
|
self._executor = None
|
|
|
|
if executor is None:
|
|
return 0
|
|
|
|
try:
|
|
executor.shutdown(wait=False, cancel_futures=True)
|
|
except TypeError:
|
|
executor.shutdown(wait=False)
|
|
|
|
# ThreadPoolExecutor.shutdown() has no timeout, so join the worker
|
|
# threads directly. `_threads` is absent on the doubles some tests
|
|
# pass in, which just means no wait.
|
|
workers = list(getattr(executor, "_threads", None) or ())
|
|
deadline = time.monotonic() + max(float(drain_timeout or 0.0), 0.0)
|
|
for worker in workers:
|
|
remaining = deadline - time.monotonic()
|
|
if remaining <= 0:
|
|
break
|
|
worker.join(remaining)
|
|
return sum(1 for worker in workers if worker.is_alive())
|
|
|
|
|
|
_MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816)
|
|
|
|
# Config keys whose values MUST invalidate the cached agent when they change: the agent bakes
|
|
# them in at construction, so a mid-gateway edit would otherwise be silently ignored until some
|
|
# other eviction. (section, key) tuples from the raw config dict; add new baked-in settings here.
|
|
_CACHE_BUSTING_CONFIG_KEYS: tuple = (
|
|
("model", "context_length"),
|
|
("model", "max_tokens"),
|
|
("compression", "enabled"),
|
|
("compression", "progress_notices"),
|
|
("compression", "threshold"),
|
|
("compression", "model_thresholds"),
|
|
("compression", "threshold_tokens"),
|
|
("compression", "codex_gpt55_autoraise"),
|
|
("compression", "codex_app_server_auto"),
|
|
("compression", "codex_responses_native"),
|
|
("compression", "codex_responses_compact_threshold"),
|
|
("compression", "in_place"),
|
|
("compression", "checkpoint_required"),
|
|
("compression", "micro_compact"),
|
|
("compression", "micro_compact_every_n_turns"),
|
|
("compression", "micro_compact_defrag_threshold_tokens"),
|
|
("compression", "target_ratio"),
|
|
("compression", "tail_mode"),
|
|
("compression", "protect_last_n"),
|
|
("compression", "proactive_prune_tokens"),
|
|
("compression", "proactive_prune_min_result_chars"),
|
|
("compression", "proactive_prune_min_reclaim_tokens"),
|
|
("compression", "min_tail_user_messages"),
|
|
("agent", "disabled_toolsets"),
|
|
("memory", "provider"),
|
|
("checkpoints", "enabled"),
|
|
("checkpoints", "max_snapshots"),
|
|
("checkpoints", "max_total_size_mb"),
|
|
("checkpoints", "max_file_size_mb"),
|
|
)
|
|
|
|
_HONCHO_CACHE_BUSTING_KEYS = (
|
|
"honcho.peer_name",
|
|
"honcho.ai_peer",
|
|
"honcho.pin_peer_name",
|
|
"honcho.runtime_peer_prefix",
|
|
"honcho.user_peer_aliases",
|
|
)
|
|
_HONCHO_CACHE_BUSTING_MEMO: dict[tuple[str, int | None], dict[str, Any]] = {}
|
|
|
|
|
|
@staticmethod
|
|
def _init_cached_agent_for_turn(agent: Any, interrupt_depth: int) -> None:
|
|
"""Reset per-turn state on a cached agent before a new turn starts.
|
|
|
|
``_last_activity_ts`` / ``_desc`` / ``_provenance`` are a semantic triple, reset together
|
|
and only for fresh external turns (depth 0) — otherwise a session idle 29 min would trip
|
|
the watchdog before the first API call. Interrupt-recursive turns preserve all three so
|
|
the inactivity watchdog can accumulate stuck-turn idle time and fire the 30-min timeout.
|
|
"""
|
|
if interrupt_depth == 0:
|
|
from agent.session_activity import ActivityProvenance
|
|
|
|
agent._last_activity_ts = time.time()
|
|
agent._last_activity_desc = "starting new turn (cached)"
|
|
agent._last_activity_provenance = ActivityProvenance.UNKNOWN
|
|
# Reset the SessionDB flush cursor so the new turn's messages are fully persisted — a stale
|
|
# value from the previous turn makes `_flush_messages_to_session_db` skip new rows.
|
|
if hasattr(agent, "_last_flushed_db_idx"):
|
|
agent._last_flushed_db_idx = 0
|
|
agent._api_call_count = 0
|
|
|
|
|
|
# ---- Proxy mode: forward messages to a remote Hermes API server ----
|
|
|
|
|
|
# ------------------------------------------------------------------
|
|
|
|
|
|
def _profile_name_for_source(self, source: SessionSource) -> Optional[str]:
|
|
"""Resolve the profile name for an inbound source via configured routes.
|
|
|
|
Returns ``None`` (= use the default/active profile) when multiplexing is off, no routes are
|
|
configured, or none match. The most specific matching route wins (guild < channel <
|
|
thread); see :mod:`gateway.profile_routing`. Gated on ``gateway.multiplex_profiles``:
|
|
routing stamps ``source.profile`` but the scoped run only activates under multiplexing,
|
|
else keys would be namespaced by profile while the agent still ran in ``agent:main``.
|
|
"""
|
|
config = getattr(self, "config", None)
|
|
if not getattr(config, "multiplex_profiles", False):
|
|
return None
|
|
routes = getattr(config, "profile_routes", None)
|
|
if not routes:
|
|
return None
|
|
from gateway.profile_routing import ProfileRouteRejected, match_profile_route
|
|
try:
|
|
matched = match_profile_route(
|
|
routes,
|
|
platform=source.platform.value,
|
|
guild_id=getattr(source, "guild_id", None),
|
|
chat_id=source.chat_id,
|
|
thread_id=getattr(source, "thread_id", None),
|
|
parent_chat_id=getattr(source, "parent_chat_id", None),
|
|
)
|
|
except Exception:
|
|
logger.warning(
|
|
"Profile route matching failed for %s/%s, falling back to default",
|
|
source.platform, source.chat_id, exc_info=True,
|
|
)
|
|
return None
|
|
if matched:
|
|
try:
|
|
served = {name for name, _home in _multiplex_profile_homes(config)}
|
|
except Exception as exc:
|
|
logger.warning(
|
|
"Rejecting profile route %r because the served-profile set "
|
|
"could not be resolved",
|
|
matched.name,
|
|
exc_info=True,
|
|
)
|
|
raise ProfileRouteRejected(matched.name) from exc
|
|
if matched.profile not in served:
|
|
logger.warning(
|
|
"Rejecting profile route %r: target profile %r is not served",
|
|
matched.name,
|
|
matched.profile,
|
|
)
|
|
raise ProfileRouteRejected(matched.name)
|
|
return matched.profile
|
|
logger.debug(
|
|
"No profile route matched: platform=%s chat_id=%s thread_id=%s parent_chat_id=%s",
|
|
source.platform.value, source.chat_id,
|
|
getattr(source, "thread_id", None), getattr(source, "parent_chat_id", None),
|
|
)
|
|
return None
|
|
|
|
def _resolve_profile_home_for_source(self, source: SessionSource) -> "Path":
|
|
"""Resolve which profile's HERMES_HOME should serve this inbound source.
|
|
|
|
Order: ``source.profile`` (URL prefix, adapter ownership, or profile_routes at
|
|
``build_source``), then ``_profile_name_for_source`` (fallback for sources bypassing
|
|
``build_source``), then the active profile.
|
|
"""
|
|
from gateway.profile_routing import ProfileRouteRejected
|
|
from hermes_cli.profiles import (
|
|
get_active_profile_name,
|
|
get_profile_dir,
|
|
profile_exists,
|
|
)
|
|
from hermes_constants import get_hermes_home
|
|
|
|
# Track whether a profile was explicitly requested (vs. falling back to default)
|
|
explicit_profile = None
|
|
try:
|
|
name = (source.profile or "").strip()
|
|
if name:
|
|
explicit_profile = name # User explicitly set this profile
|
|
if not name:
|
|
name = self._profile_name_for_source(source)
|
|
if name:
|
|
explicit_profile = name # Routing explicitly set this profile
|
|
if not name:
|
|
name = get_active_profile_name() or "default"
|
|
|
|
profile_dir = get_profile_dir(name)
|
|
# Warn if an explicit profile doesn't exist on disk
|
|
if explicit_profile and not profile_exists(name):
|
|
logger.warning(
|
|
"Profile %r does not exist for source %s/%s (guild_id=%s), "
|
|
"falling back to global HERMES_HOME",
|
|
explicit_profile,
|
|
source.platform.value,
|
|
source.chat_id,
|
|
getattr(source, "guild_id", None),
|
|
)
|
|
return get_hermes_home()
|
|
return profile_dir
|
|
except ProfileRouteRejected:
|
|
raise
|
|
except Exception:
|
|
# Catch normalization errors, path errors, etc.
|
|
logger.warning(
|
|
"Failed to resolve profile directory for source %s/%s (guild_id=%s), "
|
|
"falling back to global HERMES_HOME: %s",
|
|
source.platform.value,
|
|
source.chat_id,
|
|
getattr(source, "guild_id", None),
|
|
explicit_profile or "(no profile)",
|
|
exc_info=True,
|
|
)
|
|
return get_hermes_home()
|
|
|
|
@dataclasses.dataclass
|
|
class _RunAgentDisplay:
|
|
"""Per-turn display / progress settings resolved by ``_run_agent_display_settings``."""
|
|
|
|
user_config: Any = None
|
|
platform_key: Any = None
|
|
enabled_toolsets: Any = None
|
|
disabled_toolsets: Any = None
|
|
resolve_display_setting: Any = None
|
|
progress_mode: Any = None
|
|
progress_grouping: Any = None
|
|
_display_surface_mode: Any = None
|
|
tool_progress_enabled: Any = None
|
|
_live_status_mode: Any = None
|
|
_live_status_adapter: Any = None
|
|
log_mode_enabled: Any = None
|
|
log_queue: Any = None
|
|
interim_assistant_messages_enabled: Any = None
|
|
_thinking_enabled: Any = None
|
|
_native_slack_task_cards: Any = None
|
|
needs_progress_queue: Any = None
|
|
_generic_status_phrase: Any = None
|
|
|
|
@dataclasses.dataclass
|
|
class _RunAgentWorker:
|
|
"""Executor future + inactivity-watchdog handles for one ``_run_agent_inner`` turn."""
|
|
|
|
executor_task: Any = None
|
|
agent_timeout: Optional[float] = None
|
|
agent_warning: Optional[float] = None
|
|
task_id: str = ""
|
|
process_baseline: Any = None
|
|
worker_done: Any = None
|
|
timeout_fired: Any = None
|
|
cleanup_lock: Any = None
|
|
is_current: Any = None
|
|
|
|
|
|
def _run_planned_stop_watcher(
|
|
stop_event: threading.Event,
|
|
runner,
|
|
loop: asyncio.AbstractEventLoop,
|
|
shutdown_handler,
|
|
*,
|
|
poll_interval: float = 0.5,
|
|
) -> None:
|
|
"""Poll for the planned-stop marker and trigger graceful shutdown.
|
|
|
|
On Windows ``asyncio.add_signal_handler`` raises NotImplementedError, so ``hermes gateway
|
|
stop`` never reaches the signal-driven drain: sessions die mid-turn and ``resume_pending``
|
|
is never set. This watcher (cheap, runs on every platform) translates the marker written by
|
|
``hermes_cli.gateway_windows.stop()`` into the same shutdown-handler call a SIGTERM would; on
|
|
POSIX the synchronous signal handler consumes the marker first. ``_running`` / ``_draining``
|
|
guard against re-triggering; the handler tolerates ``signal=None``.
|
|
"""
|
|
from gateway.status import (
|
|
_get_planned_stop_marker_path,
|
|
planned_stop_marker_targets_self,
|
|
)
|
|
marker_path = _get_planned_stop_marker_path()
|
|
while not stop_event.is_set():
|
|
try:
|
|
if (
|
|
marker_path.exists()
|
|
and not getattr(runner, "_draining", False)
|
|
and getattr(runner, "_running", False)
|
|
):
|
|
# A marker existing is NOT sufficient — it may target a PREVIOUS gateway instance
|
|
# (different PID) left behind when that process exited before stop() cleaned up.
|
|
# Firing on it drives us into shutdown, an "UNKNOWN" exit, and a watchdog crash-loop.
|
|
# Only fire when the marker targets us; the probe unlinks stale/malformed markers.
|
|
if not planned_stop_marker_targets_self():
|
|
stop_event.wait(poll_interval)
|
|
continue
|
|
# Same path as a real signal handler. signal=None is tolerated; the handler consumes the
|
|
# marker via consume_planned_stop_marker_for_self (validates target_pid + start_time).
|
|
loop.call_soon_threadsafe(shutdown_handler, None)
|
|
# Done — the handler will set _draining; we exit on next tick.
|
|
break
|
|
except Exception as _e:
|
|
logger.debug("Planned-stop watcher tick error: %s", _e)
|
|
stop_event.wait(poll_interval)
|
|
|
|
|
|
def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop=None, interval: int = 60, cron_provider=None):
|
|
"""Background thread for gateway-only periodic chores (NOT cron).
|
|
|
|
Separate from the cron trigger so chores run regardless of which ``CronScheduler`` provider
|
|
fires cron (an external scale-to-zero provider has no 60s loop). Refreshes the channel
|
|
directory every 5 min; prunes media caches + expired share pastes hourly; polls the curator
|
|
hourly (its inner gate enforces the weekly cadence).
|
|
"""
|
|
from gateway.platforms.base import (
|
|
cleanup_audio_cache,
|
|
cleanup_document_cache,
|
|
cleanup_image_cache,
|
|
cleanup_screenshot_cache,
|
|
cleanup_video_cache,
|
|
)
|
|
from tools.tool_result_storage import cleanup_spillover_cache
|
|
from tools.environments.local import cleanup_terminal_temp_cache
|
|
from tools.bot_mode_dm import cleanup_bot_dm_cache
|
|
from tools.bot_relay import cleanup_bot_relay_artifacts
|
|
from hermes_cli.debug import _sweep_expired_pastes
|
|
|
|
IMAGE_CACHE_EVERY = 60 # ticks — once per hour at default 60s interval
|
|
CHANNEL_DIR_EVERY = 5 # ticks — every 5 minutes
|
|
PASTE_SWEEP_EVERY = 60 # ticks — once per hour
|
|
CURATOR_EVERY = 60 # ticks — poll hourly (inner gate handles the real cadence)
|
|
AUTO_ARCHIVE_EVERY = 60 # ticks — poll hourly (state_meta gate owns the real cadence)
|
|
MEMORY_TRIM_EVERY = 1 # shared helper cooldown bounds actual allocator work
|
|
MISFIRE_SWEEP_EVERY = 5 # ticks — every 5 minutes (grace window gates real work)
|
|
FTS_STALE_RETRY_EVERY = 1 # SessionDB rate-limits the real work (_FTS_STALE_RETRY_SECONDS)
|
|
|
|
# Every platform media cache prunes on the same hourly cadence — one loop
|
|
# over (name, cleanup_fn), not a copy-pasted try/except per cache.
|
|
MEDIA_CACHE_CLEANUPS = (
|
|
("Image", cleanup_image_cache),
|
|
("Document", cleanup_document_cache),
|
|
("Audio", cleanup_audio_cache),
|
|
("Video", cleanup_video_cache),
|
|
("Screenshot", cleanup_screenshot_cache),
|
|
("Spillover", cleanup_spillover_cache),
|
|
("Terminal temp", cleanup_terminal_temp_cache),
|
|
("Bot DM", cleanup_bot_dm_cache),
|
|
("Bot relay", cleanup_bot_relay_artifacts),
|
|
)
|
|
|
|
logger.info("Gateway housekeeping started (interval=%ds)", interval)
|
|
tick_count = 0
|
|
while not stop_event.is_set():
|
|
tick_count += 1
|
|
|
|
if tick_count % CHANNEL_DIR_EVERY == 0 and adapters:
|
|
try:
|
|
from gateway.channel_directory import build_channel_directory
|
|
if loop is not None:
|
|
# build_channel_directory is async (Slack web calls) and this is a background thread:
|
|
# schedule onto the gateway loop and wait briefly so refresh failures still log.
|
|
fut = safe_schedule_threadsafe(
|
|
build_channel_directory(adapters), loop,
|
|
logger=logger,
|
|
log_message="Channel directory refresh scheduling error",
|
|
)
|
|
if fut is not None:
|
|
fut.result(timeout=30)
|
|
except Exception as e:
|
|
logger.debug("Channel directory refresh error: %s", e)
|
|
|
|
if tick_count % IMAGE_CACHE_EVERY == 0:
|
|
for cache_name, cleanup_fn in MEDIA_CACHE_CLEANUPS:
|
|
try:
|
|
removed = cleanup_fn(max_age_hours=24)
|
|
if removed:
|
|
logger.info("%s cache cleanup: removed %d stale file(s)", cache_name, removed)
|
|
except Exception as e:
|
|
logger.debug("%s cache cleanup error: %s", cache_name, e)
|
|
|
|
if tick_count % PASTE_SWEEP_EVERY == 0:
|
|
try:
|
|
deleted, remaining = _sweep_expired_pastes()
|
|
if deleted:
|
|
logger.info(
|
|
"Paste sweep: deleted %d expired paste(s), %d pending",
|
|
deleted, remaining,
|
|
)
|
|
except Exception as e:
|
|
logger.debug("Paste sweep error: %s", e)
|
|
|
|
# Misfire catch-up (external cron providers only): fire jobs whose scheduled time passed
|
|
# with no external fire delivered (dead loopback hop: restart window, api_server not bound,
|
|
# retries exhausted). No-op for the built-in ticker; enforces cron.misfire_grace_minutes;
|
|
# the store CAS claim de-dupes against a late external retry.
|
|
if cron_provider is not None and tick_count % MISFIRE_SWEEP_EVERY == 0:
|
|
try:
|
|
from cron.scheduler_provider import fire_overdue_jobs
|
|
|
|
caught_up = fire_overdue_jobs(
|
|
cron_provider, adapters=adapters, loop=loop
|
|
)
|
|
if caught_up:
|
|
logger.info(
|
|
"Misfire catch-up: fired %d overdue job(s)", caught_up
|
|
)
|
|
except Exception as e:
|
|
logger.debug("Misfire catch-up sweep error: %s", e)
|
|
|
|
# Curator — piggy-back on housekeeping so long-running gateways get weekly skill maintenance
|
|
# without restarts. maybe_run_curator() is gated by config.interval_hours (7 days default), so
|
|
# CURATOR_EVERY is just the poll rate.
|
|
if tick_count % CURATOR_EVERY == 0:
|
|
try:
|
|
from agent.curator import maybe_run_curator
|
|
maybe_run_curator(
|
|
idle_for_seconds=float("inf"),
|
|
on_summary=lambda msg: logger.info("curator: %s", msg),
|
|
)
|
|
except Exception as e:
|
|
logger.debug("Curator tick error: %s", e)
|
|
|
|
# Skill Sync — best-effort periodic pull on the same cadence; inert unless the access gate is
|
|
# open and a sync base URL is configured; never raises.
|
|
try:
|
|
from tools.skills_sync_client import maybe_pull_skills
|
|
maybe_pull_skills()
|
|
except Exception as e:
|
|
logger.debug("Sync pull tick error: %s", e)
|
|
|
|
# Org-shared skills. Gated on real org membership (the token must
|
|
# carry an org role), so a solo account never reaches the network.
|
|
try:
|
|
from tools.skills_sync_client import maybe_pull_org_skills
|
|
maybe_pull_org_skills()
|
|
except Exception as e:
|
|
logger.debug("Org sync pull tick error: %s", e)
|
|
|
|
# Stale-session auto-archive on a live timer so long-running gateways keep sweeping (the
|
|
# startup hook fires once). maybe_auto_archive() is gated by sessions.min_interval_hours;
|
|
# this is just the poll rate. Opens its own SessionDB — SQLite connections are thread-bound.
|
|
if tick_count % AUTO_ARCHIVE_EVERY == 0:
|
|
try:
|
|
from hermes_cli.config import load_config as _load_full_config
|
|
from hermes_state import get_shared_session_db
|
|
_sess_cfg = (_load_full_config().get("sessions") or {})
|
|
if _sess_cfg.get("auto_archive", False):
|
|
_adb = get_shared_session_db()
|
|
try:
|
|
_adb.maybe_auto_archive(
|
|
idle_days=float(_sess_cfg.get("auto_archive_days", 3)),
|
|
min_interval_hours=int(_sess_cfg.get("min_interval_hours", 24)),
|
|
)
|
|
finally:
|
|
from hermes_state import release_or_close
|
|
release_or_close(_adb)
|
|
except Exception as e:
|
|
logger.debug("Auto-archive tick error: %s", e)
|
|
|
|
# Deferred stale-FTS rebuild retry: a SessionDB opened while another process held state.db
|
|
# or the rebuild lock fails closed onto the LIKE fallback, and the gateway stays up for days.
|
|
# Non-blocking admission, no new thread, rate-limited inside SessionDB; no-op when not stale.
|
|
if tick_count % FTS_STALE_RETRY_EVERY == 0:
|
|
try:
|
|
from hermes_state_registry import live_shared_session_dbs
|
|
|
|
for _sdb in live_shared_session_dbs():
|
|
_retry = getattr(_sdb, "retry_deferred_fts_recovery", None)
|
|
if callable(_retry) and _retry():
|
|
logger.info(
|
|
"Deferred state.db FTS rebuild completed in-process "
|
|
"for %s; full-text search restored.",
|
|
getattr(_sdb, "db_path", "state.db"),
|
|
)
|
|
except Exception as exc:
|
|
logger.debug("Deferred FTS retry tick error: %s", exc)
|
|
|
|
# Long-lived messaging-gateway counterpart to the TUI idle reaper; the helper is config-gated
|
|
# and rate-limited, so the 60s housekeeping cadence creates no trim storm.
|
|
if tick_count % MEMORY_TRIM_EVERY == 0:
|
|
try:
|
|
from hermes_cli.mem_trim import trim_memory
|
|
|
|
trim_memory(reason="messaging gateway housekeeping")
|
|
except Exception as exc:
|
|
# debug, not warning: sibling branches log failures at debug, and a persistent failure
|
|
# (e.g. broken import after a partial update) would otherwise warn every 60s forever.
|
|
logger.debug(
|
|
"gateway housekeeping memory trim failed: %s: %s",
|
|
type(exc).__name__,
|
|
exc,
|
|
)
|
|
|
|
stop_event.wait(timeout=interval)
|
|
logger.info("Gateway housekeeping stopped")
|
|
|
|
|
|
def _start_cron_ticker(stop_event: threading.Event, adapters=None, loop=None, interval: int = 60):
|
|
"""DEPRECATED shim — runs ONLY the built-in in-process cron tick loop.
|
|
|
|
The trigger now lives behind the ``CronScheduler`` provider (``cron.scheduler_provider``,
|
|
started in ``start_gateway``); housekeeping moved to ``_start_gateway_housekeeping``.
|
|
"""
|
|
from cron.scheduler_provider import InProcessCronScheduler
|
|
InProcessCronScheduler().start(stop_event, adapters=adapters, loop=loop, interval=interval)
|
|
|
|
|
|
def _stop_cron_provider(provider) -> None:
|
|
"""Stop a cron provider without letting it choose the gateway exit code."""
|
|
try:
|
|
provider.stop()
|
|
except SystemExit as exc:
|
|
logger.warning(
|
|
"Cron provider stop() attempted to exit the gateway with code %s; ignoring",
|
|
exc.code,
|
|
)
|
|
except Exception as exc:
|
|
logger.debug("Cron provider stop() error: %s", exc)
|
|
|
|
|
|
# Upper bound for cooperatively draining the cron ticker on shutdown: the cron thread blocks on
|
|
# ``future.result(timeout=60)`` (cron/scheduler.py::_deliver_result), so a delivery unblocks in ~60s.
|
|
_CRON_SHUTDOWN_DRAIN_TIMEOUT = 65.0
|
|
|
|
# Upper bound for draining the housekeeping ticker on shutdown: the channel-directory refresh blocks
|
|
# on ``fut.result(timeout=30)``, so cover that 30s plus margin or an in-flight refresh is abandoned.
|
|
_HOUSEKEEPING_SHUTDOWN_DRAIN_TIMEOUT = 35.0
|
|
|
|
|
|
async def _await_thread_exit(
|
|
thread: Optional[threading.Thread], timeout: float, poll: float = 0.1
|
|
) -> bool:
|
|
"""Wait for a daemon thread to exit WITHOUT blocking the event loop; True if it exited in time.
|
|
|
|
A synchronous ``join()`` freezes the loop — fatal for the cron ticker, whose in-flight delivery
|
|
is a coroutine scheduled onto *this* loop via ``safe_schedule_threadsafe``: it could never run,
|
|
so the join always timed out and the message was dropped. Polling ``is_alive()`` with
|
|
``await asyncio.sleep`` lets the delivery complete and the ticker see ``stop_event``.
|
|
"""
|
|
if thread is None:
|
|
return True
|
|
deadline = asyncio.get_running_loop().time() + max(0.0, timeout)
|
|
while thread.is_alive() and asyncio.get_running_loop().time() < deadline:
|
|
await asyncio.sleep(poll)
|
|
return not thread.is_alive()
|
|
|
|
|
|
async def _shutdown_mcp_servers_nonblocking(timeout: float = 5.0) -> bool:
|
|
"""Close MCP servers off-loop with a bounded wait; True when done within ``timeout``.
|
|
|
|
``shutdown_mcp_servers()`` is synchronous and can block ~15s when the MCP loop and stdio
|
|
children are torn down concurrently. On the loop thread that freezes the loop, so supervisors
|
|
with a short kill grace (s6 default 3s) SIGKILL us before ``lifecycle_ledger.mark_exited()``
|
|
runs and every later boot reports a phantom unclean death. Runs on a daemon thread, polled via
|
|
``_await_thread_exit``; on timeout shutdown proceeds and the thread is left to finish or die.
|
|
"""
|
|
|
|
def _do() -> None:
|
|
try:
|
|
from tools.mcp_tool import shutdown_mcp_servers
|
|
|
|
shutdown_mcp_servers()
|
|
except Exception:
|
|
logger.debug("MCP shutdown raised", exc_info=True)
|
|
|
|
thread = threading.Thread(target=_do, name="mcp-shutdown", daemon=True)
|
|
thread.start()
|
|
done = await _await_thread_exit(thread, timeout=timeout)
|
|
if not done:
|
|
logger.warning(
|
|
"MCP shutdown did not finish within %.1fs; continuing gateway "
|
|
"teardown (background thread will be reaped at process exit)",
|
|
timeout,
|
|
)
|
|
return done
|
|
|
|
|
|
def _shutdown_gateway_health_export(runner: Any) -> None:
|
|
"""Idempotently drain and detach Gateway Health OTLP export."""
|
|
runtime = getattr(runner, "_gateway_health_export_runtime", None)
|
|
if runtime is None:
|
|
return
|
|
runner._gateway_health_export_runtime = None
|
|
try:
|
|
runtime.shutdown()
|
|
except Exception:
|
|
logger.debug("gateway health OTLP export shutdown failed", exc_info=True)
|
|
|
|
|
|
def _gateway_stderr_formatter() -> logging.Formatter:
|
|
"""Return the redacting formatter used by the gateway stderr stream."""
|
|
from agent.redact import RedactingFormatter
|
|
|
|
return RedactingFormatter("%(asctime)s %(levelname)s %(name)s: %(message)s")
|
|
|
|
# ownership guard inserted below (PR #93084)
|
|
def _replace_target_belongs_to_other_profile(existing_pid: int) -> bool:
|
|
"""Return True when ``--replace`` must refuse to signal ``existing_pid``.
|
|
|
|
A poisoned/stale PID record can point at another profile's LIVE gateway; signaling it starts
|
|
a cross-profile SIGTERM restart loop. Ownership is decided by the persisted identity record
|
|
ALONE (exact ``_same_hermes_home``), and only while it stays bound to the live target by exact
|
|
PID + start-time identity. Live argv carries no HERMES_HOME so it can never PROVE ownership;
|
|
it is only a consistency check (contradicting profile flags refuse). Missing, legacy,
|
|
conflicting, stale-bound or unprovable identity → refuse (fail closed).
|
|
"""
|
|
try:
|
|
from gateway.status import (
|
|
_get_pid_path,
|
|
_get_process_hermes_home,
|
|
_get_process_start_time,
|
|
_pid_from_record,
|
|
_read_pid_record,
|
|
_record_looks_like_gateway,
|
|
_read_process_cmdline,
|
|
_same_hermes_home,
|
|
)
|
|
|
|
our_home = _get_process_hermes_home()
|
|
|
|
# Authorize from the persisted identity record — bound claim: the record must describe THIS pid
|
|
# with THIS live start time, otherwise it is stale/poisoned and proves nothing.
|
|
record = _read_pid_record(_get_pid_path())
|
|
if not isinstance(record, dict) or not _record_looks_like_gateway(record):
|
|
logger.warning(
|
|
"Refusing --replace: no valid gateway pid record to prove "
|
|
"ownership of PID %s.",
|
|
existing_pid,
|
|
)
|
|
return True
|
|
|
|
record_pid = _pid_from_record(record)
|
|
if record_pid != existing_pid:
|
|
logger.warning(
|
|
"Refusing --replace: pid record names %s, not target %s.",
|
|
record_pid, existing_pid,
|
|
)
|
|
return True
|
|
|
|
recorded_start = record.get("start_time")
|
|
if not isinstance(recorded_start, int) or isinstance(recorded_start, bool):
|
|
return True
|
|
if _get_process_start_time(existing_pid) != recorded_start:
|
|
logger.warning(
|
|
"Refusing --replace: pid record start-time does not match "
|
|
"the live process %s (stale/PID-reuse record).",
|
|
existing_pid,
|
|
)
|
|
return True
|
|
|
|
recorded_home = record.get("hermes_home")
|
|
if not isinstance(recorded_home, str) or not recorded_home.strip():
|
|
# Legacy record without hermes_home cannot prove ownership.
|
|
logger.warning(
|
|
"Refusing --replace: pid record predates hermes_home "
|
|
"stampings; ownership of PID %s unprovable.",
|
|
existing_pid,
|
|
)
|
|
return True
|
|
|
|
if not _same_hermes_home(recorded_home, our_home):
|
|
logger.error(
|
|
"Refusing --replace: pid record belongs to a different "
|
|
"HERMES_HOME (%s, ours %s). Remove the stale PID record or "
|
|
"stop the owning profile explicitly.",
|
|
recorded_home,
|
|
our_home,
|
|
)
|
|
return True
|
|
|
|
# Readable-argv consistency check (never authority): an explicit profile flag / HERMES_HOME= that
|
|
# clearly contradicts our home refuses even if the record agreed; bare/matching argv adds nothing.
|
|
try:
|
|
live_cmdline = _read_process_cmdline(existing_pid)
|
|
except Exception:
|
|
live_cmdline = None # consistency probe failure → record decides
|
|
if live_cmdline and _looks_like_profile_conflict_from_cmdline(
|
|
live_cmdline, our_home
|
|
):
|
|
logger.error(
|
|
"Refusing --replace: target PID %s command line explicitly "
|
|
"advertises a different profile than HERMES_HOME %s.",
|
|
existing_pid,
|
|
our_home,
|
|
)
|
|
return True
|
|
|
|
return False
|
|
except Exception:
|
|
# Destructive action + unknown ownership => fail closed (#89315).
|
|
logger.warning(
|
|
"cross-profile --replace ownership probe failed for PID %s; "
|
|
"refusing to signal",
|
|
existing_pid,
|
|
exc_info=True,
|
|
)
|
|
return True
|
|
|
|
|
|
def _looks_like_profile_conflict_from_cmdline(command: str, our_home) -> bool:
|
|
"""Token-exact contradiction check between a target argv and our home.
|
|
|
|
Authority lives in the pid record; this only catches argv that EXPLICITLY advertises another
|
|
profile. Substring matching is not identity: ``--profile timothy`` must NOT read as profile
|
|
``tim``. Returns False whenever the argv does not clearly contradict our home.
|
|
"""
|
|
from gateway.status import _profile_name_for_home
|
|
|
|
profile_name = _profile_name_for_home(our_home)
|
|
try:
|
|
tokens = shlex.split(command)
|
|
except ValueError:
|
|
tokens = command.split()
|
|
|
|
def _flag_value(flag: str) -> Optional[str]:
|
|
"""Value of ``--flag X`` / ``--flag=X`` occurrences, token-exact."""
|
|
values = []
|
|
i = 0
|
|
while i < len(tokens):
|
|
tok = tokens[i]
|
|
if tok == flag and i + 1 < len(tokens):
|
|
values.append(tokens[i + 1])
|
|
i += 2
|
|
continue
|
|
if tok.startswith(flag + "="):
|
|
values.append(tok[len(flag) + 1:])
|
|
i += 1
|
|
return values[-1] if values else None
|
|
|
|
def _env_home_value() -> Optional[str]:
|
|
"""HERMES_HOME=<path> env-style assignment on the argv, token-exact."""
|
|
prefix = "HERMES_HOME="
|
|
for tok in reversed(tokens):
|
|
if tok.startswith(prefix):
|
|
return tok[len(prefix):]
|
|
return None
|
|
|
|
if profile_name is not None and profile_name != "default":
|
|
# Our home is a named profile: any explicit DIFFERENT named profile on the argv contradicts it;
|
|
# bare argv stays consistent (legacy default-gateway argv never carried profile flags).
|
|
for flag in ("--profile", "-p"):
|
|
value = _flag_value(flag)
|
|
if value is not None and value != profile_name:
|
|
return True
|
|
home_value = _flag_value("--hermes-home") or _env_home_value()
|
|
return bool(home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home))))
|
|
|
|
# Our home is the default/root: ANY explicit named-profile flag on the
|
|
# argv contradicts it.
|
|
if _flag_value("--profile") is not None or _flag_value("-p") is not None:
|
|
return True
|
|
home_value = _flag_value("--hermes-home") or _env_home_value()
|
|
return bool(home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home))))
|
|
|
|
|
|
async def _start_gateway_replace_existing_instance(existing_pid: int, replace: bool) -> bool:
|
|
"""Handle a live gateway PID under this HERMES_HOME: replace it (``--replace``) or refuse.
|
|
|
|
Returns False when startup must abort (refused, permission denied, target still alive).
|
|
"""
|
|
from gateway.status import (
|
|
get_process_start_time,
|
|
remove_pid_file,
|
|
terminate_pid,
|
|
)
|
|
if replace:
|
|
# Cross-profile ownership gate: never signal a live process we cannot prove belongs to
|
|
# this HERMES_HOME. A poisoned PID record steering --replace at another profile's
|
|
# gateway is exactly the restart-loop shape this flow must not allow.
|
|
if _replace_target_belongs_to_other_profile(existing_pid):
|
|
from gateway.status import _get_process_hermes_home
|
|
|
|
logger.error(
|
|
"Refusing --replace: PID %d cannot be proven to belong "
|
|
"to this profile's gateway (HERMES_HOME %s). Remove the "
|
|
"stale PID record or stop the owning profile explicitly.",
|
|
existing_pid,
|
|
_get_process_hermes_home(),
|
|
)
|
|
return False
|
|
existing_start_time = get_process_start_time(existing_pid)
|
|
logger.info(
|
|
"Replacing existing gateway instance (PID %d) with --replace.",
|
|
existing_pid,
|
|
)
|
|
# Record a takeover marker so the target's shutdown handler recognises its SIGTERM as a
|
|
# planned takeover and exits 0 (rather than exit 1, which would trigger systemd's
|
|
# Restart=on-failure and start a flap loop against us). Best-effort — proceed on failure.
|
|
try:
|
|
from gateway.status import write_takeover_marker
|
|
write_takeover_marker(existing_pid)
|
|
except Exception as e:
|
|
logger.debug("Could not write takeover marker: %s", e)
|
|
# Snapshot the old gateway's children BEFORE signalling it: once it exits, orphans are
|
|
# reparented and invisible to a parent walk. On POSIX, surviving adapter subprocesses hold
|
|
# scoped token locks and block the replacement (Windows already tree-kills). Best-effort.
|
|
try:
|
|
from gateway.status import _snapshot_gateway_children
|
|
_old_gateway_children = _snapshot_gateway_children(existing_pid)
|
|
except Exception:
|
|
_old_gateway_children = []
|
|
try:
|
|
terminate_pid(existing_pid, force=False)
|
|
except ProcessLookupError:
|
|
pass # Already gone
|
|
except (PermissionError, OSError):
|
|
logger.error(
|
|
"Permission denied killing PID %d. Cannot replace.",
|
|
existing_pid,
|
|
)
|
|
# Marker is scoped to a specific target; clean it up on
|
|
# give-up so it doesn't grief an unrelated future shutdown.
|
|
try:
|
|
from gateway.status import clear_takeover_marker
|
|
clear_takeover_marker()
|
|
except Exception:
|
|
pass
|
|
return False
|
|
# Wait up to 10s for the old process to exit. ``os.kill(pid, 0)`` on Windows is NOT a no-op —
|
|
# use the handle-based existence check instead.
|
|
from gateway.status import _pid_exists
|
|
old_gateway_exited = False
|
|
for _ in range(20):
|
|
if not _pid_exists(existing_pid):
|
|
old_gateway_exited = True
|
|
break # Process is gone
|
|
# start_gateway is async: a blocking sleep here freezes the event loop (signal handlers,
|
|
# health checks, every coroutine) for up to 10s per replacement.
|
|
await asyncio.sleep(0.5)
|
|
else:
|
|
# Still alive after 10s — force kill
|
|
logger.warning(
|
|
"Old gateway (PID %d) did not exit after SIGTERM, sending SIGKILL.",
|
|
existing_pid,
|
|
)
|
|
try:
|
|
terminate_pid(
|
|
existing_pid,
|
|
force=True,
|
|
expected_start_time=existing_start_time,
|
|
)
|
|
except ProcessLookupError:
|
|
old_gateway_exited = True
|
|
except (PermissionError, OSError):
|
|
pass
|
|
# Confirm the force-kill actually reaped the process before clearing its PID file /
|
|
# scoped locks: SIGKILL can fail to take (uninterruptible sleep, zombie), and blindly
|
|
# clearing metadata would leave two live gateways fighting over the same token.
|
|
if not old_gateway_exited:
|
|
for _ in range(20):
|
|
if not _pid_exists(existing_pid):
|
|
old_gateway_exited = True
|
|
break
|
|
# Async context — never block the loop (#36163).
|
|
await asyncio.sleep(0.25)
|
|
if not old_gateway_exited:
|
|
logger.error(
|
|
"Old gateway (PID %d) still appears alive after SIGKILL; "
|
|
"aborting replacement to avoid a duplicate gateway.",
|
|
existing_pid,
|
|
)
|
|
try:
|
|
from gateway.status import clear_takeover_marker
|
|
clear_takeover_marker()
|
|
except Exception:
|
|
pass
|
|
return False
|
|
# Old gateway confirmed dead — reap any orphaned child processes it left behind (POSIX;
|
|
# mirrors Windows taskkill /T tree-kill). Orphaned adapter subprocesses would otherwise
|
|
# keep holding scoped token locks against us. Best-effort, never raises.
|
|
try:
|
|
from gateway.status import reap_gateway_children
|
|
reap_gateway_children(
|
|
_old_gateway_children, parent_pid=existing_pid
|
|
)
|
|
except Exception:
|
|
logger.debug(
|
|
"Child reap for replaced gateway PID %d failed",
|
|
existing_pid,
|
|
exc_info=True,
|
|
)
|
|
remove_pid_file()
|
|
# remove_pid_file() is a no-op when the PID doesn't match.
|
|
# Force-unlink to cover the old-process-crashed case.
|
|
with suppress(Exception):
|
|
(get_hermes_home() / "gateway.pid").unlink(missing_ok=True)
|
|
# Clean up any takeover marker the old process didn't consume
|
|
# (e.g. SIGKILL'd before its shutdown handler could read it).
|
|
try:
|
|
from gateway.status import clear_takeover_marker
|
|
clear_takeover_marker()
|
|
except Exception:
|
|
pass
|
|
# Release all scoped locks left by the old process: stopped (Ctrl+Z) processes don't release
|
|
# locks on exit, leaving stale lock files that block the new gateway.
|
|
try:
|
|
from gateway.status import release_all_scoped_locks
|
|
_released = release_all_scoped_locks(
|
|
owner_pid=existing_pid,
|
|
owner_start_time=existing_start_time,
|
|
)
|
|
if _released:
|
|
logger.info("Released %d stale scoped lock(s) from old gateway.", _released)
|
|
except Exception:
|
|
pass
|
|
else:
|
|
hermes_home = str(get_hermes_home())
|
|
logger.error(
|
|
"Another gateway instance is already running (PID %d, HERMES_HOME=%s). "
|
|
"Use 'hermes gateway restart' to replace it, or 'hermes gateway stop' first.",
|
|
existing_pid, hermes_home,
|
|
)
|
|
print(
|
|
f"\n❌ Gateway already running (PID {existing_pid}).\n"
|
|
f" Use 'hermes gateway restart' to replace it,\n"
|
|
f" or 'hermes gateway stop' to kill it first.\n"
|
|
f" Or use 'hermes gateway run --replace' to auto-replace.\n"
|
|
)
|
|
return False
|
|
return True
|
|
|
|
|
|
def _start_gateway_configure_logging(verbosity: Optional[int]) -> None:
|
|
"""Sync bundled skills, set up file logging + startup security audit, and the -v/-q stderr handler."""
|
|
# Sync bundled skills on gateway start (fast -- skips unchanged)
|
|
try:
|
|
from tools.skills_sync import sync_skills
|
|
sync_skills(quiet=True)
|
|
except Exception:
|
|
pass
|
|
|
|
# Centralized logging — agent.log (INFO+), errors.log (WARNING+), gateway.log (INFO+, gateway
|
|
# records only). Idempotent, so repeated calls from AIAgent.__init__ don't duplicate.
|
|
from hermes_logging import setup_logging, _safe_stderr
|
|
setup_logging(hermes_home=_hermes_home, mode="gateway")
|
|
|
|
# Startup security posture audit — warn-on-load, never blocks: surfaces root / weak-SSH /
|
|
# ephemeral-container / unauthenticated-listener posture so operators see they're exposed.
|
|
try:
|
|
from hermes_cli.security_audit_startup import log_startup_security_warnings
|
|
|
|
_audit_cfg = None
|
|
try:
|
|
from hermes_cli.config import read_raw_config
|
|
|
|
_audit_cfg = read_raw_config()
|
|
except Exception:
|
|
_audit_cfg = None
|
|
log_startup_security_warnings(hermes_home=_hermes_home, config=_audit_cfg)
|
|
except Exception as _audit_exc:
|
|
logger.debug("Startup security audit failed (non-fatal): %s", _audit_exc)
|
|
|
|
# Optional stderr handler — level driven by -v/-q flags on the CLI.
|
|
# verbosity=None (-q/--quiet): no stderr output
|
|
# verbosity=0 (default): WARNING and above
|
|
# verbosity=1 (-v): INFO and above
|
|
# verbosity=2+ (-vv/-vvv): DEBUG
|
|
if verbosity is not None:
|
|
_stderr_level = {0: logging.WARNING, 1: logging.INFO}.get(verbosity, logging.DEBUG)
|
|
_stderr_handler = logging.StreamHandler(_safe_stderr())
|
|
_stderr_handler.setLevel(_stderr_level)
|
|
_stderr_handler.setFormatter(_gateway_stderr_formatter())
|
|
logging.getLogger().addHandler(_stderr_handler)
|
|
# Lower root logger level if needed so DEBUG records can reach the handler
|
|
if _stderr_level < logging.getLogger().level:
|
|
logging.getLogger().setLevel(_stderr_level)
|
|
|
|
|
|
def _start_gateway_make_shutdown_signal_handler(runner, _signal_initiated_shutdown: list):
|
|
"""Build the SIGINT/SIGTERM handler; ``_signal_initiated_shutdown[0]`` records an unplanned signal."""
|
|
def shutdown_signal_handler(received_signal=None):
|
|
# Planned --replace takeover: the sibling wrote a marker naming this PID before SIGTERM. Treat as
|
|
# planned, exit 0 so systemd's Restart=on-failure doesn't revive us to flap-fight the replacer
|
|
# (e.g. when both hermes.service and hermes-gateway.service are enabled).
|
|
planned_takeover = False
|
|
try:
|
|
from gateway.status import consume_takeover_marker_for_self
|
|
planned_takeover = consume_takeover_marker_for_self()
|
|
except Exception as e:
|
|
logger.debug("Takeover marker check failed: %s", e)
|
|
|
|
# Planned stop: service managers and `hermes gateway stop` also send SIGTERM, indistinguishable
|
|
# from an external kill unless the CLI marks it first. SIGINT is an interactive Ctrl+C stop.
|
|
planned_stop = False
|
|
if received_signal == signal.SIGINT:
|
|
planned_stop = True
|
|
elif not planned_takeover:
|
|
try:
|
|
from gateway.status import consume_planned_stop_marker_for_self
|
|
planned_stop = consume_planned_stop_marker_for_self()
|
|
except Exception as e:
|
|
logger.debug("Planned stop marker check failed: %s", e)
|
|
|
|
# Fast (<10ms) snapshot of who's asking us to shut down — runs synchronously inside the asyncio
|
|
# signal handler: stdlib + /proc only, no subprocesses (a sync `ps aux` here once blocked ~3s).
|
|
try:
|
|
from gateway.shutdown_forensics import (
|
|
format_context_for_log,
|
|
snapshot_shutdown_context,
|
|
spawn_async_diagnostic,
|
|
)
|
|
_shutdown_ctx = snapshot_shutdown_context(received_signal)
|
|
except Exception as _e:
|
|
_shutdown_ctx = None
|
|
logger.debug("snapshot_shutdown_context failed: %s", _e)
|
|
|
|
if planned_takeover:
|
|
logger.info(
|
|
"Received %s as a planned --replace takeover — exiting cleanly",
|
|
_shutdown_ctx["signal"] if _shutdown_ctx else "SIGTERM",
|
|
)
|
|
elif planned_stop:
|
|
logger.info(
|
|
"Received %s as a planned gateway stop — exiting cleanly",
|
|
_shutdown_ctx["signal"] if _shutdown_ctx else "SIGTERM/SIGINT",
|
|
)
|
|
else:
|
|
_signal_initiated_shutdown[0] = True
|
|
# Mirror onto the runner so _stop_impl can suppress the gateway_state=stopped persist for
|
|
# unexpected signals (container/s6 SIGTERM on restart, OOM, bare kill). Operator stops set a
|
|
# planned-stop marker, take the `planned_stop` branch above and leave this False (DO persist).
|
|
runner._signal_initiated_shutdown = True
|
|
logger.info(
|
|
"Received %s — initiating shutdown",
|
|
_shutdown_ctx["signal"] if _shutdown_ctx else "SIGTERM/SIGINT",
|
|
)
|
|
|
|
# Always log who/what triggered the signal — the most useful line for "gateway keeps dying"
|
|
# tickets. One line, key=value, parent_cmdline last (often long).
|
|
if _shutdown_ctx is not None:
|
|
try:
|
|
logger.warning(
|
|
"Shutdown context: %s", format_context_for_log(_shutdown_ctx)
|
|
)
|
|
except Exception as _e:
|
|
logger.debug("format_context_for_log failed: %s", _e)
|
|
|
|
# Spawn the heavyweight diagnostic (ps auxf, pstree, dmesg) detached so it can finish
|
|
# writing even if our cgroup is torn down; bounded by an internal timeout, never blocks.
|
|
try:
|
|
_diag_log = _hermes_home / "logs" / "gateway-shutdown-diag.log"
|
|
spawn_async_diagnostic(
|
|
_diag_log, _shutdown_ctx["signal"], timeout_seconds=5.0
|
|
)
|
|
except Exception as _e:
|
|
logger.debug("spawn_async_diagnostic failed: %s", _e)
|
|
asyncio.create_task(runner.stop())
|
|
return shutdown_signal_handler
|
|
|
|
|
|
def _start_gateway_claim_pid_file() -> bool:
|
|
"""Claim the runtime lock + PID file (O_EXCL winner is the authoritative gateway). False = lost."""
|
|
import atexit
|
|
from gateway.status import (
|
|
acquire_gateway_runtime_lock,
|
|
get_running_pid,
|
|
release_gateway_runtime_lock,
|
|
remove_pid_file,
|
|
write_pid_file,
|
|
)
|
|
_current_pid = get_running_pid()
|
|
if _current_pid is not None and _current_pid != os.getpid():
|
|
logger.error(
|
|
"Another gateway instance (PID %d) started during our startup. "
|
|
"Exiting to avoid double-running.", _current_pid
|
|
)
|
|
return False
|
|
if not acquire_gateway_runtime_lock():
|
|
logger.error(
|
|
"Gateway runtime lock is already held by another instance. Exiting."
|
|
)
|
|
return False
|
|
try:
|
|
write_pid_file()
|
|
except FileExistsError:
|
|
release_gateway_runtime_lock()
|
|
logger.error(
|
|
"PID file race lost to another gateway instance. Exiting."
|
|
)
|
|
return False
|
|
atexit.register(remove_pid_file)
|
|
atexit.register(release_gateway_runtime_lock)
|
|
return True
|
|
|
|
|
|
async def _start_gateway_start_control_socket(runner):
|
|
"""Start the gateway control socket (identify/status/pause-for-update); None when unavailable."""
|
|
import atexit
|
|
_control_server = None
|
|
try:
|
|
from gateway.control_socket import GatewayControlServer
|
|
|
|
# pause-for-update: the updater asks this gateway to drain in-flight turns and exit cleanly
|
|
# (releasing every venv file handle) instead of being tree-killed mid-turn — same drain path as
|
|
# SIGUSR1/service restarts (request_restart(via_service=True)). The handler runs on the socket's
|
|
# executor thread, so the request is marshalled onto the loop; the ACK returns the drain budget.
|
|
_main_loop = asyncio.get_running_loop()
|
|
|
|
def _pause_for_update_handler() -> dict:
|
|
try:
|
|
from hermes_cli.gateway import _get_restart_drain_timeout
|
|
|
|
_drain = float(_get_restart_drain_timeout())
|
|
except Exception:
|
|
_drain = 30.0
|
|
accepted_box: list[bool] = []
|
|
_done = threading.Event()
|
|
|
|
def _request() -> None:
|
|
try:
|
|
accepted_box.append(
|
|
runner.request_restart(detached=False, via_service=True)
|
|
)
|
|
finally:
|
|
_done.set()
|
|
|
|
_main_loop.call_soon_threadsafe(_request)
|
|
_done.wait(timeout=5.0)
|
|
accepted = bool(accepted_box and accepted_box[0])
|
|
return {
|
|
"pausing": accepted,
|
|
"already_stopping": not accepted,
|
|
"pid": os.getpid(),
|
|
"drain_timeout": _drain,
|
|
}
|
|
|
|
_control_server = GatewayControlServer(
|
|
verb_handlers={"pause-for-update": _pause_for_update_handler}
|
|
)
|
|
if not await _control_server.start():
|
|
_control_server = None
|
|
else:
|
|
atexit.register(_control_server.cleanup_files)
|
|
except Exception as _cs_exc:
|
|
logger.debug("Control socket startup failed (non-fatal): %s", _cs_exc)
|
|
_control_server = None
|
|
return _control_server
|
|
|
|
|
|
def _start_gateway_start_cron_and_housekeeping(runner):
|
|
"""Start the cron scheduler thread + gateway housekeeping thread.
|
|
|
|
Returns ``(cron_stop, cron_provider, cron_thread, housekeeping_thread)``.
|
|
"""
|
|
# Start the background cron scheduler via the resolved provider so scheduled jobs fire
|
|
# automatically. Pass the event loop so cron delivery can use live adapters (E2EE support).
|
|
from cron.scheduler_provider import (
|
|
InProcessCronScheduler,
|
|
resolve_cron_scheduler,
|
|
scheduler_for_profile_mode,
|
|
)
|
|
cron_stop = threading.Event()
|
|
multiplex_cron = bool(getattr(runner.config, "multiplex_profiles", False))
|
|
cron_provider = scheduler_for_profile_mode(
|
|
resolve_cron_scheduler(),
|
|
multiplex_profiles=multiplex_cron,
|
|
)
|
|
cron_start_kwargs: Dict[str, Any] = {"adapters": runner.adapters, "loop": asyncio.get_running_loop()}
|
|
|
|
# Multiplex profiles: tell the built-in ticker which profile homes to tick. Otherwise only the
|
|
# process-global HERMES_HOME is iterated and secondary profiles' cron jobs show as "scheduled"
|
|
# with a valid next_run_at but never execute because no ticker owns that store.
|
|
if (
|
|
isinstance(cron_provider, InProcessCronScheduler)
|
|
and multiplex_cron
|
|
):
|
|
try:
|
|
profile_homes = _multiplex_profile_homes(runner.config)
|
|
if profile_homes:
|
|
cron_start_kwargs["profile_homes"] = profile_homes
|
|
# Per-profile adapters so each profile's cron output goes via its own bot/adapter, not the
|
|
# default profile's.
|
|
cron_start_kwargs["profile_adapters"] = getattr(
|
|
runner, "_profile_adapters", None
|
|
)
|
|
# runner.adapters belongs to the default profile ("default" in the multiplex list).
|
|
# Thread that identity so the ticker reserves the shared adapters for the default
|
|
# profile alone and never routes a secondary's cron through the default bot (even
|
|
# before its adapter connects, when profile_adapters[name] is still absent/empty).
|
|
cron_start_kwargs["default_profile"] = "default"
|
|
logger.info(
|
|
"Cron scheduler will tick %d profile(s) under multiplex: %s",
|
|
len(profile_homes),
|
|
[p[0] if isinstance(p, tuple) else p for p in profile_homes],
|
|
)
|
|
except Exception as exc:
|
|
logger.warning(
|
|
"Could not resolve profile homes for multiplex cron: %s",
|
|
exc,
|
|
)
|
|
|
|
# External cron providers own their remote scheduling contract; only the in-process ticker polls
|
|
# local due jobs, so only it receives the local external-drain dispatch gate.
|
|
if isinstance(cron_provider, InProcessCronScheduler):
|
|
cron_start_kwargs["can_dispatch"] = lambda: not (
|
|
runner._draining or runner._external_drain_active
|
|
)
|
|
cron_thread = threading.Thread(
|
|
target=cron_provider.start,
|
|
args=(cron_stop,),
|
|
kwargs=cron_start_kwargs,
|
|
daemon=True,
|
|
name="cron-scheduler",
|
|
)
|
|
cron_thread.start()
|
|
|
|
# Preflight tell for the hosted fire path: an external cron provider fires over HTTP to THIS
|
|
# process's api_server adapter on loopback. If it never came up (typically API_SERVER_KEY missing)
|
|
# every fire fails with ConnectError while manual runs work — misread as a job bug. Say it ONCE now.
|
|
if not isinstance(cron_provider, InProcessCronScheduler):
|
|
try:
|
|
_has_api_server = Platform.API_SERVER in (runner.adapters or {})
|
|
except Exception:
|
|
_has_api_server = True # never let the tell break startup
|
|
if not _has_api_server:
|
|
logger.warning(
|
|
"Cron provider '%s' is active but the api_server adapter is "
|
|
"NOT running in this gateway — scheduled fires arrive over "
|
|
"loopback HTTP and will all fail (jobs only run when "
|
|
"triggered manually). Most common cause: API_SERVER_KEY is "
|
|
"missing from this gateway process's environment. Restart "
|
|
"the gateway through its supervisor (`hermes gateway "
|
|
"restart`) so the profile env loads.",
|
|
getattr(cron_provider, "name", "external"),
|
|
)
|
|
|
|
# Gateway-only periodic housekeeping (channel dir, cache cleanup, paste sweep, curator) — runs
|
|
# independently of the active cron provider; shares cron_stop as the shutdown signal.
|
|
housekeeping_thread = threading.Thread(
|
|
target=_start_gateway_housekeeping,
|
|
args=(cron_stop,),
|
|
kwargs={
|
|
"adapters": runner.adapters,
|
|
"loop": asyncio.get_running_loop(),
|
|
"cron_provider": cron_provider,
|
|
},
|
|
daemon=True,
|
|
name="gateway-housekeeping",
|
|
)
|
|
housekeeping_thread.start()
|
|
return cron_stop, cron_provider, cron_thread, housekeeping_thread
|
|
|
|
|
|
async def _start_gateway_shutdown_tail(
|
|
runner,
|
|
_control_server,
|
|
cron_stop: threading.Event,
|
|
cron_provider,
|
|
cron_thread: threading.Thread,
|
|
housekeeping_thread: threading.Thread,
|
|
_planned_stop_watcher_stop: threading.Event,
|
|
_planned_stop_watcher_thread: threading.Thread,
|
|
_signal_initiated_shutdown: list,
|
|
) -> bool:
|
|
"""Post-``wait_for_shutdown`` teardown; returns the process exit verdict (True = exit 0)."""
|
|
# Stop the control socket first: once shutdown begins this process is no longer a truthful "the
|
|
# gateway is serving here" answer, and a successor (--replace / supervisor respawn) must be able
|
|
# to bind. Early-exit paths above don't reach this; their atexit cleanup_files hook runs, and a
|
|
# successor clears any stale socket on bind.
|
|
if _control_server is not None:
|
|
try:
|
|
await _control_server.stop()
|
|
except Exception:
|
|
logger.debug("Control socket stop failed (non-fatal)", exc_info=True)
|
|
|
|
try:
|
|
from hermes_cli.nous_auth_keepalive import stop_nous_auth_keepalive
|
|
|
|
stop_nous_auth_keepalive()
|
|
except Exception:
|
|
pass
|
|
|
|
if runner.should_exit_with_failure:
|
|
if runner.exit_reason:
|
|
logger.error("Gateway exiting with failure: %s", runner.exit_reason)
|
|
return False
|
|
|
|
# Stop cron scheduler + housekeeping cooperatively, never join()ed: an in-flight cron delivery
|
|
# is a coroutine scheduled onto THIS loop while the ticker thread blocks on future.result(); a
|
|
# synchronous join would block the loop so the delivery never ran and the message was dropped.
|
|
cron_stop.set()
|
|
_stop_cron_provider(cron_provider)
|
|
if not await _await_thread_exit(cron_thread, timeout=_CRON_SHUTDOWN_DRAIN_TIMEOUT):
|
|
logger.warning(
|
|
"Cron ticker did not exit within %.0fs of shutdown — an in-flight "
|
|
"delivery may have been dropped.", _CRON_SHUTDOWN_DRAIN_TIMEOUT,
|
|
)
|
|
await _await_thread_exit(
|
|
housekeeping_thread, timeout=_HOUSEKEEPING_SHUTDOWN_DRAIN_TIMEOUT
|
|
)
|
|
|
|
# Stop the planned-stop watcher (daemon=True so this is belt-and-suspenders).
|
|
_planned_stop_watcher_stop.set()
|
|
_planned_stop_watcher_thread.join(timeout=2)
|
|
|
|
# Close MCP server connections (off-loop, bounded — #82874)
|
|
with suppress(Exception):
|
|
await _shutdown_mcp_servers_nonblocking()
|
|
|
|
if runner.exit_code is not None:
|
|
raise SystemExit(runner.exit_code)
|
|
|
|
# An unexpected SIGTERM that wasn't a planned restart (/restart, /update, SIGUSR1) exits
|
|
# non-zero so systemd's Restart=on-failure revives the process (hermes update killing the
|
|
# gateway mid-work, external kills, WSL2/container runtime signals). `hermes gateway stop` and
|
|
# Ctrl+C are handled above as planned stops and must not trigger revival.
|
|
if _signal_initiated_shutdown[0] and not runner._restart_requested:
|
|
logger.info(
|
|
"Exiting with code 1 (signal-initiated shutdown without restart "
|
|
"request) so systemd Restart=on-failure can revive the gateway."
|
|
)
|
|
return False # → sys.exit(1) in the caller
|
|
|
|
# Older restart paths may reach here without ``runner.exit_code`` set.
|
|
# Keep the historical non-zero fallback for service-managed restarts.
|
|
if runner._restart_via_service:
|
|
logger.info(
|
|
"Exiting with code 75 (service-restart requested) so the service "
|
|
"manager relaunches the gateway."
|
|
)
|
|
raise SystemExit(75)
|
|
|
|
return True
|
|
|
|
|
|
async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = False, verbosity: Optional[int] = 0) -> bool:
|
|
"""Start the gateway and run until interrupted.
|
|
|
|
Returns True if the gateway ran, False if it failed to start (non-zero exit so systemd can
|
|
auto-restart). ``replace`` kills any existing instance first — avoids systemd restart-loop
|
|
deadlocks when the previous process hasn't fully exited.
|
|
"""
|
|
# Enable interactive exec approval on messaging platforms. Set here (not at module import) so
|
|
# incidental imports of gateway.run from CLI/tool code don't poison HERMES_EXEC_ASK.
|
|
os.environ["HERMES_EXEC_ASK"] = "1"
|
|
|
|
from hermes_cli.resource_limits import apply_nofile_soft_limit
|
|
|
|
apply_nofile_soft_limit()
|
|
|
|
# Snapshot the checkout revision now, while sys.modules still matches disk, so a later `git
|
|
# pull` under this long-lived process can be detected (and risky work like model switching
|
|
# refused) instead of crashing on a stale in-memory module.
|
|
from gateway.code_skew import record_boot_fingerprint
|
|
record_boot_fingerprint()
|
|
|
|
# Duplicate-instance guard: no two gateways under one HERMES_HOME. The PID file is scoped to
|
|
# HERMES_HOME, so multi-profile setups (distinct HERMES_HOME each) run concurrently untripped.
|
|
from gateway.status import get_running_pid
|
|
existing_pid = get_running_pid()
|
|
if existing_pid is not None and existing_pid != os.getpid():
|
|
if not await _start_gateway_replace_existing_instance(existing_pid, replace):
|
|
return False
|
|
|
|
_start_gateway_configure_logging(verbosity)
|
|
|
|
runner = GatewayRunner(config)
|
|
# Multiplex: swap the launch-home file handlers for per-profile routers so each profile's records
|
|
# land in its own logs/. Must run after the runner resolved (possibly None) config and setup_logging.
|
|
_enable_multiplex_log_routing(runner.config)
|
|
# ``--replace`` is explicit startup authority, not a durable reconnect policy: GatewayRunner scopes
|
|
# it to cold adapter connects and clears it before the background reconnect watcher starts.
|
|
runner._platform_lock_takeover_on_start = bool(replace)
|
|
|
|
# Track whether an unexpected signal initiated shutdown: an unexpected SIGTERM exits non-zero so
|
|
# service managers revive us; planned stop paths write a marker first so they exit cleanly.
|
|
_signal_initiated_shutdown = [False]
|
|
|
|
# Set up signal handlers
|
|
shutdown_signal_handler = _start_gateway_make_shutdown_signal_handler(
|
|
runner, _signal_initiated_shutdown
|
|
)
|
|
|
|
def restart_signal_handler():
|
|
runner.request_restart(detached=False, via_service=True)
|
|
|
|
loop = asyncio.get_running_loop()
|
|
|
|
# Loop-level exception handler swallowing transient network errors from background tasks: an
|
|
# unhandled telegram TimedOut / NetworkError / httpx connection error in any awaited coroutine
|
|
# would kill the whole gateway. Deliberately narrow — everything else hits the default handler.
|
|
loop.set_exception_handler(_gateway_loop_exception_handler)
|
|
|
|
if threading.current_thread() is threading.main_thread():
|
|
for sig in (signal.SIGINT, signal.SIGTERM):
|
|
try:
|
|
loop.add_signal_handler(sig, shutdown_signal_handler, sig) # windows-footgun: ok — wrapped in try/except NotImplementedError for Windows
|
|
except NotImplementedError:
|
|
pass
|
|
if hasattr(signal, "SIGUSR1"):
|
|
try:
|
|
loop.add_signal_handler(signal.SIGUSR1, restart_signal_handler) # windows-footgun: ok — POSIX signal, guarded by hasattr above + try/except NotImplementedError
|
|
except NotImplementedError:
|
|
pass
|
|
else:
|
|
logger.info("Skipping signal handlers (not running in main thread).")
|
|
|
|
# Windows fallback: asyncio.add_signal_handler raises NotImplementedError there, so `hermes
|
|
# gateway stop`'s SIGTERM never reaches shutdown_signal_handler (no drain, sessions lost). A
|
|
# marker-polling thread notices the planned-stop marker written BEFORE the kill and drives the
|
|
# same shutdown path. Runs everywhere (cheap) so environments masking SIGTERM still drain cleanly.
|
|
_planned_stop_watcher_stop = threading.Event()
|
|
_planned_stop_watcher_thread = threading.Thread(
|
|
target=_run_planned_stop_watcher,
|
|
args=(_planned_stop_watcher_stop, runner, loop, shutdown_signal_handler),
|
|
daemon=True,
|
|
name="planned-stop-watcher",
|
|
)
|
|
_planned_stop_watcher_thread.start()
|
|
|
|
# Claim the PID file BEFORE bringing up any platform adapters: two concurrent `gateway run
|
|
# --replace` invocations both pass the termination-wait above, but only the O_CREAT|O_EXCL
|
|
# winner ever opens Telegram polling, Discord sockets, etc. The loser exits cleanly first.
|
|
if not _start_gateway_claim_pid_file():
|
|
return False
|
|
|
|
# Control socket — the gateway-owned identify/status surface. Started right after the PID-file
|
|
# claim, since winning that O_EXCL race makes this process the authoritative gateway for its
|
|
# HERMES_HOME. Non-fatal: a bind failure just leaves consumers on the process-scan/state-file layer.
|
|
_control_server = await _start_gateway_start_control_socket(runner)
|
|
|
|
# Lifecycle ledger: report if the previous life died uncleanly (SIGKILL / OOM / VM death), then
|
|
# claim the sentinel for this life. Placed after the PID-file/lock claim so only the
|
|
# authoritative gateway touches it — a --replace loser exiting above must not clobber it.
|
|
try:
|
|
from gateway.lifecycle_ledger import record_startup as _lifecycle_record_startup
|
|
_lifecycle_record_startup()
|
|
except Exception as _lc_exc:
|
|
logger.debug("Lifecycle ledger startup record failed: %s", _lc_exc)
|
|
|
|
try:
|
|
from hermes_cli.nous_auth_keepalive import start_nous_auth_keepalive
|
|
|
|
start_nous_auth_keepalive()
|
|
except Exception as exc:
|
|
logger.debug("Nous auth keepalive did not start: %s", exc)
|
|
|
|
_ensure_windows_gateway_venv_imports()
|
|
|
|
# MCP tool discovery in an executor so the loop stays responsive when a configured MCP server is
|
|
# slow/unreachable: discover_mcp_tools() blocks up to 120s, which on the loop thread would freeze
|
|
# platform heartbeats (Discord shard, Telegram polling).
|
|
try:
|
|
await _discover_gateway_mcp_tools(runner.config)
|
|
except Exception as e:
|
|
logger.debug("MCP tool discovery failed: %s", e)
|
|
|
|
# Start the gateway
|
|
try:
|
|
success = await runner.start()
|
|
except BaseException:
|
|
_shutdown_gateway_health_export(runner)
|
|
raise
|
|
if not success:
|
|
_shutdown_gateway_health_export(runner)
|
|
return False
|
|
# Recover any pending messages flushed during a previous shutdown (#72680).
|
|
try:
|
|
from gateway.shutdown_flush import recover_pending_to_db
|
|
recovered = recover_pending_to_db()
|
|
if recovered:
|
|
logger.info(
|
|
"Recovered %d pending message(s) from shutdown flush", recovered,
|
|
)
|
|
except Exception:
|
|
pass
|
|
if runner.should_exit_cleanly:
|
|
_shutdown_gateway_health_export(runner)
|
|
if runner.exit_reason:
|
|
logger.error("Gateway exiting cleanly: %s", runner.exit_reason)
|
|
# A clean exit carrying an explicit exit code (e.g. GATEWAY_FATAL_CONFIG_EXIT_CODE) must
|
|
# propagate so the s6 finish script can translate it (78 → 125) and stop the restart loop;
|
|
# otherwise the early `return True` exits 0 and s6 crash-loops the gateway anyway.
|
|
if runner.exit_code is not None:
|
|
raise SystemExit(runner.exit_code)
|
|
return True
|
|
if not runner._running:
|
|
# Startup was intentionally aborted by restart/shutdown before entering
|
|
# running mode; preserve that lifecycle path without starting cron.
|
|
try:
|
|
await runner.wait_for_shutdown()
|
|
if runner.should_exit_with_failure:
|
|
if runner.exit_reason:
|
|
logger.error("Gateway exiting with failure: %s", runner.exit_reason)
|
|
return False
|
|
with suppress(Exception):
|
|
await _shutdown_mcp_servers_nonblocking()
|
|
if runner.exit_code is not None:
|
|
raise SystemExit(runner.exit_code)
|
|
return True
|
|
finally:
|
|
_shutdown_gateway_health_export(runner)
|
|
|
|
cron_stop, cron_provider, cron_thread, housekeeping_thread = (
|
|
_start_gateway_start_cron_and_housekeeping(runner)
|
|
)
|
|
|
|
# READY is emitted only after adapters, cron and housekeeping reach their running boundary;
|
|
# missing config/systemd runtime state leaves the watchdog disabled without changing behavior.
|
|
start_watchdog = getattr(runner, "_start_systemd_watchdog", None)
|
|
if callable(start_watchdog):
|
|
start_watchdog()
|
|
|
|
# Wait for shutdown
|
|
await runner.wait_for_shutdown()
|
|
|
|
return await _start_gateway_shutdown_tail(
|
|
runner,
|
|
_control_server,
|
|
cron_stop,
|
|
cron_provider,
|
|
cron_thread,
|
|
housekeeping_thread,
|
|
_planned_stop_watcher_stop,
|
|
_planned_stop_watcher_thread,
|
|
_signal_initiated_shutdown,
|
|
)
|
|
|
|
|
|
def _guard_corrupt_user_config() -> None:
|
|
"""Fail closed when the active profile's config.yaml cannot be parsed.
|
|
|
|
Nobody is present to repair a corrupt config on this non-interactive surface, and continuing
|
|
on defaults lets provider auto-detection adopt ``.env`` credentials the config never named.
|
|
Same policy and escape hatch (``HERMES_IGNORE_USER_CONFIG=1``) as ``hermes_cli/main.py``.
|
|
"""
|
|
from hermes_cli.config import (
|
|
InvalidUserConfigError,
|
|
require_parseable_user_config,
|
|
)
|
|
|
|
try:
|
|
require_parseable_user_config()
|
|
except InvalidUserConfigError as exc:
|
|
print(f"Error: {exc}", file=sys.stderr)
|
|
raise SystemExit(2) from exc
|
|
|
|
|
|
def main():
|
|
"""CLI entry point for the gateway."""
|
|
# Refuse to start on a corrupt config.yaml — before any config-dependent
|
|
# startup (watchdog, DB opens, provider resolution). See _guard docstring.
|
|
_guard_corrupt_user_config()
|
|
|
|
# Advertise the agent harness to children (AI_AGENT = cross-agent standard, HERMES_AGENT = Hermes
|
|
# marker — mirrors _advertise_agent_env in hermes_cli/main.py, inlined to avoid its startup
|
|
# side effects). Value must equal registry id ``hermes-agent`` exactly; setdefault never clobbers.
|
|
os.environ.setdefault("AI_AGENT", "hermes-agent")
|
|
os.environ.setdefault("HERMES_AGENT", "true")
|
|
|
|
# Positive process identity: ledger registration + Windows job-object self-attach, so update-time
|
|
# reapers can identify this gateway (and its child tree dies with it on Windows). Best-effort.
|
|
try:
|
|
from hermes_cli.process_identity import (
|
|
attach_self_to_kill_on_close_job,
|
|
register_self,
|
|
)
|
|
|
|
register_self("gateway")
|
|
attach_self_to_kill_on_close_job()
|
|
except Exception:
|
|
pass
|
|
|
|
# Startup-liveness watchdog: armed before config load, DB opens and the rest of pre-loop startup
|
|
# so a deadlock there still gets the process respawned by the supervisor instead of wedging as a
|
|
# live-PID zombie (hermes_cli.main's argv fast-path covers import time). GatewayRunner disarms it.
|
|
try:
|
|
from gateway.startup_watchdog import arm_startup_watchdog
|
|
arm_startup_watchdog()
|
|
except Exception:
|
|
pass
|
|
|
|
# Force UTF-8 stdio on Windows — gateway logs and startup banner would
|
|
# otherwise UnicodeEncodeError on cp1252 consoles. No-op on POSIX.
|
|
try:
|
|
from hermes_cli.stdio import configure_windows_stdio
|
|
configure_windows_stdio()
|
|
except Exception:
|
|
pass
|
|
|
|
import argparse
|
|
|
|
parser = argparse.ArgumentParser(description="Hermes Gateway - Multi-platform messaging")
|
|
parser.add_argument("--config", "-c", help="Path to gateway config file")
|
|
parser.add_argument("--verbose", "-v", action="store_true", help="Verbose output")
|
|
|
|
args = parser.parse_args()
|
|
|
|
config = None
|
|
if args.config:
|
|
import yaml
|
|
with open(args.config, encoding="utf-8") as f:
|
|
data = yaml.safe_load(f) or {}
|
|
config = GatewayConfig.from_dict(data)
|
|
|
|
# start_gateway() completes full graceful teardown before returning OR raising SystemExit. Force-
|
|
# exit afterwards so a wedged non-daemon worker thread (e.g. an executor call with no timeout)
|
|
# can't block Py_FinalizeEx's thread join and strand the gateway half-shut. SystemExit is caught
|
|
# explicitly (all its paths finish teardown first) so EVERY exit path hits the os._exit backstop.
|
|
try:
|
|
success = asyncio.run(start_gateway(config))
|
|
exit_code = 0 if success else 1
|
|
except SystemExit as e:
|
|
# e.code may be None (→ 0), an int, or a str (→ 1, like CPython).
|
|
if e.code is None:
|
|
exit_code = 0
|
|
elif isinstance(e.code, int):
|
|
exit_code = e.code
|
|
else:
|
|
exit_code = 1
|
|
_exit_after_graceful_shutdown(exit_code)
|
|
|
|
|
|
def _exit_after_graceful_shutdown(exit_code: int) -> None:
|
|
"""Flush stdio, release the PID file + runtime lock, then hard-exit.
|
|
|
|
Graceful teardown is already complete, so ``os._exit`` (not ``sys.exit``): SystemExit triggers
|
|
``Py_FinalizeEx`` → joins every non-daemon thread — exactly the hang a wedged worker causes.
|
|
``os._exit`` bypasses ``atexit``, so ``remove_pid_file`` / ``release_gateway_runtime_lock`` are
|
|
called here explicitly (idempotent; the EARLY exit paths relied on atexit). Logging is drained
|
|
explicitly (bounded): file handlers sit behind a ``QueueListener`` thread whose atexit drain
|
|
never runs, so the last records would otherwise be lost.
|
|
"""
|
|
for stream in (sys.stdout, sys.stderr):
|
|
with suppress(Exception):
|
|
stream.flush()
|
|
# Release PID + runtime lock BEFORE the log drain: the drain is bounded but could take its full
|
|
# timeout on a wedged disk, and these locks must never be stranded. os._exit skips atexit and the
|
|
# early SystemExit paths never run _stop_impl, so release here (idempotent).
|
|
try:
|
|
from gateway.status import remove_pid_file, release_gateway_runtime_lock
|
|
remove_pid_file()
|
|
release_gateway_runtime_lock()
|
|
except Exception:
|
|
pass
|
|
# Mark this life cleanly exited in the lifecycle sentinel: the single funnel every graceful exit
|
|
# passes through, so the next boot's unclean-death detector fires only for genuine SIGKILL/OOM/VM
|
|
# deaths. Ownership-guarded: an old --replace life won't clobber the replacement's fresh sentinel.
|
|
try:
|
|
from gateway.lifecycle_ledger import mark_exited
|
|
mark_exited(exit_code, reason="graceful_shutdown")
|
|
except Exception:
|
|
pass
|
|
# Drain the async log queue (os._exit bypasses the listener's atexit drain). drain_log_queue() is
|
|
# bounded with no restart — NOT flush_log_queue(): a listener wedged on the rotation lock would
|
|
# re-freeze shutdown in an unbounded stop() join. No-op when logging never initialized a queue.
|
|
try:
|
|
from hermes_logging import drain_log_queue
|
|
drain_log_queue(timeout=1.0)
|
|
except Exception:
|
|
pass
|
|
os._exit(exit_code)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|