test: split tests/conftest.py into topic modules under tests/_fixtures
tests/conftest.py had grown to 2,343 lines, past the 2,000-line gate. Move three self-contained topics out, unchanged: - env_filter.py: the credential / behavioral env-var name tables the hermetic fixture blanks; - live_system_guard.py: the autouse live-system guard fixture, its marks and the protected checkout roots; - platform_gating.py: the platforms() marker evaluation used by the collection hook. conftest imports them instead of listing them in pytest_plugins: it is not the rootdir conftest, and pytest fails a run that loads a non-root conftest carrying pytest_plugins after startup. Imported fixtures register on the conftest module under their old names, so autouse order is unchanged. Tests that reached into the moved symbols now import the new modules.
This commit is contained in:
@@ -44,7 +44,7 @@ from pathlib import Path
|
||||
|
||||
_VALID_PLATFORMS = ("linux", "macos", "windows")
|
||||
|
||||
# Mirrors tests/conftest.py::_PLATFORM_ALIASES, keyed by lane name rather than
|
||||
# Mirrors tests/_fixtures/platform_gating.py::_PLATFORM_ALIASES, keyed by lane name rather than
|
||||
# sys.platform value: this side never runs on the host it is asking about.
|
||||
_SPEC_HOSTS = {
|
||||
"linux": frozenset({"linux"}),
|
||||
|
||||
1
tests/_fixtures/__init__.py
Normal file
1
tests/_fixtures/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
"""Topic modules of the suite-wide ``tests/conftest.py``, imported by it."""
|
||||
330
tests/_fixtures/env_filter.py
Normal file
330
tests/_fixtures/env_filter.py
Normal file
@@ -0,0 +1,330 @@
|
||||
"""Env-var names the hermetic ``_hermetic_environment`` fixture blanks before every test.
|
||||
|
||||
Pure data, split out of ``tests/conftest.py`` to keep it under the file-size
|
||||
gate; the fixture itself stays in conftest next to the sandbox it maintains.
|
||||
"""
|
||||
|
||||
|
||||
# ── Credential env-var filter ──────────────────────────────────────────────
|
||||
#
|
||||
# Any env var in the current process matching ONE of these patterns is
|
||||
# unset for every test. Developers' local keys cannot leak into assertions
|
||||
# about "auto-detect provider when key present".
|
||||
|
||||
_CREDENTIAL_SUFFIXES = (
|
||||
"_API_KEY",
|
||||
"_TOKEN",
|
||||
"_SECRET",
|
||||
"_PASSWORD",
|
||||
"_CREDENTIALS",
|
||||
"_ACCESS_KEY",
|
||||
"_SECRET_ACCESS_KEY",
|
||||
"_PRIVATE_KEY",
|
||||
"_OAUTH_TOKEN",
|
||||
"_WEBHOOK_SECRET",
|
||||
"_ENCRYPT_KEY",
|
||||
"_APP_SECRET",
|
||||
"_CLIENT_SECRET",
|
||||
"_CORP_SECRET",
|
||||
"_AES_KEY",
|
||||
)
|
||||
|
||||
# Explicit names (for ones that don't fit the suffix pattern)
|
||||
_CREDENTIAL_NAMES = frozenset({
|
||||
"AWS_ACCESS_KEY_ID",
|
||||
"AWS_SECRET_ACCESS_KEY",
|
||||
"AWS_SESSION_TOKEN",
|
||||
"ANTHROPIC_TOKEN",
|
||||
"FAL_KEY",
|
||||
"GH_TOKEN",
|
||||
"GITHUB_TOKEN",
|
||||
"OPENAI_API_KEY",
|
||||
"OPENROUTER_API_KEY",
|
||||
"NOUS_API_KEY",
|
||||
"GEMINI_API_KEY",
|
||||
"GOOGLE_API_KEY",
|
||||
"GROQ_API_KEY",
|
||||
"XAI_API_KEY",
|
||||
"MISTRAL_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"KIMI_API_KEY",
|
||||
"MOONSHOT_API_KEY",
|
||||
"GLM_API_KEY",
|
||||
"ZAI_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"OLLAMA_API_KEY",
|
||||
"OPENVIKING_API_KEY",
|
||||
"COPILOT_API_KEY",
|
||||
"CLAUDE_CODE_OAUTH_TOKEN",
|
||||
"BROWSERBASE_API_KEY",
|
||||
"FIRECRAWL_API_KEY",
|
||||
"PARALLEL_API_KEY",
|
||||
"EXA_API_KEY",
|
||||
"TAVILY_API_KEY",
|
||||
"PERPLEXITY_API_KEY",
|
||||
"WANDB_API_KEY",
|
||||
"ELEVENLABS_API_KEY",
|
||||
"HONCHO_API_KEY",
|
||||
"MEM0_API_KEY",
|
||||
"SUPERMEMORY_API_KEY",
|
||||
"RETAINDB_API_KEY",
|
||||
"HINDSIGHT_API_KEY",
|
||||
"HINDSIGHT_LLM_API_KEY",
|
||||
"DAYTONA_API_KEY",
|
||||
"TWILIO_AUTH_TOKEN",
|
||||
"TELEGRAM_BOT_TOKEN",
|
||||
"DISCORD_BOT_TOKEN",
|
||||
"SLACK_BOT_TOKEN",
|
||||
"SLACK_APP_TOKEN",
|
||||
"MATTERMOST_TOKEN",
|
||||
"MATRIX_ACCESS_TOKEN",
|
||||
"MATRIX_PASSWORD",
|
||||
"MATRIX_RECOVERY_KEY",
|
||||
"HASS_TOKEN",
|
||||
"EMAIL_PASSWORD",
|
||||
"BLUEBUBBLES_PASSWORD",
|
||||
"FEISHU_APP_SECRET",
|
||||
"FEISHU_ENCRYPT_KEY",
|
||||
"FEISHU_VERIFICATION_TOKEN",
|
||||
"DINGTALK_CLIENT_SECRET",
|
||||
"QQ_CLIENT_SECRET",
|
||||
"QQ_STT_API_KEY",
|
||||
"WECOM_SECRET",
|
||||
"WECOM_CALLBACK_CORP_SECRET",
|
||||
"WECOM_CALLBACK_TOKEN",
|
||||
"WECOM_CALLBACK_ENCODING_AES_KEY",
|
||||
"WEIXIN_TOKEN",
|
||||
"MODAL_TOKEN_ID",
|
||||
"MODAL_TOKEN_SECRET",
|
||||
"TERMINAL_SSH_KEY",
|
||||
"SUDO_PASSWORD",
|
||||
"GATEWAY_PROXY_KEY",
|
||||
"API_SERVER_KEY",
|
||||
"TOOL_GATEWAY_USER_TOKEN",
|
||||
"TELEGRAM_WEBHOOK_SECRET",
|
||||
"WEBHOOK_SECRET",
|
||||
"AI_GATEWAY_API_KEY",
|
||||
"VOICE_TOOLS_OPENAI_KEY",
|
||||
"BROWSER_USE_API_KEY",
|
||||
"CUSTOM_API_KEY",
|
||||
"GATEWAY_PROXY_URL",
|
||||
"GEMINI_BASE_URL",
|
||||
"OPENAI_BASE_URL",
|
||||
"OPENROUTER_BASE_URL",
|
||||
"OLLAMA_BASE_URL",
|
||||
"GROQ_BASE_URL",
|
||||
"XAI_BASE_URL",
|
||||
"AI_GATEWAY_BASE_URL",
|
||||
"ANTHROPIC_BASE_URL",
|
||||
})
|
||||
|
||||
|
||||
def _looks_like_credential(name: str) -> bool:
|
||||
"""True if env var name matches a credential-shaped pattern."""
|
||||
if name in _CREDENTIAL_NAMES:
|
||||
return True
|
||||
return any(name.endswith(suf) for suf in _CREDENTIAL_SUFFIXES)
|
||||
|
||||
|
||||
# HERMES_* vars that change test behavior by being set. Unset all of these
|
||||
# unconditionally — individual tests that need them set do so explicitly.
|
||||
_HERMES_BEHAVIORAL_VARS = frozenset({
|
||||
# Voice/TTS runtime flags. ``tui_gateway/server.py`` reads these straight
|
||||
# off ``os.environ`` at call time (``_voice_mode_enabled`` /
|
||||
# ``_voice_tts_enabled``) and, on every completed turn, hands the turn's
|
||||
# final response text to ``hermes_cli.voice.speak_text`` — real synthesis,
|
||||
# real playback, out of the developer's speakers. Blank them per-test so a
|
||||
# leak (from the shell, or from an earlier test that drove the
|
||||
# ``voice.toggle`` RPC, which writes ``os.environ`` directly) cannot carry
|
||||
# into the next test. See ``_audio_playback_guard`` for the second layer.
|
||||
"HERMES_VOICE",
|
||||
"HERMES_VOICE_TTS",
|
||||
"HERMES_YOLO_MODE",
|
||||
# Injected into subprocess envs by the terminal tool (_make_run_env), so
|
||||
# any test run launched FROM a Hermes agent session inherits them and
|
||||
# hermes_constants home-resolution helpers prefer them over monkeypatched
|
||||
# HOME (test_subprocess_home_isolation red locally, green on CI).
|
||||
"HERMES_REAL_HOME",
|
||||
"TERMINAL_HOME_MODE",
|
||||
"HERMES_INTERACTIVE",
|
||||
"HERMES_QUIET",
|
||||
"HERMES_TOOL_PROGRESS",
|
||||
"HERMES_TOOL_PROGRESS_MODE",
|
||||
"HERMES_MAX_ITERATIONS",
|
||||
"HERMES_SESSION_PLATFORM",
|
||||
"HERMES_SESSION_CHAT_ID",
|
||||
"HERMES_SESSION_CHAT_NAME",
|
||||
"HERMES_SESSION_CHAT_TYPE",
|
||||
"HERMES_SESSION_THREAD_ID",
|
||||
"HERMES_SESSION_SOURCE",
|
||||
"HERMES_SESSION_KEY",
|
||||
"HERMES_GATEWAY_SESSION",
|
||||
"HERMES_CRON_SESSION",
|
||||
"_HERMES_GATEWAY",
|
||||
"HERMES_PLATFORM",
|
||||
"HERMES_MODEL",
|
||||
"HERMES_INFERENCE_MODEL",
|
||||
"HERMES_INFERENCE_PROVIDER",
|
||||
"HERMES_TUI_PROVIDER",
|
||||
"HERMES_MANAGED",
|
||||
"HERMES_MANAGED_DIR",
|
||||
# A Nix-wrapped `hermes` on the developer's host exports the store's read-only plugins
|
||||
# tree; tests must discover the checkout's plugins/ (get_bundled_plugins_dir), not a
|
||||
# different release's.
|
||||
"HERMES_BUNDLED_PLUGINS",
|
||||
"HERMES_DEV",
|
||||
"HERMES_CONTAINER",
|
||||
"HERMES_EPHEMERAL_SYSTEM_PROMPT",
|
||||
"HERMES_TIMEZONE",
|
||||
"HERMES_REDACT_SECRETS",
|
||||
"HERMES_BACKGROUND_NOTIFICATIONS",
|
||||
"HERMES_EXEC_ASK",
|
||||
"HERMES_HOME_MODE",
|
||||
"HERMES_AGENT_USE_LEGACY_SESSION_KEYS",
|
||||
# Kanban path/board pins must never leak from a developer shell or
|
||||
# dispatched worker into tests; otherwise tests can write fake tasks to
|
||||
# the real ~/.hermes/kanban.db instead of the per-test HERMES_HOME.
|
||||
"HERMES_KANBAN_DB",
|
||||
"HERMES_KANBAN_BOARD",
|
||||
"HERMES_KANBAN_HOME",
|
||||
"HERMES_KANBAN_WORKSPACES_ROOT",
|
||||
"HERMES_KANBAN_LOGS_ROOT",
|
||||
"HERMES_KANBAN_TASK",
|
||||
"HERMES_KANBAN_WORKSPACE",
|
||||
"HERMES_KANBAN_RUN_ID",
|
||||
"HERMES_KANBAN_CLAIM_LOCK",
|
||||
"HERMES_KANBAN_DISPATCH_IN_GATEWAY",
|
||||
# Pytest is routinely launched from a delegated worker. The worker
|
||||
# lineage marker must not make parent-state tests run as delegated
|
||||
# children; tests that exercise child behavior set it explicitly.
|
||||
"HERMES_DELEGATED_CHILD_CONTEXT",
|
||||
"HERMES_TENANT",
|
||||
# Honcho host selection changes which nested config block wins. A local
|
||||
# shell override leaked "myhost" into the full suite and flipped 20
|
||||
# otherwise-unrelated config tests away from the default "hermes" host.
|
||||
"HERMES_HONCHO_HOST",
|
||||
# Dashboard OAuth auth gate (PR #30156). When set, the bundled
|
||||
# dashboard-auth `nous` plugin auto-registers itself on plugin discovery,
|
||||
# which is triggered by any `/api/status` call. That leaks a provider
|
||||
# into the dashboard_auth registry across tests in the same worker and
|
||||
# makes assertions like `auth_providers == []` flaky. CI never sets
|
||||
# these, so production tests must not see them either.
|
||||
"HERMES_DASHBOARD_OAUTH_CLIENT_ID",
|
||||
"HERMES_DASHBOARD_PORTAL_URL",
|
||||
"TERMINAL_CWD",
|
||||
"TERMINAL_ENV",
|
||||
"TERMINAL_VERCEL_RUNTIME",
|
||||
"TERMINAL_CONTAINER_CPU",
|
||||
"TERMINAL_CONTAINER_DISK",
|
||||
"TERMINAL_CONTAINER_MEMORY",
|
||||
"TERMINAL_CONTAINER_PERSISTENT",
|
||||
"TERMINAL_DOCKER_PERSIST_ACROSS_PROCESSES",
|
||||
"TERMINAL_DOCKER_ORPHAN_REAPER",
|
||||
"TERMINAL_DOCKER_RUN_AS_HOST_USER",
|
||||
"BROWSER_CDP_URL",
|
||||
"CAMOFOX_URL",
|
||||
# Platform allowlists — not credentials, but if set from any source
|
||||
# (user shell, earlier leaky test, CI env), they change gateway auth
|
||||
# behavior and flake button-authorization tests.
|
||||
"TELEGRAM_ALLOWED_USERS",
|
||||
"TELEGRAM_GROUP_ALLOWED_USERS",
|
||||
"TELEGRAM_GROUP_ALLOWED_CHATS",
|
||||
"QQ_ALLOWED_USERS",
|
||||
"QQ_GROUP_ALLOWED_USERS",
|
||||
"DISCORD_ALLOWED_USERS",
|
||||
"WHATSAPP_ALLOWED_USERS",
|
||||
"SLACK_ALLOWED_USERS",
|
||||
"SIGNAL_ALLOWED_USERS",
|
||||
"SIGNAL_GROUP_ALLOWED_USERS",
|
||||
"EMAIL_ALLOWED_USERS",
|
||||
"SMS_ALLOWED_USERS",
|
||||
"MATTERMOST_ALLOWED_USERS",
|
||||
"MATRIX_ALLOWED_USERS",
|
||||
"DINGTALK_ALLOWED_USERS",
|
||||
"FEISHU_ALLOWED_USERS",
|
||||
"WECOM_ALLOWED_USERS",
|
||||
"PHOTON_ALLOWED_USERS",
|
||||
"GATEWAY_ALLOWED_USERS",
|
||||
"GATEWAY_ALLOW_ALL_USERS",
|
||||
"TELEGRAM_ALLOW_ALL_USERS",
|
||||
"DISCORD_ALLOW_ALL_USERS",
|
||||
"WHATSAPP_ALLOW_ALL_USERS",
|
||||
"SLACK_ALLOW_ALL_USERS",
|
||||
"SIGNAL_ALLOW_ALL_USERS",
|
||||
"EMAIL_ALLOW_ALL_USERS",
|
||||
"SMS_ALLOW_ALL_USERS",
|
||||
"PHOTON_ALLOW_ALL_USERS",
|
||||
# Gateway home channels are set by /sethome in real profiles. Tests that
|
||||
# exercise dashboard notification toggles must opt in explicitly or they
|
||||
# can accidentally subscribe against a developer's real home channel.
|
||||
"TELEGRAM_HOME_CHANNEL",
|
||||
"TELEGRAM_HOME_CHANNEL_THREAD_ID",
|
||||
"TELEGRAM_HOME_CHANNEL_NAME",
|
||||
"TELEGRAM_CRON_THREAD_ID",
|
||||
"DISCORD_HOME_CHANNEL",
|
||||
"DISCORD_HOME_CHANNEL_THREAD_ID",
|
||||
"DISCORD_HOME_CHANNEL_NAME",
|
||||
"SLACK_HOME_CHANNEL",
|
||||
"SLACK_HOME_CHANNEL_THREAD_ID",
|
||||
"SLACK_HOME_CHANNEL_NAME",
|
||||
"WHATSAPP_HOME_CHANNEL",
|
||||
"WHATSAPP_HOME_CHANNEL_THREAD_ID",
|
||||
"WHATSAPP_HOME_CHANNEL_NAME",
|
||||
"SIGNAL_HOME_CHANNEL",
|
||||
"SIGNAL_HOME_CHANNEL_THREAD_ID",
|
||||
"SIGNAL_HOME_CHANNEL_NAME",
|
||||
"EMAIL_HOME_CHANNEL",
|
||||
"EMAIL_HOME_CHANNEL_THREAD_ID",
|
||||
"EMAIL_HOME_CHANNEL_NAME",
|
||||
"SMS_HOME_CHANNEL",
|
||||
"SMS_HOME_CHANNEL_THREAD_ID",
|
||||
"SMS_HOME_CHANNEL_NAME",
|
||||
"MATTERMOST_HOME_CHANNEL",
|
||||
"MATTERMOST_HOME_CHANNEL_THREAD_ID",
|
||||
"MATTERMOST_HOME_CHANNEL_NAME",
|
||||
"MATRIX_HOME_CHANNEL",
|
||||
"MATRIX_HOME_CHANNEL_THREAD_ID",
|
||||
"MATRIX_HOME_CHANNEL_NAME",
|
||||
"DINGTALK_HOME_CHANNEL",
|
||||
"DINGTALK_HOME_CHANNEL_THREAD_ID",
|
||||
"DINGTALK_HOME_CHANNEL_NAME",
|
||||
"FEISHU_HOME_CHANNEL",
|
||||
"FEISHU_HOME_CHANNEL_THREAD_ID",
|
||||
"FEISHU_HOME_CHANNEL_NAME",
|
||||
"WECOM_HOME_CHANNEL",
|
||||
"WECOM_HOME_CHANNEL_THREAD_ID",
|
||||
"WECOM_HOME_CHANNEL_NAME",
|
||||
"PHOTON_HOME_CHANNEL",
|
||||
"PHOTON_HOME_CHANNEL_THREAD_ID",
|
||||
"PHOTON_HOME_CHANNEL_NAME",
|
||||
# API server bind/auth settings are common in local gateway profiles and
|
||||
# change adapter defaults plus load_gateway_config() enablement. Tests that
|
||||
# need them set opt in explicitly with monkeypatch.
|
||||
"API_SERVER_ENABLED",
|
||||
"API_SERVER_HOST",
|
||||
"API_SERVER_PORT",
|
||||
"API_SERVER_KEY",
|
||||
"API_SERVER_CORS_ORIGINS",
|
||||
"API_SERVER_MODEL_NAME",
|
||||
# Platform gating — set by load_gateway_config() as a side effect when
|
||||
# a config.yaml is present, so individual test bodies that call the
|
||||
# loader leak these values into later tests in the same process.
|
||||
# Force-clear on every test setup so the leak can't happen.
|
||||
"SLACK_REQUIRE_MENTION",
|
||||
"SLACK_STRICT_MENTION",
|
||||
"SLACK_THREAD_REQUIRE_MENTION",
|
||||
"SLACK_IGNORE_OTHER_USER_MENTIONS",
|
||||
"SLACK_REQUIRE_MENTION_CHANNELS",
|
||||
"SLACK_FREE_RESPONSE_CHANNELS",
|
||||
"SLACK_ALLOWED_CHANNELS",
|
||||
"SLACK_IGNORED_CHANNELS",
|
||||
"SLACK_DISABLE_DMS",
|
||||
"SLACK_ALLOW_BOTS",
|
||||
"SLACK_REACTIONS",
|
||||
"DISCORD_REQUIRE_MENTION",
|
||||
"DISCORD_FREE_RESPONSE_CHANNELS",
|
||||
"TELEGRAM_REQUIRE_MENTION",
|
||||
"WHATSAPP_REQUIRE_MENTION",
|
||||
"DINGTALK_REQUIRE_MENTION",
|
||||
"MATRIX_REQUIRE_MENTION",
|
||||
})
|
||||
465
tests/_fixtures/live_system_guard.py
Normal file
465
tests/_fixtures/live_system_guard.py
Normal file
@@ -0,0 +1,465 @@
|
||||
"""The autouse live-system guard: no real kills, systemctl writes, gateway spawns or checkout writes.
|
||||
|
||||
Imported into ``tests/conftest.py`` so pytest registers the fixture there;
|
||||
``pytest_plugins`` is not an option because ``tests/conftest.py`` is not the
|
||||
rootdir conftest (the rootdir is the repo root, where ``pyproject.toml`` lives).
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent.parent
|
||||
|
||||
|
||||
# ── Live-system guard ──────────────────────────────────────────────────────
|
||||
#
|
||||
# Several test files exercise the gateway-restart / kill code paths
|
||||
# (``cmd_update``, ``kill_gateway_processes``, ``stop_profile_gateway``).
|
||||
# When a single test forgets to mock either ``os.kill`` or the global
|
||||
# ``find_gateway_pids`` helper, the real call leaks out of the hermetic
|
||||
# environment and finds the developer's live ``hermes-gateway`` process
|
||||
# via ``psutil`` — sending it SIGTERM mid-test. The shutdown forensics in
|
||||
# PR #23285 caught this happening 5+ times in 3 days, every time
|
||||
# correlated with a ``tests/hermes_cli/`` pytest run starting up.
|
||||
#
|
||||
# This fixture makes the leak impossible by intercepting the two
|
||||
# primitives that actually do damage:
|
||||
#
|
||||
# • ``os.kill`` rejects any PID outside the test process subtree with
|
||||
# a hard ``RuntimeError`` so the offending test gets a stack trace
|
||||
# instead of silently murdering the real gateway.
|
||||
# • ``subprocess.run`` / ``subprocess.Popen`` / ``call`` / ``check_call`` /
|
||||
# ``check_output`` reject any ``systemctl ... <verb> hermes-gateway``
|
||||
# invocation that would mutate the live unit. Read-only systemctl
|
||||
# calls (``status``, ``show``, ``list-units``) still pass through.
|
||||
#
|
||||
# We intentionally do NOT stub ``find_gateway_pids`` / ``_scan_gateway_pids``
|
||||
# here — tests of those functions themselves need the real implementation.
|
||||
# Even if a test gets the live gateway PID back from a real scan, the
|
||||
# ``os.kill`` guard above catches the actual signal call, and the
|
||||
# ``systemctl`` guard catches the systemd path. Discovery without
|
||||
# delivery is harmless.
|
||||
|
||||
_LIVE_SYSTEM_GUARD_BYPASS_MARK = "live_system_guard_bypass"
|
||||
_GATEWAY_LOOKALIKE_MARK = "spawns_gateway_lookalike"
|
||||
|
||||
# Tests may designate a temporary repo to exercise the real guard safely.
|
||||
_LIVE_GUARD_PROTECTED_GIT_ROOTS = (PROJECT_ROOT,)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _live_system_guard(request, monkeypatch):
|
||||
"""Block real os.kill / systemctl / gateway-pid scans during tests.
|
||||
|
||||
See block comment above for the why. Tests that genuinely need
|
||||
real signal delivery (e.g. PTY tests that SIGINT their own child)
|
||||
can opt out with ``@pytest.mark.live_system_guard_bypass``.
|
||||
|
||||
Coverage (every primitive that can deliver a signal to or otherwise
|
||||
terminate a foreign process):
|
||||
• os.kill, os.killpg (POSIX)
|
||||
• subprocess.run / Popen / call / check_call / check_output
|
||||
• subprocess.getoutput / getstatusoutput
|
||||
• os.system / os.popen
|
||||
• pty.spawn
|
||||
• asyncio.create_subprocess_exec / create_subprocess_shell
|
||||
Subprocess inspection looks at the WHOLE command string (not just
|
||||
tokens[0]), so ``bash -c "systemctl restart hermes-gateway"``,
|
||||
``sudo systemctl ...``, ``env systemctl ...``, ``setsid systemctl ...``
|
||||
are all caught. ``pkill``/``killall``/``taskkill`` invocations
|
||||
targeting hermes/python patterns are also blocked. Bare ``git``
|
||||
commands may not mutate a protected checkout. Git writes against
|
||||
temporary repositories and read-only Git commands remain allowed.
|
||||
"""
|
||||
if request.node.get_closest_marker(_LIVE_SYSTEM_GUARD_BYPASS_MARK):
|
||||
yield
|
||||
return
|
||||
|
||||
import os as _os
|
||||
import shlex as _shlex
|
||||
import subprocess as _subprocess
|
||||
|
||||
test_pid = _os.getpid()
|
||||
lookalike_ok = request.node.get_closest_marker(_GATEWAY_LOOKALIKE_MARK) is not None
|
||||
# Capture the test process's existing children at fixture start —
|
||||
# any *new* children spawned by the test are also allowlisted via
|
||||
# the live psutil walk below. Static set keeps the fast path cheap.
|
||||
try:
|
||||
import psutil as _psutil
|
||||
_initial_children = {
|
||||
c.pid for c in _psutil.Process(test_pid).children(recursive=True)
|
||||
}
|
||||
except Exception:
|
||||
_psutil = None
|
||||
_initial_children = set()
|
||||
|
||||
def _is_own_subtree(pid: int) -> bool:
|
||||
# PID 0 means "our own process group"; -1 means "every process we
|
||||
# can signal". Both are dangerous when paired with SIGTERM/SIGKILL,
|
||||
# but pid 0 is technically scoped to our group so allow it; pid -1
|
||||
# is treated as foreign (refuse).
|
||||
if pid == 0:
|
||||
return True
|
||||
if pid < 0:
|
||||
return False
|
||||
if pid == test_pid or pid in _initial_children:
|
||||
return True
|
||||
if _psutil is None:
|
||||
return False
|
||||
try:
|
||||
walker = _psutil.Process(pid)
|
||||
except Exception:
|
||||
# Stale PID — kill would be a no-op anyway, allow it.
|
||||
return True
|
||||
try:
|
||||
for parent in walker.parents():
|
||||
if parent.pid == test_pid:
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
return False
|
||||
|
||||
real_kill = _os.kill
|
||||
|
||||
def _guarded_kill(pid, sig, *args, **kwargs):
|
||||
# Signal 0 is a pure liveness probe — it cannot terminate anything.
|
||||
# psutil.pid_exists() uses os.kill(pid, 0) on POSIX, and probing a
|
||||
# just-killed grandchild that was reparented to init (zombie with a
|
||||
# foreign parent chain) must not trip the guard. Flaked in CI on
|
||||
# test_entire_tree_is_sigkilled_not_just_parent.
|
||||
if int(sig) == 0:
|
||||
return real_kill(pid, sig, *args, **kwargs)
|
||||
if _is_own_subtree(int(pid)):
|
||||
return real_kill(pid, sig, *args, **kwargs)
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked os.kill("
|
||||
f"{pid}, {sig}) — PID is outside the test process subtree. "
|
||||
"If this fired in CI it means the test reached a real "
|
||||
"kill_gateway_processes / stop_profile_gateway / cmd_update "
|
||||
"code path without mocking find_gateway_pids and os.kill. "
|
||||
"Mock both, or mark the test with "
|
||||
"@pytest.mark.live_system_guard_bypass if real signal "
|
||||
"delivery is genuinely required."
|
||||
)
|
||||
|
||||
monkeypatch.setattr(_os, "kill", _guarded_kill)
|
||||
|
||||
# ``os.killpg`` is the same risk class — sends a signal to every
|
||||
# process in a group. The gateway is a session leader (its own
|
||||
# PGID == its PID), so killpg(gateway_pid, SIGTERM) is a one-shot
|
||||
# kill of the live process. Allow it only when the target PGID is
|
||||
# the test process's own group.
|
||||
if hasattr(_os, "killpg"):
|
||||
real_killpg = _os.killpg
|
||||
own_pgid = _os.getpgrp()
|
||||
|
||||
def _guarded_killpg(pgid, sig, *args, **kwargs):
|
||||
# Signal 0 is a pure liveness probe — never destructive.
|
||||
if int(sig) == 0:
|
||||
return real_killpg(pgid, sig, *args, **kwargs)
|
||||
if int(pgid) == own_pgid or _is_own_subtree(int(pgid)):
|
||||
return real_killpg(pgid, sig, *args, **kwargs)
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"os.killpg({pgid}, {sig}) — PGID is outside the test "
|
||||
"process group. See _live_system_guard for the why."
|
||||
)
|
||||
|
||||
monkeypatch.setattr(_os, "killpg", _guarded_killpg)
|
||||
|
||||
# ── Subprocess command-string inspection (whole-line) ──────────
|
||||
_HERMES_TOKENS = (
|
||||
"hermes-gateway",
|
||||
"hermes.service",
|
||||
"hermes_cli.main gateway",
|
||||
"hermes_cli/main.py gateway",
|
||||
"gateway/run.py",
|
||||
"hermes gateway",
|
||||
)
|
||||
_MUTATING_VERBS = (
|
||||
"restart", "start", "stop", "kill", "reload",
|
||||
"reset-failed", "enable", "disable", "mask", "unmask",
|
||||
"daemon-reload", "try-restart", "reload-or-restart",
|
||||
)
|
||||
_PROCESS_KILLERS = ("pkill", "killall", "taskkill", "skill", "fuser")
|
||||
_CONTAINER_RUNTIMES = ("docker", "podman", "nerdctl")
|
||||
|
||||
def _first_token_basename(cmd_str: str) -> str:
|
||||
try:
|
||||
tokens = _shlex.split(cmd_str)
|
||||
except ValueError:
|
||||
tokens = cmd_str.split()
|
||||
return tokens[0].rsplit("/", 1)[-1].lower() if tokens else ""
|
||||
# Shell/launcher executables whose arguments are themselves commands —
|
||||
# argv[0]-only scanning must not exempt what they wrap.
|
||||
_WRAPPER_COMMANDS = (
|
||||
"sh", "bash", "zsh", "dash", "env", "nohup", "setsid",
|
||||
"timeout", "sudo", "xargs", "nice", "ionice", "stdbuf", "flock",
|
||||
)
|
||||
|
||||
def _cmd_to_string(cmd) -> str:
|
||||
if cmd is None:
|
||||
return ""
|
||||
if isinstance(cmd, (bytes, bytearray)):
|
||||
try:
|
||||
return bytes(cmd).decode(errors="replace")
|
||||
except Exception:
|
||||
return ""
|
||||
if isinstance(cmd, str):
|
||||
return cmd
|
||||
if isinstance(cmd, (list, tuple)):
|
||||
try:
|
||||
return " ".join(str(t) for t in cmd)
|
||||
except Exception:
|
||||
return ""
|
||||
return str(cmd)
|
||||
|
||||
def _matches_hermes_gateway(cmd_str: str) -> bool:
|
||||
low = cmd_str.lower()
|
||||
return any(tok in low for tok in _HERMES_TOKENS)
|
||||
|
||||
def _is_blocked_systemctl(cmd) -> bool:
|
||||
cmd_str = _cmd_to_string(cmd)
|
||||
if "systemctl" not in cmd_str:
|
||||
return False
|
||||
if not _matches_hermes_gateway(cmd_str):
|
||||
return False
|
||||
try:
|
||||
tokens = _shlex.split(cmd_str)
|
||||
except ValueError:
|
||||
tokens = cmd_str.split()
|
||||
return any(verb in tokens for verb in _MUTATING_VERBS)
|
||||
|
||||
def _is_process_killer(cmd) -> bool:
|
||||
cmd_str = _cmd_to_string(cmd)
|
||||
try:
|
||||
tokens = _shlex.split(cmd_str)
|
||||
except ValueError:
|
||||
tokens = cmd_str.split()
|
||||
if not tokens:
|
||||
return False
|
||||
|
||||
# For argv-style calls only argv[0] is the executable; scanning every
|
||||
# argument blocked innocent commands like ``cat /tmp/.../skill``
|
||||
# ("skill" is in _PROCESS_KILLERS). Wrapper executables still get
|
||||
# full-token scanning so ``["bash", "-c", "pkill ..."]`` stays caught.
|
||||
if isinstance(cmd, (list, tuple)):
|
||||
head0 = tokens[0].rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
|
||||
killer_tokens = tokens if head0 in _WRAPPER_COMMANDS else tokens[:1]
|
||||
else:
|
||||
killer_tokens = tokens
|
||||
for tok in killer_tokens:
|
||||
head = tok.rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
|
||||
if head in _PROCESS_KILLERS:
|
||||
low = cmd_str.lower()
|
||||
# pkill -f pattern: catch hermes-themed patterns + a
|
||||
# plain "python" -f which would catch the live gateway
|
||||
# whose cmdline contains "python -m hermes_cli.main".
|
||||
if (
|
||||
"hermes" in low
|
||||
or "gateway" in low
|
||||
or ("python" in low and "-f" in tokens)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
from tests.git_safety import blocked_git_mutation
|
||||
|
||||
def _check_subprocess_cmd(name, cmd, kwargs=None):
|
||||
git_verb = blocked_git_mutation(cmd, kwargs, _LIVE_GUARD_PROTECTED_GIT_ROOTS)
|
||||
if git_verb is not None:
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — `git {git_verb}` would mutate "
|
||||
"the protected checkout. Use a temporary repository or mock "
|
||||
"the update boundary; live_system_guard_bypass is for deliberate live tests."
|
||||
)
|
||||
if _is_blocked_systemctl(cmd):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — would mutate the "
|
||||
"live hermes-gateway systemd unit. Mock "
|
||||
"subprocess.run / _run_systemctl in the test, or "
|
||||
"mark with @pytest.mark.live_system_guard_bypass."
|
||||
)
|
||||
if _is_process_killer(cmd):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — process-killer command "
|
||||
"targeting hermes/python could hit the live gateway. "
|
||||
"Mark with @pytest.mark.live_system_guard_bypass if "
|
||||
"intentional."
|
||||
)
|
||||
# Block any subprocess that would run `hermes update` (or the
|
||||
# equivalent `python -m hermes_cli.main update`). These commands
|
||||
# run `git fetch origin + git pull` against the REAL checkout,
|
||||
# overwriting files like pyproject.toml mid-test-run and corrupting
|
||||
# every subsequent subprocess that reads them. The corruption is
|
||||
# especially insidious because the spawned process uses setsid/
|
||||
# start_new_session=True, making it invisible to pytest's process
|
||||
# tree (PPid=1) and nearly impossible to trace without explicit
|
||||
# inotify/SHA watchdogs. Any test that legitimately needs to exercise
|
||||
# the update-spawn path must mock subprocess.Popen explicitly.
|
||||
cmd_str = _cmd_to_string(cmd)
|
||||
low = cmd_str.lower()
|
||||
if "update" in low and (
|
||||
# hermes update / hermes update --gateway / setsid bash -c ... hermes update
|
||||
("hermes" in low and "update" in low.split())
|
||||
or
|
||||
# python -m hermes_cli.main update --gateway
|
||||
("hermes_cli" in low and "update" in low.split())
|
||||
or
|
||||
# venv/bin/hermes update (absolute path variant used in tests)
|
||||
(".venv/bin/hermes" in low and "update" in low)
|
||||
):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — this command would run "
|
||||
"`hermes update` against the real checkout, fetching "
|
||||
"from origin and overwriting repo files (e.g. "
|
||||
"pyproject.toml) mid-test-run. This corrupts every "
|
||||
"subsequent subprocess in the same runner. "
|
||||
"Mock subprocess.Popen (and subprocess.run if used) "
|
||||
"in the test instead, or mark with "
|
||||
"@pytest.mark.live_system_guard_bypass if genuinely "
|
||||
"needed (e.g. an integration test testing the update "
|
||||
"flow against a dedicated throwaway repo)."
|
||||
)
|
||||
# Block spawning a REAL gateway runtime (``python -m hermes_cli.main
|
||||
# gateway run|start|restart``). ``_spawn_hermes_action`` launches it
|
||||
# with start_new_session=True, so it outlives the pytest worker; the
|
||||
# child inherits the pytest-tmp HERMES_HOME, resolves the DEVELOPER's
|
||||
# ``hermes-gateway`` systemd unit (a tmp home hashes to no profile
|
||||
# suffix), restarts the live gateway, and the survivors squat the
|
||||
# webhook port. 2026-09-03: 39 such orphans lived 6 days after a
|
||||
# sibling refactor moved the spawn seam and left tests patching the
|
||||
# facade. The canonical matcher, never an argv substring.
|
||||
from gateway.status import _gateway_command_subcommand
|
||||
# A gateway launched INSIDE a container (`docker exec … hermes gateway start`) cannot
|
||||
# reach the host's systemd unit or webhook port; tests/docker/ exists to exercise it.
|
||||
in_container = _first_token_basename(cmd_str) in _CONTAINER_RUNTIMES
|
||||
if (
|
||||
not lookalike_ok
|
||||
and not in_container
|
||||
and _gateway_command_subcommand(cmd_str) in ("run", "start", "restart")
|
||||
):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — this would spawn a REAL "
|
||||
"hermes gateway runtime that outlives the test (it is "
|
||||
"detached), restarts the developer's live gateway, and "
|
||||
"holds the webhook port. Patch the spawn seam where "
|
||||
"production reads it (hermes_cli.web_server_gateway."
|
||||
"_spawn_hermes_action), or mark with "
|
||||
"@pytest.mark.spawns_gateway_lookalike a test that spawns "
|
||||
"and reaps its own stub child."
|
||||
)
|
||||
|
||||
def _wrap_subprocess(name, real):
|
||||
def _guarded(cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd(name, cmd, kwargs)
|
||||
return real(cmd, *args, **kwargs)
|
||||
_guarded.__name__ = f"_guarded_{name}"
|
||||
# Make the wrapper subscriptable like the wrapped callable when
|
||||
# the wrapped object is. ``subprocess.Popen[bytes]`` is used as
|
||||
# a type annotation in third-party packages (mcp, etc.); replacing
|
||||
# ``Popen`` with a plain function breaks ``Popen[bytes]`` at
|
||||
# import time. Defer ``__class_getitem__`` to the original.
|
||||
if hasattr(real, "__class_getitem__"):
|
||||
_guarded.__class_getitem__ = real.__class_getitem__
|
||||
return _guarded
|
||||
|
||||
def _wrap_popen():
|
||||
"""Subclass Popen so isinstance checks AND Popen[bytes] still work."""
|
||||
real = _subprocess.Popen
|
||||
|
||||
class _GuardedPopen(real): # type: ignore[misc, valid-type]
|
||||
def __init__(self, cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd("Popen", cmd, kwargs)
|
||||
super().__init__(cmd, *args, **kwargs)
|
||||
|
||||
_GuardedPopen.__name__ = "Popen"
|
||||
_GuardedPopen.__qualname__ = "Popen"
|
||||
return _GuardedPopen
|
||||
|
||||
real_run = _subprocess.run
|
||||
real_popen = _subprocess.Popen
|
||||
real_call = _subprocess.call
|
||||
real_check_call = _subprocess.check_call
|
||||
real_check_output = _subprocess.check_output
|
||||
real_getoutput = _subprocess.getoutput
|
||||
real_getstatusoutput = _subprocess.getstatusoutput
|
||||
|
||||
monkeypatch.setattr(_subprocess, "run", _wrap_subprocess("run", real_run))
|
||||
monkeypatch.setattr(_subprocess, "Popen", _wrap_popen())
|
||||
monkeypatch.setattr(_subprocess, "call", _wrap_subprocess("call", real_call))
|
||||
monkeypatch.setattr(
|
||||
_subprocess, "check_call", _wrap_subprocess("check_call", real_check_call)
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
_subprocess,
|
||||
"check_output",
|
||||
_wrap_subprocess("check_output", real_check_output),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
_subprocess, "getoutput", _wrap_subprocess("getoutput", real_getoutput)
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
_subprocess,
|
||||
"getstatusoutput",
|
||||
_wrap_subprocess("getstatusoutput", real_getstatusoutput),
|
||||
)
|
||||
|
||||
# os.system / os.popen — same risk class, completely unwrapped before.
|
||||
real_os_system = _os.system
|
||||
real_os_popen = _os.popen
|
||||
|
||||
def _guarded_os_system(command):
|
||||
_check_subprocess_cmd("os.system", command)
|
||||
return real_os_system(command)
|
||||
|
||||
def _guarded_os_popen(cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd("os.popen", cmd, kwargs)
|
||||
return real_os_popen(cmd, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_os, "system", _guarded_os_system)
|
||||
monkeypatch.setattr(_os, "popen", _guarded_os_popen)
|
||||
|
||||
# pty.spawn — POSIX-only.
|
||||
try:
|
||||
import pty as _pty
|
||||
if hasattr(_pty, "spawn"):
|
||||
real_pty_spawn = _pty.spawn
|
||||
|
||||
def _guarded_pty_spawn(argv, *args, **kwargs):
|
||||
_check_subprocess_cmd("pty.spawn", argv, kwargs)
|
||||
return real_pty_spawn(argv, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_pty, "spawn", _guarded_pty_spawn)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# asyncio.create_subprocess_* — bypasses subprocess module entirely.
|
||||
try:
|
||||
import asyncio as _asyncio
|
||||
real_async_exec = _asyncio.create_subprocess_exec
|
||||
real_async_shell = _asyncio.create_subprocess_shell
|
||||
|
||||
async def _guarded_async_exec(program, *args, **kwargs):
|
||||
_check_subprocess_cmd(
|
||||
"asyncio.create_subprocess_exec", [program, *args], kwargs
|
||||
)
|
||||
return await real_async_exec(program, *args, **kwargs)
|
||||
|
||||
async def _guarded_async_shell(cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd("asyncio.create_subprocess_shell", cmd, kwargs)
|
||||
return await real_async_shell(cmd, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_asyncio, "create_subprocess_exec", _guarded_async_exec)
|
||||
monkeypatch.setattr(
|
||||
_asyncio, "create_subprocess_shell", _guarded_async_shell
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
yield
|
||||
165
tests/_fixtures/platform_gating.py
Normal file
165
tests/_fixtures/platform_gating.py
Normal file
@@ -0,0 +1,165 @@
|
||||
"""Host-OS gating for ``@pytest.mark.platforms(...)``, applied by the conftest collection hook."""
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# OS gating
|
||||
#
|
||||
# Hermes runs on Linux, macOS and native Windows, and a lot of its behaviour
|
||||
# genuinely differs per host: PTY vs pywinpty, taskkill vs SIGTERM, launchd
|
||||
# vs systemd, Keychain vs libsecret, ``%LOCALAPPDATA%`` vs ``~/.hermes``.
|
||||
#
|
||||
# Historically those code paths were tested by *faking* the host — patching
|
||||
# ``sys.platform`` to ``"win32"`` inside a Linux CI job. That gives a green
|
||||
# test on a machine where the code under test could not actually run: the
|
||||
# fake covers the ``if sys.platform == "win32"`` branch selection but nothing
|
||||
# underneath it (``msvcrt`` still isn't importable, ``taskkill`` still isn't
|
||||
# on PATH, paths are still POSIX, ``signal.SIGKILL`` still exists). The
|
||||
# result was tests that pass on Linux and tell us nothing about Windows.
|
||||
#
|
||||
# So: a test whose subject is genuinely OS-specific declares the OS it
|
||||
# belongs to and runs there for real —
|
||||
#
|
||||
# @pytest.mark.platforms("windows") → only on native Windows (``sys.platform == "win32"``)
|
||||
# @pytest.mark.platforms("macos") → only on macOS (``sys.platform == "darwin"``)
|
||||
# @pytest.mark.platforms("linux") → only on Linux (``sys.platform.startswith("linux")``)
|
||||
#
|
||||
# Elsewhere the test is skipped, not faked. CI runs a dedicated macOS job
|
||||
# (``-m platforms("macos")``) and a dedicated Windows job (``-m platforms("windows")``) so
|
||||
# those markers are actually exercised on their own host rather than
|
||||
# quietly skipped everywhere.
|
||||
#
|
||||
# This does NOT mean every mention of another platform must be gated. Two
|
||||
# things are legitimately host-independent and stay on the Linux runner:
|
||||
#
|
||||
# • Pure functions that TAKE a platform as data — e.g.
|
||||
# ``hidden_windows_child_options(opts, is_windows=True)`` or a
|
||||
# ``resolve_launcher(platform_name)`` helper. Passing "win32" as an
|
||||
# argument is not faking the host; the function's whole contract is
|
||||
# that it maps input to output.
|
||||
# • Declaration/packaging invariants — e.g. "pyproject declares tzdata
|
||||
# with a ``sys_platform == 'win32'`` marker". That's an assertion about
|
||||
# a file, not about runtime behaviour.
|
||||
#
|
||||
# The line is: if the test needs the interpreter to BELIEVE it is on
|
||||
# another OS in order to pass, it belongs on that OS.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_PLATFORM_ALIASES = {
|
||||
"linux": ("linux",),
|
||||
"macos": ("darwin", "macos"),
|
||||
"windows": ("win32", "windows"),
|
||||
"posix": ("linux", "darwin"),
|
||||
"any": (),
|
||||
}
|
||||
|
||||
|
||||
def _platform_machine() -> str:
|
||||
import platform as _platform
|
||||
|
||||
machine = (_platform.machine() or "").lower()
|
||||
return {"amd64": "x86_64", "x86": "x86_64", "aarch64": "arm64"}.get(machine, machine)
|
||||
|
||||
|
||||
def _host_matches_platforms(conditions, arch=None, arch_negate=False):
|
||||
"""Evaluate a platforms() marker payload against the running host.
|
||||
|
||||
Returns ``(ok, skip_reason)``.
|
||||
"""
|
||||
host = sys.platform.lower()
|
||||
machine = _platform_machine()
|
||||
specs = [str(c).strip().lower() for c in conditions if str(c).strip()]
|
||||
if not specs:
|
||||
return True, "platforms() with no specs matches every host"
|
||||
# An unknown spec is a collection error, never a skip: a typo like
|
||||
# platforms("linx") would otherwise drop the test on every host while
|
||||
# both lanes stay green — the exact failure the gate exists to catch.
|
||||
unknown = [spec for spec in specs if spec.removeprefix("not ").strip() not in _PLATFORM_ALIASES]
|
||||
if unknown:
|
||||
raise pytest.UsageError(
|
||||
f"platforms(): unknown spec(s) {', '.join(map(repr, unknown))} — valid: "
|
||||
f"{', '.join(sorted(_PLATFORM_ALIASES))}, each optionally prefixed with 'not '"
|
||||
)
|
||||
for spec in specs:
|
||||
negate = spec.startswith("not ")
|
||||
leaf = spec[4:].strip() if negate else spec
|
||||
wanted = _PLATFORM_ALIASES[leaf]
|
||||
matched = (not wanted) or host in wanted
|
||||
if negate:
|
||||
matched = not matched
|
||||
if matched:
|
||||
break
|
||||
else:
|
||||
return False, f"platforms({', '.join(specs)}); host is {sys.platform}"
|
||||
if arch is not None:
|
||||
arch_l = str(arch).lower()
|
||||
arch_hit = machine == arch_l or (
|
||||
arch_l in {"arm64", "aarch64"} and machine == "arm64"
|
||||
)
|
||||
if arch_negate:
|
||||
arch_hit = not arch_hit
|
||||
if not arch_hit:
|
||||
return False, (
|
||||
f"platforms(arch={'not ' if arch_negate else ''}{arch}); "
|
||||
f"host machine is {machine or 'unknown'}"
|
||||
)
|
||||
return True, ""
|
||||
|
||||
|
||||
def _platforms_gate_reason(item):
|
||||
"""Skip reason when the item's platforms() gating excludes this host."""
|
||||
for mark in item.iter_markers("platforms"):
|
||||
kwargs = dict(mark.kwargs)
|
||||
conds = list(mark.args)
|
||||
try:
|
||||
ok, reason = _host_matches_platforms(
|
||||
conds,
|
||||
arch=kwargs.pop("arch", None),
|
||||
arch_negate=kwargs.pop("arch_negate", False),
|
||||
)
|
||||
except pytest.UsageError as exc:
|
||||
raise pytest.UsageError(f"{item.nodeid}: {exc}") from None
|
||||
if kwargs:
|
||||
raise pytest.UsageError(
|
||||
f"{item.nodeid}: platforms() got unexpected keyword(s) "
|
||||
f"{sorted(kwargs)} — valid: arch, arch_negate"
|
||||
)
|
||||
if not ok:
|
||||
return reason
|
||||
return None
|
||||
|
||||
|
||||
def _reject_contradictory_platform_marks(items):
|
||||
"""Fail collection when one test carries two platforms() markers.
|
||||
|
||||
Two markers are ANDed by the gate, so a stacked pair is not always wrong
|
||||
in principle — but the historic failure this guard exists for (a
|
||||
module-level gate stacking with a per-test gate so the test is skipped
|
||||
on every host while both lanes report green) is only diagnosable at
|
||||
collection time. A test that needs a compound condition writes ONE
|
||||
marker: platforms("linux", arch="arm64").
|
||||
"""
|
||||
offenders = []
|
||||
retired = []
|
||||
for item in items:
|
||||
marks = list(item.iter_markers("platforms"))
|
||||
if len(marks) > 1:
|
||||
offenders.append(f" {item.nodeid}: {len(marks)} platforms() marks")
|
||||
for legacy in ("linux_only", "macos_only", "windows_only"):
|
||||
if item.get_closest_marker(legacy):
|
||||
retired.append(f" {item.nodeid}: {legacy}")
|
||||
if retired:
|
||||
# An unregistered mark is a warning, so a merge that resurrects the old
|
||||
# trio would make a host-gated test RUN on every host, unnoticed.
|
||||
raise pytest.UsageError(
|
||||
"linux_only/macos_only/windows_only were replaced by platforms(...); rewrite:\n"
|
||||
+ "\n".join(retired)
|
||||
)
|
||||
if offenders:
|
||||
raise pytest.UsageError(
|
||||
"a test may carry at most one platforms() marker — combine the "
|
||||
'specs into one call (platforms("linux", arch="arm64") instead '
|
||||
"of stacking two markers); these carry several:\n"
|
||||
+ "\n".join(offenders)
|
||||
)
|
||||
@@ -230,329 +230,18 @@ if not HOST_LOCK_DIR_AT_CONFTEST_IMPORT:
|
||||
# See ``scripts/run_tests.sh`` for the runner.
|
||||
|
||||
|
||||
# ── Credential env-var filter ──────────────────────────────────────────────
|
||||
#
|
||||
# Any env var in the current process matching ONE of these patterns is
|
||||
# unset for every test. Developers' local keys cannot leak into assertions
|
||||
# about "auto-detect provider when key present".
|
||||
|
||||
_CREDENTIAL_SUFFIXES = (
|
||||
"_API_KEY",
|
||||
"_TOKEN",
|
||||
"_SECRET",
|
||||
"_PASSWORD",
|
||||
"_CREDENTIALS",
|
||||
"_ACCESS_KEY",
|
||||
"_SECRET_ACCESS_KEY",
|
||||
"_PRIVATE_KEY",
|
||||
"_OAUTH_TOKEN",
|
||||
"_WEBHOOK_SECRET",
|
||||
"_ENCRYPT_KEY",
|
||||
"_APP_SECRET",
|
||||
"_CLIENT_SECRET",
|
||||
"_CORP_SECRET",
|
||||
"_AES_KEY",
|
||||
# Topic modules split out to keep this file under the size gate. They are
|
||||
# imported rather than listed in ``pytest_plugins``: this is not the rootdir
|
||||
# conftest (that is the repo root), and pytest fails a run that loads a
|
||||
# non-root conftest carrying ``pytest_plugins`` after startup (e.g. ``pytest .``).
|
||||
# Fixtures imported here register exactly as if they were defined here.
|
||||
from tests._fixtures.env_filter import _HERMES_BEHAVIORAL_VARS, _looks_like_credential
|
||||
from tests._fixtures.live_system_guard import ( # noqa: F401 — _live_system_guard registers here
|
||||
_GATEWAY_LOOKALIKE_MARK,
|
||||
_LIVE_SYSTEM_GUARD_BYPASS_MARK,
|
||||
_live_system_guard,
|
||||
)
|
||||
|
||||
# Explicit names (for ones that don't fit the suffix pattern)
|
||||
_CREDENTIAL_NAMES = frozenset({
|
||||
"AWS_ACCESS_KEY_ID",
|
||||
"AWS_SECRET_ACCESS_KEY",
|
||||
"AWS_SESSION_TOKEN",
|
||||
"ANTHROPIC_TOKEN",
|
||||
"FAL_KEY",
|
||||
"GH_TOKEN",
|
||||
"GITHUB_TOKEN",
|
||||
"OPENAI_API_KEY",
|
||||
"OPENROUTER_API_KEY",
|
||||
"NOUS_API_KEY",
|
||||
"GEMINI_API_KEY",
|
||||
"GOOGLE_API_KEY",
|
||||
"GROQ_API_KEY",
|
||||
"XAI_API_KEY",
|
||||
"MISTRAL_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"KIMI_API_KEY",
|
||||
"MOONSHOT_API_KEY",
|
||||
"GLM_API_KEY",
|
||||
"ZAI_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"OLLAMA_API_KEY",
|
||||
"OPENVIKING_API_KEY",
|
||||
"COPILOT_API_KEY",
|
||||
"CLAUDE_CODE_OAUTH_TOKEN",
|
||||
"BROWSERBASE_API_KEY",
|
||||
"FIRECRAWL_API_KEY",
|
||||
"PARALLEL_API_KEY",
|
||||
"EXA_API_KEY",
|
||||
"TAVILY_API_KEY",
|
||||
"PERPLEXITY_API_KEY",
|
||||
"WANDB_API_KEY",
|
||||
"ELEVENLABS_API_KEY",
|
||||
"HONCHO_API_KEY",
|
||||
"MEM0_API_KEY",
|
||||
"SUPERMEMORY_API_KEY",
|
||||
"RETAINDB_API_KEY",
|
||||
"HINDSIGHT_API_KEY",
|
||||
"HINDSIGHT_LLM_API_KEY",
|
||||
"DAYTONA_API_KEY",
|
||||
"TWILIO_AUTH_TOKEN",
|
||||
"TELEGRAM_BOT_TOKEN",
|
||||
"DISCORD_BOT_TOKEN",
|
||||
"SLACK_BOT_TOKEN",
|
||||
"SLACK_APP_TOKEN",
|
||||
"MATTERMOST_TOKEN",
|
||||
"MATRIX_ACCESS_TOKEN",
|
||||
"MATRIX_PASSWORD",
|
||||
"MATRIX_RECOVERY_KEY",
|
||||
"HASS_TOKEN",
|
||||
"EMAIL_PASSWORD",
|
||||
"BLUEBUBBLES_PASSWORD",
|
||||
"FEISHU_APP_SECRET",
|
||||
"FEISHU_ENCRYPT_KEY",
|
||||
"FEISHU_VERIFICATION_TOKEN",
|
||||
"DINGTALK_CLIENT_SECRET",
|
||||
"QQ_CLIENT_SECRET",
|
||||
"QQ_STT_API_KEY",
|
||||
"WECOM_SECRET",
|
||||
"WECOM_CALLBACK_CORP_SECRET",
|
||||
"WECOM_CALLBACK_TOKEN",
|
||||
"WECOM_CALLBACK_ENCODING_AES_KEY",
|
||||
"WEIXIN_TOKEN",
|
||||
"MODAL_TOKEN_ID",
|
||||
"MODAL_TOKEN_SECRET",
|
||||
"TERMINAL_SSH_KEY",
|
||||
"SUDO_PASSWORD",
|
||||
"GATEWAY_PROXY_KEY",
|
||||
"API_SERVER_KEY",
|
||||
"TOOL_GATEWAY_USER_TOKEN",
|
||||
"TELEGRAM_WEBHOOK_SECRET",
|
||||
"WEBHOOK_SECRET",
|
||||
"AI_GATEWAY_API_KEY",
|
||||
"VOICE_TOOLS_OPENAI_KEY",
|
||||
"BROWSER_USE_API_KEY",
|
||||
"CUSTOM_API_KEY",
|
||||
"GATEWAY_PROXY_URL",
|
||||
"GEMINI_BASE_URL",
|
||||
"OPENAI_BASE_URL",
|
||||
"OPENROUTER_BASE_URL",
|
||||
"OLLAMA_BASE_URL",
|
||||
"GROQ_BASE_URL",
|
||||
"XAI_BASE_URL",
|
||||
"AI_GATEWAY_BASE_URL",
|
||||
"ANTHROPIC_BASE_URL",
|
||||
})
|
||||
|
||||
|
||||
def _looks_like_credential(name: str) -> bool:
|
||||
"""True if env var name matches a credential-shaped pattern."""
|
||||
if name in _CREDENTIAL_NAMES:
|
||||
return True
|
||||
return any(name.endswith(suf) for suf in _CREDENTIAL_SUFFIXES)
|
||||
|
||||
|
||||
# HERMES_* vars that change test behavior by being set. Unset all of these
|
||||
# unconditionally — individual tests that need them set do so explicitly.
|
||||
_HERMES_BEHAVIORAL_VARS = frozenset({
|
||||
# Voice/TTS runtime flags. ``tui_gateway/server.py`` reads these straight
|
||||
# off ``os.environ`` at call time (``_voice_mode_enabled`` /
|
||||
# ``_voice_tts_enabled``) and, on every completed turn, hands the turn's
|
||||
# final response text to ``hermes_cli.voice.speak_text`` — real synthesis,
|
||||
# real playback, out of the developer's speakers. Blank them per-test so a
|
||||
# leak (from the shell, or from an earlier test that drove the
|
||||
# ``voice.toggle`` RPC, which writes ``os.environ`` directly) cannot carry
|
||||
# into the next test. See ``_audio_playback_guard`` for the second layer.
|
||||
"HERMES_VOICE",
|
||||
"HERMES_VOICE_TTS",
|
||||
"HERMES_YOLO_MODE",
|
||||
# Injected into subprocess envs by the terminal tool (_make_run_env), so
|
||||
# any test run launched FROM a Hermes agent session inherits them and
|
||||
# hermes_constants home-resolution helpers prefer them over monkeypatched
|
||||
# HOME (test_subprocess_home_isolation red locally, green on CI).
|
||||
"HERMES_REAL_HOME",
|
||||
"TERMINAL_HOME_MODE",
|
||||
"HERMES_INTERACTIVE",
|
||||
"HERMES_QUIET",
|
||||
"HERMES_TOOL_PROGRESS",
|
||||
"HERMES_TOOL_PROGRESS_MODE",
|
||||
"HERMES_MAX_ITERATIONS",
|
||||
"HERMES_SESSION_PLATFORM",
|
||||
"HERMES_SESSION_CHAT_ID",
|
||||
"HERMES_SESSION_CHAT_NAME",
|
||||
"HERMES_SESSION_CHAT_TYPE",
|
||||
"HERMES_SESSION_THREAD_ID",
|
||||
"HERMES_SESSION_SOURCE",
|
||||
"HERMES_SESSION_KEY",
|
||||
"HERMES_GATEWAY_SESSION",
|
||||
"HERMES_CRON_SESSION",
|
||||
"_HERMES_GATEWAY",
|
||||
"HERMES_PLATFORM",
|
||||
"HERMES_MODEL",
|
||||
"HERMES_INFERENCE_MODEL",
|
||||
"HERMES_INFERENCE_PROVIDER",
|
||||
"HERMES_TUI_PROVIDER",
|
||||
"HERMES_MANAGED",
|
||||
"HERMES_MANAGED_DIR",
|
||||
# A Nix-wrapped `hermes` on the developer's host exports the store's read-only plugins
|
||||
# tree; tests must discover the checkout's plugins/ (get_bundled_plugins_dir), not a
|
||||
# different release's.
|
||||
"HERMES_BUNDLED_PLUGINS",
|
||||
"HERMES_DEV",
|
||||
"HERMES_CONTAINER",
|
||||
"HERMES_EPHEMERAL_SYSTEM_PROMPT",
|
||||
"HERMES_TIMEZONE",
|
||||
"HERMES_REDACT_SECRETS",
|
||||
"HERMES_BACKGROUND_NOTIFICATIONS",
|
||||
"HERMES_EXEC_ASK",
|
||||
"HERMES_HOME_MODE",
|
||||
"HERMES_AGENT_USE_LEGACY_SESSION_KEYS",
|
||||
# Kanban path/board pins must never leak from a developer shell or
|
||||
# dispatched worker into tests; otherwise tests can write fake tasks to
|
||||
# the real ~/.hermes/kanban.db instead of the per-test HERMES_HOME.
|
||||
"HERMES_KANBAN_DB",
|
||||
"HERMES_KANBAN_BOARD",
|
||||
"HERMES_KANBAN_HOME",
|
||||
"HERMES_KANBAN_WORKSPACES_ROOT",
|
||||
"HERMES_KANBAN_LOGS_ROOT",
|
||||
"HERMES_KANBAN_TASK",
|
||||
"HERMES_KANBAN_WORKSPACE",
|
||||
"HERMES_KANBAN_RUN_ID",
|
||||
"HERMES_KANBAN_CLAIM_LOCK",
|
||||
"HERMES_KANBAN_DISPATCH_IN_GATEWAY",
|
||||
# Pytest is routinely launched from a delegated worker. The worker
|
||||
# lineage marker must not make parent-state tests run as delegated
|
||||
# children; tests that exercise child behavior set it explicitly.
|
||||
"HERMES_DELEGATED_CHILD_CONTEXT",
|
||||
"HERMES_TENANT",
|
||||
# Honcho host selection changes which nested config block wins. A local
|
||||
# shell override leaked "myhost" into the full suite and flipped 20
|
||||
# otherwise-unrelated config tests away from the default "hermes" host.
|
||||
"HERMES_HONCHO_HOST",
|
||||
# Dashboard OAuth auth gate (PR #30156). When set, the bundled
|
||||
# dashboard-auth `nous` plugin auto-registers itself on plugin discovery,
|
||||
# which is triggered by any `/api/status` call. That leaks a provider
|
||||
# into the dashboard_auth registry across tests in the same worker and
|
||||
# makes assertions like `auth_providers == []` flaky. CI never sets
|
||||
# these, so production tests must not see them either.
|
||||
"HERMES_DASHBOARD_OAUTH_CLIENT_ID",
|
||||
"HERMES_DASHBOARD_PORTAL_URL",
|
||||
"TERMINAL_CWD",
|
||||
"TERMINAL_ENV",
|
||||
"TERMINAL_VERCEL_RUNTIME",
|
||||
"TERMINAL_CONTAINER_CPU",
|
||||
"TERMINAL_CONTAINER_DISK",
|
||||
"TERMINAL_CONTAINER_MEMORY",
|
||||
"TERMINAL_CONTAINER_PERSISTENT",
|
||||
"TERMINAL_DOCKER_PERSIST_ACROSS_PROCESSES",
|
||||
"TERMINAL_DOCKER_ORPHAN_REAPER",
|
||||
"TERMINAL_DOCKER_RUN_AS_HOST_USER",
|
||||
"BROWSER_CDP_URL",
|
||||
"CAMOFOX_URL",
|
||||
# Platform allowlists — not credentials, but if set from any source
|
||||
# (user shell, earlier leaky test, CI env), they change gateway auth
|
||||
# behavior and flake button-authorization tests.
|
||||
"TELEGRAM_ALLOWED_USERS",
|
||||
"TELEGRAM_GROUP_ALLOWED_USERS",
|
||||
"TELEGRAM_GROUP_ALLOWED_CHATS",
|
||||
"QQ_ALLOWED_USERS",
|
||||
"QQ_GROUP_ALLOWED_USERS",
|
||||
"DISCORD_ALLOWED_USERS",
|
||||
"WHATSAPP_ALLOWED_USERS",
|
||||
"SLACK_ALLOWED_USERS",
|
||||
"SIGNAL_ALLOWED_USERS",
|
||||
"SIGNAL_GROUP_ALLOWED_USERS",
|
||||
"EMAIL_ALLOWED_USERS",
|
||||
"SMS_ALLOWED_USERS",
|
||||
"MATTERMOST_ALLOWED_USERS",
|
||||
"MATRIX_ALLOWED_USERS",
|
||||
"DINGTALK_ALLOWED_USERS",
|
||||
"FEISHU_ALLOWED_USERS",
|
||||
"WECOM_ALLOWED_USERS",
|
||||
"PHOTON_ALLOWED_USERS",
|
||||
"GATEWAY_ALLOWED_USERS",
|
||||
"GATEWAY_ALLOW_ALL_USERS",
|
||||
"TELEGRAM_ALLOW_ALL_USERS",
|
||||
"DISCORD_ALLOW_ALL_USERS",
|
||||
"WHATSAPP_ALLOW_ALL_USERS",
|
||||
"SLACK_ALLOW_ALL_USERS",
|
||||
"SIGNAL_ALLOW_ALL_USERS",
|
||||
"EMAIL_ALLOW_ALL_USERS",
|
||||
"SMS_ALLOW_ALL_USERS",
|
||||
"PHOTON_ALLOW_ALL_USERS",
|
||||
# Gateway home channels are set by /sethome in real profiles. Tests that
|
||||
# exercise dashboard notification toggles must opt in explicitly or they
|
||||
# can accidentally subscribe against a developer's real home channel.
|
||||
"TELEGRAM_HOME_CHANNEL",
|
||||
"TELEGRAM_HOME_CHANNEL_THREAD_ID",
|
||||
"TELEGRAM_HOME_CHANNEL_NAME",
|
||||
"TELEGRAM_CRON_THREAD_ID",
|
||||
"DISCORD_HOME_CHANNEL",
|
||||
"DISCORD_HOME_CHANNEL_THREAD_ID",
|
||||
"DISCORD_HOME_CHANNEL_NAME",
|
||||
"SLACK_HOME_CHANNEL",
|
||||
"SLACK_HOME_CHANNEL_THREAD_ID",
|
||||
"SLACK_HOME_CHANNEL_NAME",
|
||||
"WHATSAPP_HOME_CHANNEL",
|
||||
"WHATSAPP_HOME_CHANNEL_THREAD_ID",
|
||||
"WHATSAPP_HOME_CHANNEL_NAME",
|
||||
"SIGNAL_HOME_CHANNEL",
|
||||
"SIGNAL_HOME_CHANNEL_THREAD_ID",
|
||||
"SIGNAL_HOME_CHANNEL_NAME",
|
||||
"EMAIL_HOME_CHANNEL",
|
||||
"EMAIL_HOME_CHANNEL_THREAD_ID",
|
||||
"EMAIL_HOME_CHANNEL_NAME",
|
||||
"SMS_HOME_CHANNEL",
|
||||
"SMS_HOME_CHANNEL_THREAD_ID",
|
||||
"SMS_HOME_CHANNEL_NAME",
|
||||
"MATTERMOST_HOME_CHANNEL",
|
||||
"MATTERMOST_HOME_CHANNEL_THREAD_ID",
|
||||
"MATTERMOST_HOME_CHANNEL_NAME",
|
||||
"MATRIX_HOME_CHANNEL",
|
||||
"MATRIX_HOME_CHANNEL_THREAD_ID",
|
||||
"MATRIX_HOME_CHANNEL_NAME",
|
||||
"DINGTALK_HOME_CHANNEL",
|
||||
"DINGTALK_HOME_CHANNEL_THREAD_ID",
|
||||
"DINGTALK_HOME_CHANNEL_NAME",
|
||||
"FEISHU_HOME_CHANNEL",
|
||||
"FEISHU_HOME_CHANNEL_THREAD_ID",
|
||||
"FEISHU_HOME_CHANNEL_NAME",
|
||||
"WECOM_HOME_CHANNEL",
|
||||
"WECOM_HOME_CHANNEL_THREAD_ID",
|
||||
"WECOM_HOME_CHANNEL_NAME",
|
||||
"PHOTON_HOME_CHANNEL",
|
||||
"PHOTON_HOME_CHANNEL_THREAD_ID",
|
||||
"PHOTON_HOME_CHANNEL_NAME",
|
||||
# API server bind/auth settings are common in local gateway profiles and
|
||||
# change adapter defaults plus load_gateway_config() enablement. Tests that
|
||||
# need them set opt in explicitly with monkeypatch.
|
||||
"API_SERVER_ENABLED",
|
||||
"API_SERVER_HOST",
|
||||
"API_SERVER_PORT",
|
||||
"API_SERVER_KEY",
|
||||
"API_SERVER_CORS_ORIGINS",
|
||||
"API_SERVER_MODEL_NAME",
|
||||
# Platform gating — set by load_gateway_config() as a side effect when
|
||||
# a config.yaml is present, so individual test bodies that call the
|
||||
# loader leak these values into later tests in the same process.
|
||||
# Force-clear on every test setup so the leak can't happen.
|
||||
"SLACK_REQUIRE_MENTION",
|
||||
"SLACK_STRICT_MENTION",
|
||||
"SLACK_THREAD_REQUIRE_MENTION",
|
||||
"SLACK_IGNORE_OTHER_USER_MENTIONS",
|
||||
"SLACK_REQUIRE_MENTION_CHANNELS",
|
||||
"SLACK_FREE_RESPONSE_CHANNELS",
|
||||
"SLACK_ALLOWED_CHANNELS",
|
||||
"SLACK_IGNORED_CHANNELS",
|
||||
"SLACK_DISABLE_DMS",
|
||||
"SLACK_ALLOW_BOTS",
|
||||
"SLACK_REACTIONS",
|
||||
"DISCORD_REQUIRE_MENTION",
|
||||
"DISCORD_FREE_RESPONSE_CHANNELS",
|
||||
"TELEGRAM_REQUIRE_MENTION",
|
||||
"WHATSAPP_REQUIRE_MENTION",
|
||||
"DINGTALK_REQUIRE_MENTION",
|
||||
"MATRIX_REQUIRE_MENTION",
|
||||
})
|
||||
from tests._fixtures.platform_gating import _platforms_gate_reason, _reject_contradictory_platform_marks
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
@@ -1209,42 +898,8 @@ def _ensure_current_event_loop(request):
|
||||
asyncio.set_event_loop(None)
|
||||
|
||||
|
||||
# ── Live-system guard ──────────────────────────────────────────────────────
|
||||
#
|
||||
# Several test files exercise the gateway-restart / kill code paths
|
||||
# (``cmd_update``, ``kill_gateway_processes``, ``stop_profile_gateway``).
|
||||
# When a single test forgets to mock either ``os.kill`` or the global
|
||||
# ``find_gateway_pids`` helper, the real call leaks out of the hermetic
|
||||
# environment and finds the developer's live ``hermes-gateway`` process
|
||||
# via ``psutil`` — sending it SIGTERM mid-test. The shutdown forensics in
|
||||
# PR #23285 caught this happening 5+ times in 3 days, every time
|
||||
# correlated with a ``tests/hermes_cli/`` pytest run starting up.
|
||||
#
|
||||
# This fixture makes the leak impossible by intercepting the two
|
||||
# primitives that actually do damage:
|
||||
#
|
||||
# • ``os.kill`` rejects any PID outside the test process subtree with
|
||||
# a hard ``RuntimeError`` so the offending test gets a stack trace
|
||||
# instead of silently murdering the real gateway.
|
||||
# • ``subprocess.run`` / ``subprocess.Popen`` / ``call`` / ``check_call`` /
|
||||
# ``check_output`` reject any ``systemctl ... <verb> hermes-gateway``
|
||||
# invocation that would mutate the live unit. Read-only systemctl
|
||||
# calls (``status``, ``show``, ``list-units``) still pass through.
|
||||
#
|
||||
# We intentionally do NOT stub ``find_gateway_pids`` / ``_scan_gateway_pids``
|
||||
# here — tests of those functions themselves need the real implementation.
|
||||
# Even if a test gets the live gateway PID back from a real scan, the
|
||||
# ``os.kill`` guard above catches the actual signal call, and the
|
||||
# ``systemctl`` guard catches the systemd path. Discovery without
|
||||
# delivery is harmless.
|
||||
|
||||
_LIVE_SYSTEM_GUARD_BYPASS_MARK = "live_system_guard_bypass"
|
||||
_GATEWAY_LOOKALIKE_MARK = "spawns_gateway_lookalike"
|
||||
_REQUIRES_WAL_MARK = "requires_wal"
|
||||
|
||||
# Tests may designate a temporary repo to exercise the real guard safely.
|
||||
_LIVE_GUARD_PROTECTED_GIT_ROOTS = (PROJECT_ROOT,)
|
||||
|
||||
|
||||
def _wal_is_usable() -> bool:
|
||||
"""True when Hermes will actually put a database into WAL mode here.
|
||||
@@ -1285,7 +940,8 @@ def _wal_is_usable() -> bool:
|
||||
|
||||
# ── Audio-playback guard ───────────────────────────────────────────────────
|
||||
#
|
||||
# Same class of incident as the live-system guard above, different primitive:
|
||||
# Same class of incident as the live-system guard (``tests/_fixtures/live_system_guard.py``),
|
||||
# different primitive:
|
||||
# a test run spoke the string "partial answer complete" out of the developer's
|
||||
# speakers. That string is a test fixture
|
||||
# (``tests/tui_gateway/test_tui_gateway_server.py``'s fake ``final_response``), and the
|
||||
@@ -1328,132 +984,6 @@ def _wal_is_usable() -> bool:
|
||||
_AUDIO_GUARD_BYPASS_MARK = "real_audio_playback"
|
||||
_ALLOW_MACOS_KEYCHAIN_MARK = "allow_macos_keychain"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# OS gating
|
||||
#
|
||||
# Hermes runs on Linux, macOS and native Windows, and a lot of its behaviour
|
||||
# genuinely differs per host: PTY vs pywinpty, taskkill vs SIGTERM, launchd
|
||||
# vs systemd, Keychain vs libsecret, ``%LOCALAPPDATA%`` vs ``~/.hermes``.
|
||||
#
|
||||
# Historically those code paths were tested by *faking* the host — patching
|
||||
# ``sys.platform`` to ``"win32"`` inside a Linux CI job. That gives a green
|
||||
# test on a machine where the code under test could not actually run: the
|
||||
# fake covers the ``if sys.platform == "win32"`` branch selection but nothing
|
||||
# underneath it (``msvcrt`` still isn't importable, ``taskkill`` still isn't
|
||||
# on PATH, paths are still POSIX, ``signal.SIGKILL`` still exists). The
|
||||
# result was tests that pass on Linux and tell us nothing about Windows.
|
||||
#
|
||||
# So: a test whose subject is genuinely OS-specific declares the OS it
|
||||
# belongs to and runs there for real —
|
||||
#
|
||||
# @pytest.mark.platforms("windows") → only on native Windows (``sys.platform == "win32"``)
|
||||
# @pytest.mark.platforms("macos") → only on macOS (``sys.platform == "darwin"``)
|
||||
# @pytest.mark.platforms("linux") → only on Linux (``sys.platform.startswith("linux")``)
|
||||
#
|
||||
# Elsewhere the test is skipped, not faked. CI runs a dedicated macOS job
|
||||
# (``-m platforms("macos")``) and a dedicated Windows job (``-m platforms("windows")``) so
|
||||
# those markers are actually exercised on their own host rather than
|
||||
# quietly skipped everywhere.
|
||||
#
|
||||
# This does NOT mean every mention of another platform must be gated. Two
|
||||
# things are legitimately host-independent and stay on the Linux runner:
|
||||
#
|
||||
# • Pure functions that TAKE a platform as data — e.g.
|
||||
# ``hidden_windows_child_options(opts, is_windows=True)`` or a
|
||||
# ``resolve_launcher(platform_name)`` helper. Passing "win32" as an
|
||||
# argument is not faking the host; the function's whole contract is
|
||||
# that it maps input to output.
|
||||
# • Declaration/packaging invariants — e.g. "pyproject declares tzdata
|
||||
# with a ``sys_platform == 'win32'`` marker". That's an assertion about
|
||||
# a file, not about runtime behaviour.
|
||||
#
|
||||
# The line is: if the test needs the interpreter to BELIEVE it is on
|
||||
# another OS in order to pass, it belongs on that OS.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_PLATFORM_ALIASES = {
|
||||
"linux": ("linux",),
|
||||
"macos": ("darwin", "macos"),
|
||||
"windows": ("win32", "windows"),
|
||||
"posix": ("linux", "darwin"),
|
||||
"any": (),
|
||||
}
|
||||
|
||||
|
||||
def _platform_machine() -> str:
|
||||
import platform as _platform
|
||||
|
||||
machine = (_platform.machine() or "").lower()
|
||||
return {"amd64": "x86_64", "x86": "x86_64", "aarch64": "arm64"}.get(machine, machine)
|
||||
|
||||
|
||||
def _host_matches_platforms(conditions, arch=None, arch_negate=False):
|
||||
"""Evaluate a platforms() marker payload against the running host.
|
||||
|
||||
Returns ``(ok, skip_reason)``.
|
||||
"""
|
||||
host = sys.platform.lower()
|
||||
machine = _platform_machine()
|
||||
specs = [str(c).strip().lower() for c in conditions if str(c).strip()]
|
||||
if not specs:
|
||||
return True, "platforms() with no specs matches every host"
|
||||
# An unknown spec is a collection error, never a skip: a typo like
|
||||
# platforms("linx") would otherwise drop the test on every host while
|
||||
# both lanes stay green — the exact failure the gate exists to catch.
|
||||
unknown = [spec for spec in specs if spec.removeprefix("not ").strip() not in _PLATFORM_ALIASES]
|
||||
if unknown:
|
||||
raise pytest.UsageError(
|
||||
f"platforms(): unknown spec(s) {', '.join(map(repr, unknown))} — valid: "
|
||||
f"{', '.join(sorted(_PLATFORM_ALIASES))}, each optionally prefixed with 'not '"
|
||||
)
|
||||
for spec in specs:
|
||||
negate = spec.startswith("not ")
|
||||
leaf = spec[4:].strip() if negate else spec
|
||||
wanted = _PLATFORM_ALIASES[leaf]
|
||||
matched = (not wanted) or host in wanted
|
||||
if negate:
|
||||
matched = not matched
|
||||
if matched:
|
||||
break
|
||||
else:
|
||||
return False, f"platforms({', '.join(specs)}); host is {sys.platform}"
|
||||
if arch is not None:
|
||||
arch_l = str(arch).lower()
|
||||
arch_hit = machine == arch_l or (
|
||||
arch_l in {"arm64", "aarch64"} and machine == "arm64"
|
||||
)
|
||||
if arch_negate:
|
||||
arch_hit = not arch_hit
|
||||
if not arch_hit:
|
||||
return False, (
|
||||
f"platforms(arch={'not ' if arch_negate else ''}{arch}); "
|
||||
f"host machine is {machine or 'unknown'}"
|
||||
)
|
||||
return True, ""
|
||||
|
||||
|
||||
def _platforms_gate_reason(item):
|
||||
"""Skip reason when the item's platforms() gating excludes this host."""
|
||||
for mark in item.iter_markers("platforms"):
|
||||
kwargs = dict(mark.kwargs)
|
||||
conds = list(mark.args)
|
||||
try:
|
||||
ok, reason = _host_matches_platforms(
|
||||
conds,
|
||||
arch=kwargs.pop("arch", None),
|
||||
arch_negate=kwargs.pop("arch_negate", False),
|
||||
)
|
||||
except pytest.UsageError as exc:
|
||||
raise pytest.UsageError(f"{item.nodeid}: {exc}") from None
|
||||
if kwargs:
|
||||
raise pytest.UsageError(
|
||||
f"{item.nodeid}: platforms() got unexpected keyword(s) "
|
||||
f"{sorted(kwargs)} — valid: arch, arch_negate"
|
||||
)
|
||||
if not ok:
|
||||
return reason
|
||||
return None
|
||||
|
||||
|
||||
def _relocate_basetemp_outside_operator_home(config) -> None:
|
||||
"""Move pytest's basetemp out of the operator's platform-native Hermes home.
|
||||
@@ -1682,47 +1212,12 @@ def pytest_runtest_setup(item):
|
||||
)
|
||||
|
||||
|
||||
def _reject_contradictory_platform_marks(items):
|
||||
"""Fail collection when one test carries two platforms() markers.
|
||||
|
||||
Two markers are ANDed by the gate, so a stacked pair is not always wrong
|
||||
in principle — but the historic failure this guard exists for (a
|
||||
module-level gate stacking with a per-test gate so the test is skipped
|
||||
on every host while both lanes report green) is only diagnosable at
|
||||
collection time. A test that needs a compound condition writes ONE
|
||||
marker: platforms("linux", arch="arm64").
|
||||
"""
|
||||
offenders = []
|
||||
retired = []
|
||||
for item in items:
|
||||
marks = list(item.iter_markers("platforms"))
|
||||
if len(marks) > 1:
|
||||
offenders.append(f" {item.nodeid}: {len(marks)} platforms() marks")
|
||||
for legacy in ("linux_only", "macos_only", "windows_only"):
|
||||
if item.get_closest_marker(legacy):
|
||||
retired.append(f" {item.nodeid}: {legacy}")
|
||||
if retired:
|
||||
# An unregistered mark is a warning, so a merge that resurrects the old
|
||||
# trio would make a host-gated test RUN on every host, unnoticed.
|
||||
raise pytest.UsageError(
|
||||
"linux_only/macos_only/windows_only were replaced by platforms(...); rewrite:\n"
|
||||
+ "\n".join(retired)
|
||||
)
|
||||
if offenders:
|
||||
raise pytest.UsageError(
|
||||
"a test may carry at most one platforms() marker — combine the "
|
||||
'specs into one call (platforms("linux", arch="arm64") instead '
|
||||
"of stacking two markers); these carry several:\n"
|
||||
+ "\n".join(offenders)
|
||||
)
|
||||
|
||||
|
||||
def pytest_collection_modifyitems(config, items): # noqa: D401 — pytest hook
|
||||
"""Apply host-OS gating, then skip ``requires_wal`` where WAL is unusable.
|
||||
|
||||
OS gating: a test marked ``platforms(...)`` runs only on hosts its
|
||||
specs match. See the platform-gating block comment above for why these
|
||||
tests are skipped rather than run against a patched ``sys.platform``.
|
||||
specs match. See the block comment in ``tests/_fixtures/platform_gating.py``
|
||||
for why these tests are skipped rather than run against a patched ``sys.platform``.
|
||||
|
||||
WAL gating is cheaper and more honest than each test hand-rolling a
|
||||
version check: the reason string names the actual linked version so the
|
||||
@@ -1752,424 +1247,6 @@ def pytest_collection_modifyitems(config, items): # noqa: D401 — pytest hook
|
||||
item.add_marker(skip_marker)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _live_system_guard(request, monkeypatch):
|
||||
"""Block real os.kill / systemctl / gateway-pid scans during tests.
|
||||
|
||||
See block comment above for the why. Tests that genuinely need
|
||||
real signal delivery (e.g. PTY tests that SIGINT their own child)
|
||||
can opt out with ``@pytest.mark.live_system_guard_bypass``.
|
||||
|
||||
Coverage (every primitive that can deliver a signal to or otherwise
|
||||
terminate a foreign process):
|
||||
• os.kill, os.killpg (POSIX)
|
||||
• subprocess.run / Popen / call / check_call / check_output
|
||||
• subprocess.getoutput / getstatusoutput
|
||||
• os.system / os.popen
|
||||
• pty.spawn
|
||||
• asyncio.create_subprocess_exec / create_subprocess_shell
|
||||
Subprocess inspection looks at the WHOLE command string (not just
|
||||
tokens[0]), so ``bash -c "systemctl restart hermes-gateway"``,
|
||||
``sudo systemctl ...``, ``env systemctl ...``, ``setsid systemctl ...``
|
||||
are all caught. ``pkill``/``killall``/``taskkill`` invocations
|
||||
targeting hermes/python patterns are also blocked. Bare ``git``
|
||||
commands may not mutate a protected checkout. Git writes against
|
||||
temporary repositories and read-only Git commands remain allowed.
|
||||
"""
|
||||
if request.node.get_closest_marker(_LIVE_SYSTEM_GUARD_BYPASS_MARK):
|
||||
yield
|
||||
return
|
||||
|
||||
import os as _os
|
||||
import shlex as _shlex
|
||||
import subprocess as _subprocess
|
||||
|
||||
test_pid = _os.getpid()
|
||||
lookalike_ok = request.node.get_closest_marker(_GATEWAY_LOOKALIKE_MARK) is not None
|
||||
# Capture the test process's existing children at fixture start —
|
||||
# any *new* children spawned by the test are also allowlisted via
|
||||
# the live psutil walk below. Static set keeps the fast path cheap.
|
||||
try:
|
||||
import psutil as _psutil
|
||||
_initial_children = {
|
||||
c.pid for c in _psutil.Process(test_pid).children(recursive=True)
|
||||
}
|
||||
except Exception:
|
||||
_psutil = None
|
||||
_initial_children = set()
|
||||
|
||||
def _is_own_subtree(pid: int) -> bool:
|
||||
# PID 0 means "our own process group"; -1 means "every process we
|
||||
# can signal". Both are dangerous when paired with SIGTERM/SIGKILL,
|
||||
# but pid 0 is technically scoped to our group so allow it; pid -1
|
||||
# is treated as foreign (refuse).
|
||||
if pid == 0:
|
||||
return True
|
||||
if pid < 0:
|
||||
return False
|
||||
if pid == test_pid or pid in _initial_children:
|
||||
return True
|
||||
if _psutil is None:
|
||||
return False
|
||||
try:
|
||||
walker = _psutil.Process(pid)
|
||||
except Exception:
|
||||
# Stale PID — kill would be a no-op anyway, allow it.
|
||||
return True
|
||||
try:
|
||||
for parent in walker.parents():
|
||||
if parent.pid == test_pid:
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
return False
|
||||
|
||||
real_kill = _os.kill
|
||||
|
||||
def _guarded_kill(pid, sig, *args, **kwargs):
|
||||
# Signal 0 is a pure liveness probe — it cannot terminate anything.
|
||||
# psutil.pid_exists() uses os.kill(pid, 0) on POSIX, and probing a
|
||||
# just-killed grandchild that was reparented to init (zombie with a
|
||||
# foreign parent chain) must not trip the guard. Flaked in CI on
|
||||
# test_entire_tree_is_sigkilled_not_just_parent.
|
||||
if int(sig) == 0:
|
||||
return real_kill(pid, sig, *args, **kwargs)
|
||||
if _is_own_subtree(int(pid)):
|
||||
return real_kill(pid, sig, *args, **kwargs)
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked os.kill("
|
||||
f"{pid}, {sig}) — PID is outside the test process subtree. "
|
||||
"If this fired in CI it means the test reached a real "
|
||||
"kill_gateway_processes / stop_profile_gateway / cmd_update "
|
||||
"code path without mocking find_gateway_pids and os.kill. "
|
||||
"Mock both, or mark the test with "
|
||||
"@pytest.mark.live_system_guard_bypass if real signal "
|
||||
"delivery is genuinely required."
|
||||
)
|
||||
|
||||
monkeypatch.setattr(_os, "kill", _guarded_kill)
|
||||
|
||||
# ``os.killpg`` is the same risk class — sends a signal to every
|
||||
# process in a group. The gateway is a session leader (its own
|
||||
# PGID == its PID), so killpg(gateway_pid, SIGTERM) is a one-shot
|
||||
# kill of the live process. Allow it only when the target PGID is
|
||||
# the test process's own group.
|
||||
if hasattr(_os, "killpg"):
|
||||
real_killpg = _os.killpg
|
||||
own_pgid = _os.getpgrp()
|
||||
|
||||
def _guarded_killpg(pgid, sig, *args, **kwargs):
|
||||
# Signal 0 is a pure liveness probe — never destructive.
|
||||
if int(sig) == 0:
|
||||
return real_killpg(pgid, sig, *args, **kwargs)
|
||||
if int(pgid) == own_pgid or _is_own_subtree(int(pgid)):
|
||||
return real_killpg(pgid, sig, *args, **kwargs)
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"os.killpg({pgid}, {sig}) — PGID is outside the test "
|
||||
"process group. See _live_system_guard for the why."
|
||||
)
|
||||
|
||||
monkeypatch.setattr(_os, "killpg", _guarded_killpg)
|
||||
|
||||
# ── Subprocess command-string inspection (whole-line) ──────────
|
||||
_HERMES_TOKENS = (
|
||||
"hermes-gateway",
|
||||
"hermes.service",
|
||||
"hermes_cli.main gateway",
|
||||
"hermes_cli/main.py gateway",
|
||||
"gateway/run.py",
|
||||
"hermes gateway",
|
||||
)
|
||||
_MUTATING_VERBS = (
|
||||
"restart", "start", "stop", "kill", "reload",
|
||||
"reset-failed", "enable", "disable", "mask", "unmask",
|
||||
"daemon-reload", "try-restart", "reload-or-restart",
|
||||
)
|
||||
_PROCESS_KILLERS = ("pkill", "killall", "taskkill", "skill", "fuser")
|
||||
_CONTAINER_RUNTIMES = ("docker", "podman", "nerdctl")
|
||||
|
||||
def _first_token_basename(cmd_str: str) -> str:
|
||||
try:
|
||||
tokens = _shlex.split(cmd_str)
|
||||
except ValueError:
|
||||
tokens = cmd_str.split()
|
||||
return tokens[0].rsplit("/", 1)[-1].lower() if tokens else ""
|
||||
# Shell/launcher executables whose arguments are themselves commands —
|
||||
# argv[0]-only scanning must not exempt what they wrap.
|
||||
_WRAPPER_COMMANDS = (
|
||||
"sh", "bash", "zsh", "dash", "env", "nohup", "setsid",
|
||||
"timeout", "sudo", "xargs", "nice", "ionice", "stdbuf", "flock",
|
||||
)
|
||||
|
||||
def _cmd_to_string(cmd) -> str:
|
||||
if cmd is None:
|
||||
return ""
|
||||
if isinstance(cmd, (bytes, bytearray)):
|
||||
try:
|
||||
return bytes(cmd).decode(errors="replace")
|
||||
except Exception:
|
||||
return ""
|
||||
if isinstance(cmd, str):
|
||||
return cmd
|
||||
if isinstance(cmd, (list, tuple)):
|
||||
try:
|
||||
return " ".join(str(t) for t in cmd)
|
||||
except Exception:
|
||||
return ""
|
||||
return str(cmd)
|
||||
|
||||
def _matches_hermes_gateway(cmd_str: str) -> bool:
|
||||
low = cmd_str.lower()
|
||||
return any(tok in low for tok in _HERMES_TOKENS)
|
||||
|
||||
def _is_blocked_systemctl(cmd) -> bool:
|
||||
cmd_str = _cmd_to_string(cmd)
|
||||
if "systemctl" not in cmd_str:
|
||||
return False
|
||||
if not _matches_hermes_gateway(cmd_str):
|
||||
return False
|
||||
try:
|
||||
tokens = _shlex.split(cmd_str)
|
||||
except ValueError:
|
||||
tokens = cmd_str.split()
|
||||
return any(verb in tokens for verb in _MUTATING_VERBS)
|
||||
|
||||
def _is_process_killer(cmd) -> bool:
|
||||
cmd_str = _cmd_to_string(cmd)
|
||||
try:
|
||||
tokens = _shlex.split(cmd_str)
|
||||
except ValueError:
|
||||
tokens = cmd_str.split()
|
||||
if not tokens:
|
||||
return False
|
||||
|
||||
# For argv-style calls only argv[0] is the executable; scanning every
|
||||
# argument blocked innocent commands like ``cat /tmp/.../skill``
|
||||
# ("skill" is in _PROCESS_KILLERS). Wrapper executables still get
|
||||
# full-token scanning so ``["bash", "-c", "pkill ..."]`` stays caught.
|
||||
if isinstance(cmd, (list, tuple)):
|
||||
head0 = tokens[0].rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
|
||||
killer_tokens = tokens if head0 in _WRAPPER_COMMANDS else tokens[:1]
|
||||
else:
|
||||
killer_tokens = tokens
|
||||
for tok in killer_tokens:
|
||||
head = tok.rsplit("/", 1)[-1].rsplit("\\", 1)[-1]
|
||||
if head in _PROCESS_KILLERS:
|
||||
low = cmd_str.lower()
|
||||
# pkill -f pattern: catch hermes-themed patterns + a
|
||||
# plain "python" -f which would catch the live gateway
|
||||
# whose cmdline contains "python -m hermes_cli.main".
|
||||
if (
|
||||
"hermes" in low
|
||||
or "gateway" in low
|
||||
or ("python" in low and "-f" in tokens)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
from tests.git_safety import blocked_git_mutation
|
||||
|
||||
def _check_subprocess_cmd(name, cmd, kwargs=None):
|
||||
git_verb = blocked_git_mutation(cmd, kwargs, _LIVE_GUARD_PROTECTED_GIT_ROOTS)
|
||||
if git_verb is not None:
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — `git {git_verb}` would mutate "
|
||||
"the protected checkout. Use a temporary repository or mock "
|
||||
"the update boundary; live_system_guard_bypass is for deliberate live tests."
|
||||
)
|
||||
if _is_blocked_systemctl(cmd):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — would mutate the "
|
||||
"live hermes-gateway systemd unit. Mock "
|
||||
"subprocess.run / _run_systemctl in the test, or "
|
||||
"mark with @pytest.mark.live_system_guard_bypass."
|
||||
)
|
||||
if _is_process_killer(cmd):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — process-killer command "
|
||||
"targeting hermes/python could hit the live gateway. "
|
||||
"Mark with @pytest.mark.live_system_guard_bypass if "
|
||||
"intentional."
|
||||
)
|
||||
# Block any subprocess that would run `hermes update` (or the
|
||||
# equivalent `python -m hermes_cli.main update`). These commands
|
||||
# run `git fetch origin + git pull` against the REAL checkout,
|
||||
# overwriting files like pyproject.toml mid-test-run and corrupting
|
||||
# every subsequent subprocess that reads them. The corruption is
|
||||
# especially insidious because the spawned process uses setsid/
|
||||
# start_new_session=True, making it invisible to pytest's process
|
||||
# tree (PPid=1) and nearly impossible to trace without explicit
|
||||
# inotify/SHA watchdogs. Any test that legitimately needs to exercise
|
||||
# the update-spawn path must mock subprocess.Popen explicitly.
|
||||
cmd_str = _cmd_to_string(cmd)
|
||||
low = cmd_str.lower()
|
||||
if "update" in low and (
|
||||
# hermes update / hermes update --gateway / setsid bash -c ... hermes update
|
||||
("hermes" in low and "update" in low.split())
|
||||
or
|
||||
# python -m hermes_cli.main update --gateway
|
||||
("hermes_cli" in low and "update" in low.split())
|
||||
or
|
||||
# venv/bin/hermes update (absolute path variant used in tests)
|
||||
(".venv/bin/hermes" in low and "update" in low)
|
||||
):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — this command would run "
|
||||
"`hermes update` against the real checkout, fetching "
|
||||
"from origin and overwriting repo files (e.g. "
|
||||
"pyproject.toml) mid-test-run. This corrupts every "
|
||||
"subsequent subprocess in the same runner. "
|
||||
"Mock subprocess.Popen (and subprocess.run if used) "
|
||||
"in the test instead, or mark with "
|
||||
"@pytest.mark.live_system_guard_bypass if genuinely "
|
||||
"needed (e.g. an integration test testing the update "
|
||||
"flow against a dedicated throwaway repo)."
|
||||
)
|
||||
# Block spawning a REAL gateway runtime (``python -m hermes_cli.main
|
||||
# gateway run|start|restart``). ``_spawn_hermes_action`` launches it
|
||||
# with start_new_session=True, so it outlives the pytest worker; the
|
||||
# child inherits the pytest-tmp HERMES_HOME, resolves the DEVELOPER's
|
||||
# ``hermes-gateway`` systemd unit (a tmp home hashes to no profile
|
||||
# suffix), restarts the live gateway, and the survivors squat the
|
||||
# webhook port. 2026-09-03: 39 such orphans lived 6 days after a
|
||||
# sibling refactor moved the spawn seam and left tests patching the
|
||||
# facade. The canonical matcher, never an argv substring.
|
||||
from gateway.status import _gateway_command_subcommand
|
||||
# A gateway launched INSIDE a container (`docker exec … hermes gateway start`) cannot
|
||||
# reach the host's systemd unit or webhook port; tests/docker/ exists to exercise it.
|
||||
in_container = _first_token_basename(cmd_str) in _CONTAINER_RUNTIMES
|
||||
if (
|
||||
not lookalike_ok
|
||||
and not in_container
|
||||
and _gateway_command_subcommand(cmd_str) in ("run", "start", "restart")
|
||||
):
|
||||
raise RuntimeError(
|
||||
f"tests/conftest.py live-system guard: blocked "
|
||||
f"subprocess.{name}({cmd!r}) — this would spawn a REAL "
|
||||
"hermes gateway runtime that outlives the test (it is "
|
||||
"detached), restarts the developer's live gateway, and "
|
||||
"holds the webhook port. Patch the spawn seam where "
|
||||
"production reads it (hermes_cli.web_server_gateway."
|
||||
"_spawn_hermes_action), or mark with "
|
||||
"@pytest.mark.spawns_gateway_lookalike a test that spawns "
|
||||
"and reaps its own stub child."
|
||||
)
|
||||
|
||||
def _wrap_subprocess(name, real):
|
||||
def _guarded(cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd(name, cmd, kwargs)
|
||||
return real(cmd, *args, **kwargs)
|
||||
_guarded.__name__ = f"_guarded_{name}"
|
||||
# Make the wrapper subscriptable like the wrapped callable when
|
||||
# the wrapped object is. ``subprocess.Popen[bytes]`` is used as
|
||||
# a type annotation in third-party packages (mcp, etc.); replacing
|
||||
# ``Popen`` with a plain function breaks ``Popen[bytes]`` at
|
||||
# import time. Defer ``__class_getitem__`` to the original.
|
||||
if hasattr(real, "__class_getitem__"):
|
||||
_guarded.__class_getitem__ = real.__class_getitem__
|
||||
return _guarded
|
||||
|
||||
def _wrap_popen():
|
||||
"""Subclass Popen so isinstance checks AND Popen[bytes] still work."""
|
||||
real = _subprocess.Popen
|
||||
|
||||
class _GuardedPopen(real): # type: ignore[misc, valid-type]
|
||||
def __init__(self, cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd("Popen", cmd, kwargs)
|
||||
super().__init__(cmd, *args, **kwargs)
|
||||
|
||||
_GuardedPopen.__name__ = "Popen"
|
||||
_GuardedPopen.__qualname__ = "Popen"
|
||||
return _GuardedPopen
|
||||
|
||||
real_run = _subprocess.run
|
||||
real_popen = _subprocess.Popen
|
||||
real_call = _subprocess.call
|
||||
real_check_call = _subprocess.check_call
|
||||
real_check_output = _subprocess.check_output
|
||||
real_getoutput = _subprocess.getoutput
|
||||
real_getstatusoutput = _subprocess.getstatusoutput
|
||||
|
||||
monkeypatch.setattr(_subprocess, "run", _wrap_subprocess("run", real_run))
|
||||
monkeypatch.setattr(_subprocess, "Popen", _wrap_popen())
|
||||
monkeypatch.setattr(_subprocess, "call", _wrap_subprocess("call", real_call))
|
||||
monkeypatch.setattr(
|
||||
_subprocess, "check_call", _wrap_subprocess("check_call", real_check_call)
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
_subprocess,
|
||||
"check_output",
|
||||
_wrap_subprocess("check_output", real_check_output),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
_subprocess, "getoutput", _wrap_subprocess("getoutput", real_getoutput)
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
_subprocess,
|
||||
"getstatusoutput",
|
||||
_wrap_subprocess("getstatusoutput", real_getstatusoutput),
|
||||
)
|
||||
|
||||
# os.system / os.popen — same risk class, completely unwrapped before.
|
||||
real_os_system = _os.system
|
||||
real_os_popen = _os.popen
|
||||
|
||||
def _guarded_os_system(command):
|
||||
_check_subprocess_cmd("os.system", command)
|
||||
return real_os_system(command)
|
||||
|
||||
def _guarded_os_popen(cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd("os.popen", cmd, kwargs)
|
||||
return real_os_popen(cmd, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_os, "system", _guarded_os_system)
|
||||
monkeypatch.setattr(_os, "popen", _guarded_os_popen)
|
||||
|
||||
# pty.spawn — POSIX-only.
|
||||
try:
|
||||
import pty as _pty
|
||||
if hasattr(_pty, "spawn"):
|
||||
real_pty_spawn = _pty.spawn
|
||||
|
||||
def _guarded_pty_spawn(argv, *args, **kwargs):
|
||||
_check_subprocess_cmd("pty.spawn", argv, kwargs)
|
||||
return real_pty_spawn(argv, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_pty, "spawn", _guarded_pty_spawn)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# asyncio.create_subprocess_* — bypasses subprocess module entirely.
|
||||
try:
|
||||
import asyncio as _asyncio
|
||||
real_async_exec = _asyncio.create_subprocess_exec
|
||||
real_async_shell = _asyncio.create_subprocess_shell
|
||||
|
||||
async def _guarded_async_exec(program, *args, **kwargs):
|
||||
_check_subprocess_cmd(
|
||||
"asyncio.create_subprocess_exec", [program, *args], kwargs
|
||||
)
|
||||
return await real_async_exec(program, *args, **kwargs)
|
||||
|
||||
async def _guarded_async_shell(cmd, *args, **kwargs):
|
||||
_check_subprocess_cmd("asyncio.create_subprocess_shell", cmd, kwargs)
|
||||
return await real_async_shell(cmd, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_asyncio, "create_subprocess_exec", _guarded_async_exec)
|
||||
monkeypatch.setattr(
|
||||
_asyncio, "create_subprocess_shell", _guarded_async_shell
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
yield
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _audio_playback_guard(request, monkeypatch):
|
||||
"""Stub TTS synthesis + speaker playback for every test.
|
||||
|
||||
@@ -8,7 +8,7 @@ import pytest
|
||||
|
||||
@pytest.mark.platforms("posix")
|
||||
def test_guard_blocks_native_and_shell_git_mutations_without_touching_checkout(tmp_path, monkeypatch):
|
||||
from tests import conftest
|
||||
from tests._fixtures import live_system_guard
|
||||
|
||||
def git(repo, *args):
|
||||
result = subprocess.run(["git", "-C", str(repo), *args], capture_output=True, text=True, check=True)
|
||||
@@ -22,7 +22,7 @@ def test_guard_blocks_native_and_shell_git_mutations_without_touching_checkout(t
|
||||
(repo / "sentinel").write_text("committed", encoding="utf-8")
|
||||
git(repo, "add", "sentinel")
|
||||
git(repo, "-c", "user.name=Test", "-c", "user.email=test@example.invalid", "commit", "-m", "second")
|
||||
monkeypatch.setattr(conftest, "_LIVE_GUARD_PROTECTED_GIT_ROOTS", (protected,))
|
||||
monkeypatch.setattr(live_system_guard, "_LIVE_GUARD_PROTECTED_GIT_ROOTS", (protected,))
|
||||
head = git(protected, "rev-parse", "HEAD")
|
||||
(protected / "sentinel").write_bytes(b"uncommitted user data")
|
||||
for command in (
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
from tests.conftest import _host_matches_platforms, _platform_machine
|
||||
from tests._fixtures.platform_gating import _host_matches_platforms, _platform_machine
|
||||
|
||||
|
||||
@pytest.mark.parametrize("specs,hosts", [
|
||||
|
||||
Reference in New Issue
Block a user