# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
725 lines
28 KiB
Python
725 lines
28 KiB
Python
"""Tests for the shell-hooks subprocess bridge (agent.shell_hooks).
|
|
|
|
These tests focus on the pure translation layer — JSON serialisation,
|
|
JSON parsing, matcher behaviour, block-schema correctness, and the
|
|
subprocess runner's graceful error handling. Consent prompts are
|
|
covered in ``test_shell_hooks_consent.py``.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from agent import shell_hooks
|
|
|
|
|
|
# ── helpers ───────────────────────────────────────────────────────────────
|
|
|
|
|
|
def _write_script(tmp_path: Path, name: str, body: str) -> Path:
|
|
path = tmp_path / name
|
|
path.write_text(body)
|
|
path.chmod(0o755)
|
|
return path
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _reset_registration_state():
|
|
shell_hooks.reset_for_tests()
|
|
yield
|
|
shell_hooks.reset_for_tests()
|
|
|
|
|
|
# ── _parse_response ───────────────────────────────────────────────────────
|
|
|
|
|
|
class TestParseResponse:
|
|
def test_block_claude_code_style(self):
|
|
r = shell_hooks._parse_response(
|
|
"pre_tool_call",
|
|
'{"decision": "block", "reason": "nope"}',
|
|
)
|
|
assert r == {"action": "block", "message": "nope"}
|
|
|
|
@pytest.mark.parametrize("stdout, expected", [
|
|
('{"action": "approve", "message": " needs a human ", "rule_key": " terminal:rm "}',
|
|
{"action": "approve", "message": "needs a human", "rule_key": "terminal:rm"}),
|
|
('{"action": "approve", "message": "", "rule_key": 7}', {"action": "approve"}),
|
|
# Claude-Code's ``decision: approve`` means auto-ALLOW, not "ask a human": never mapped.
|
|
('{"decision": "approve", "reason": "ok"}', None),
|
|
('{"action": "approve", "decision": "block", "reason": "no"}', {"action": "block", "message": "no"}),
|
|
])
|
|
def test_approve_is_parsed_like_the_plugin_directive(self, stdout, expected):
|
|
"""The documented ``approve`` action used to parse to None, so the tool ran with no
|
|
approval prompt (#92553). It now yields the same shape Python plugins return."""
|
|
assert shell_hooks._parse_response("pre_tool_call", stdout) == expected
|
|
|
|
def test_empty_stdout_returns_none(self):
|
|
assert shell_hooks._parse_response("pre_tool_call", "") is None
|
|
assert shell_hooks._parse_response("pre_tool_call", " ") is None
|
|
|
|
|
|
# ── _serialize_payload ────────────────────────────────────────────────────
|
|
|
|
|
|
class TestSerializePayload:
|
|
|
|
def test_args_not_dict_becomes_null(self):
|
|
raw = shell_hooks._serialize_payload(
|
|
"pre_tool_call", {"args": ["not", "a", "dict"]},
|
|
)
|
|
payload = json.loads(raw)
|
|
assert payload["tool_input"] is None
|
|
|
|
|
|
# ── Matcher behaviour ─────────────────────────────────────────────────────
|
|
|
|
|
|
class TestMatcher:
|
|
|
|
|
|
def test_alternation_matcher(self):
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command="echo", matcher="terminal|file",
|
|
)
|
|
assert spec.matches_tool("terminal")
|
|
assert spec.matches_tool("file")
|
|
assert not spec.matches_tool("web")
|
|
|
|
|
|
def test_matcher_leading_whitespace_stripped(self):
|
|
"""YAML quirks can introduce leading/trailing whitespace — must
|
|
not silently break the matcher."""
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command="echo", matcher=" terminal ",
|
|
)
|
|
assert spec.matcher == "terminal"
|
|
assert spec.matches_tool("terminal")
|
|
|
|
|
|
def test_whitespace_only_matcher_becomes_none(self):
|
|
"""A matcher that's pure whitespace is treated as 'no matcher'."""
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command="echo", matcher=" ",
|
|
)
|
|
assert spec.matcher is None
|
|
assert spec.matches_tool("anything")
|
|
|
|
|
|
# ── End-to-end subprocess behaviour ───────────────────────────────────────
|
|
|
|
|
|
@pytest.mark.platforms("linux")
|
|
class TestCallbackSubprocess:
|
|
|
|
|
|
def test_block_aggregation_through_plugin_manager(self, tmp_path, monkeypatch):
|
|
"""Registering via register_from_config makes
|
|
get_pre_tool_call_block_message surface the block — the real
|
|
end-to-end control flow used by run_agent._invoke_tool."""
|
|
from hermes_cli import plugins
|
|
|
|
script = _write_script(
|
|
tmp_path, "block.sh",
|
|
"#!/usr/bin/env bash\n"
|
|
'printf \'{"decision": "block", "reason": "blocked-by-shell"}\\n\'\n',
|
|
)
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
|
|
monkeypatch.setenv("HERMES_ACCEPT_HOOKS", "1")
|
|
|
|
# Fresh manager
|
|
plugins._plugin_manager = plugins.PluginManager()
|
|
|
|
cfg = {
|
|
"hooks": {
|
|
"pre_tool_call": [
|
|
{"matcher": "terminal", "command": str(script)},
|
|
],
|
|
},
|
|
}
|
|
registered = shell_hooks.register_from_config(cfg, accept_hooks=True)
|
|
assert len(registered) == 1
|
|
|
|
msg = plugins.get_pre_tool_call_block_message(
|
|
tool_name="terminal",
|
|
args={"command": "rm"},
|
|
)
|
|
assert msg == "blocked-by-shell"
|
|
|
|
def test_approve_reaches_the_human_gate_through_plugin_manager(self, tmp_path, monkeypatch):
|
|
"""End to end: a shell hook's approve directive escalates to request_tool_approval with its
|
|
message and rule_key, and the gate's denial blocks the tool (#92553)."""
|
|
from hermes_cli import plugins
|
|
|
|
script = _write_script(
|
|
tmp_path, "approve.sh",
|
|
"#!/usr/bin/env bash\n"
|
|
'printf \'{"action": "approve", "message": "risky", "rule_key": "terminal:rm"}\\n\'\n',
|
|
)
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
|
|
monkeypatch.setenv("HERMES_ACCEPT_HOOKS", "1")
|
|
plugins._plugin_manager = plugins.PluginManager()
|
|
cfg = {"hooks": {"pre_tool_call": [{"matcher": "terminal", "command": str(script)}]}}
|
|
assert len(shell_hooks.register_from_config(cfg, accept_hooks=True)) == 1
|
|
|
|
seen = []
|
|
|
|
def _gate(tool_name, reason, **kwargs):
|
|
seen.append((tool_name, reason, kwargs.get("rule_key")))
|
|
return {"approved": False, "message": "denied by human"}
|
|
|
|
monkeypatch.setattr("tools.approval.request_tool_approval", _gate)
|
|
assert plugins.resolve_pre_tool_block("terminal", {"command": "rm"}) == "denied by human"
|
|
assert seen == [("terminal", "risky", "terminal:rm")]
|
|
|
|
def test_matcher_regex_filters_callback(self, tmp_path, monkeypatch):
|
|
"""A matcher set to 'terminal' must not fire for 'web_search'."""
|
|
calls = tmp_path / "calls.log"
|
|
script = _write_script(
|
|
tmp_path, "log.sh",
|
|
f"#!/usr/bin/env bash\n"
|
|
f"echo \"$(cat -)\" >> {calls}\n"
|
|
f"printf '{{}}\\n'\n",
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call",
|
|
command=str(script),
|
|
matcher="terminal",
|
|
)
|
|
cb = shell_hooks._make_callback(spec)
|
|
cb(tool_name="terminal", args={"command": "ls"})
|
|
cb(tool_name="web_search", args={"q": "x"})
|
|
cb(tool_name="file_read", args={"path": "x"})
|
|
assert calls.exists()
|
|
# Only the terminal call wrote to the log
|
|
assert calls.read_text().count("pre_tool_call") == 1
|
|
|
|
def test_payload_schema_delivered(self, tmp_path):
|
|
capture = tmp_path / "payload.json"
|
|
script = _write_script(
|
|
tmp_path, "capture.sh",
|
|
f"#!/usr/bin/env bash\ncat - > {capture}\nprintf '{{}}\\n'\n",
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command=str(script),
|
|
)
|
|
cb = shell_hooks._make_callback(spec)
|
|
cb(
|
|
tool_name="terminal",
|
|
args={"command": "echo hi"},
|
|
session_id="sess-77",
|
|
task_id="task-77",
|
|
)
|
|
payload = json.loads(capture.read_text())
|
|
assert payload["hook_event_name"] == "pre_tool_call"
|
|
assert payload["tool_name"] == "terminal"
|
|
assert payload["tool_input"] == {"command": "echo hi"}
|
|
assert payload["session_id"] == "sess-77"
|
|
assert "cwd" in payload
|
|
assert payload["extra"]["task_id"] == "task-77"
|
|
|
|
|
|
def test_modify_canonical_parsing(self, tmp_path):
|
|
"""Shell hook returning canonical modify is parsed correctly."""
|
|
script = _write_script(
|
|
tmp_path, "mod_canon.sh",
|
|
"#!/usr/bin/env bash\n"
|
|
'printf \'{"action": "modify", "args": {"path": "/safe"}}\\n\'',
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command=str(script),
|
|
)
|
|
cb = shell_hooks._make_callback(spec)
|
|
result = cb(tool_name="write_file", args={"path": "/unsafe"})
|
|
assert result == {"action": "modify", "args": {"path": "/safe"}}
|
|
|
|
def test_modify_claude_code_parsing(self, tmp_path):
|
|
"""Shell hook returning Claude-Code modify is normalised."""
|
|
script = _write_script(
|
|
tmp_path, "mod_cc.sh",
|
|
"#!/usr/bin/env bash\n"
|
|
'printf \'{"decision": "modify", "tool_input": {"content": "safe"}}\\n\'',
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command=str(script),
|
|
)
|
|
cb = shell_hooks._make_callback(spec)
|
|
result = cb(tool_name="write_file", args={"content": "danger"})
|
|
assert result == {"action": "modify", "args": {"content": "safe"}}
|
|
|
|
|
|
# ── config parsing ────────────────────────────────────────────────────────
|
|
|
|
|
|
class TestParseHooksBlock:
|
|
def test_valid_entry(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"pre_tool_call": [
|
|
{"matcher": "terminal", "command": "/tmp/hook.sh", "timeout": 30},
|
|
],
|
|
})
|
|
assert len(specs) == 1
|
|
assert specs[0].event == "pre_tool_call"
|
|
assert specs[0].matcher == "terminal"
|
|
assert specs[0].command == "/tmp/hook.sh"
|
|
assert specs[0].timeout == 30
|
|
|
|
|
|
def test_python_only_event_refused(self):
|
|
# transform_api_error_classification returns a classification directive that
|
|
# _parse_response has no channel for — a shell registration would
|
|
# be silently ignored, so it must be refused with a warning.
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"transform_api_error_classification": [
|
|
{"command": "/tmp/hook.sh"},
|
|
],
|
|
})
|
|
assert specs == []
|
|
|
|
def test_timeout_clamped_to_max(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"post_tool_call": [
|
|
{"command": "/tmp/slow.sh", "timeout": 9999},
|
|
],
|
|
})
|
|
assert specs[0].timeout == shell_hooks.MAX_TIMEOUT_SECONDS
|
|
|
|
|
|
def test_none_hooks_block(self):
|
|
assert shell_hooks._parse_hooks_block(None) == []
|
|
assert shell_hooks._parse_hooks_block("string") == []
|
|
assert shell_hooks._parse_hooks_block([]) == []
|
|
|
|
def test_non_tool_event_matcher_warns_and_drops(self):
|
|
"""matcher: is only honored for pre/post_tool_call; must drop it
|
|
on other events so the spec reflects runtime."""
|
|
cfg = {"pre_llm_call": [{"matcher": "terminal", "command": "/bin/echo"}]}
|
|
specs = shell_hooks._parse_hooks_block(cfg)
|
|
assert len(specs) == 1 and specs[0].matcher is None
|
|
|
|
|
|
# ── Idempotent registration ───────────────────────────────────────────────
|
|
|
|
|
|
class TestIdempotentRegistration:
|
|
def test_double_call_registers_once(self, tmp_path, monkeypatch):
|
|
from hermes_cli import plugins
|
|
|
|
script = _write_script(tmp_path, "h.sh",
|
|
"#!/usr/bin/env bash\nprintf '{}\\n'\n")
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
|
|
monkeypatch.setenv("HERMES_ACCEPT_HOOKS", "1")
|
|
|
|
plugins._plugin_manager = plugins.PluginManager()
|
|
|
|
cfg = {"hooks": {"on_session_start": [{"command": str(script)}]}}
|
|
|
|
first = shell_hooks.register_from_config(cfg, accept_hooks=True)
|
|
second = shell_hooks.register_from_config(cfg, accept_hooks=True)
|
|
assert len(first) == 1
|
|
assert second == []
|
|
# Only one callback on the manager
|
|
mgr = plugins.get_plugin_manager()
|
|
assert len(mgr._hooks.get("on_session_start", [])) == 1
|
|
|
|
def test_same_command_different_matcher_registers_both(
|
|
self, tmp_path, monkeypatch,
|
|
):
|
|
"""Same script used for different matchers under one event must
|
|
register both callbacks — dedupe keys on (event, matcher, command)."""
|
|
from hermes_cli import plugins
|
|
|
|
script = _write_script(tmp_path, "h.sh",
|
|
"#!/usr/bin/env bash\nprintf '{}\\n'\n")
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
|
|
monkeypatch.setenv("HERMES_ACCEPT_HOOKS", "1")
|
|
|
|
plugins._plugin_manager = plugins.PluginManager()
|
|
|
|
cfg = {
|
|
"hooks": {
|
|
"pre_tool_call": [
|
|
{"matcher": "terminal", "command": str(script)},
|
|
{"matcher": "web_search", "command": str(script)},
|
|
],
|
|
},
|
|
}
|
|
|
|
registered = shell_hooks.register_from_config(cfg, accept_hooks=True)
|
|
assert len(registered) == 2
|
|
mgr = plugins.get_plugin_manager()
|
|
assert len(mgr._hooks.get("pre_tool_call", [])) == 2
|
|
|
|
|
|
# ── Allowlist concurrency ─────────────────────────────────────────────────
|
|
|
|
|
|
class TestAllowlistConcurrency:
|
|
"""Regression tests for the Codex#1 finding: simultaneous
|
|
_record_approval() calls used to collide on a fixed tmp path and
|
|
silently lose entries under read-modify-write races."""
|
|
|
|
|
|
def test_save_allowlist_uses_unique_tmp_paths(self, tmp_path, monkeypatch):
|
|
"""Two save_allowlist calls in flight must use distinct tmp files
|
|
so the loser's os.replace does not ENOENT on the winner's sweep."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
|
|
p = shell_hooks.allowlist_path()
|
|
p.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
tmp_paths_seen: list = []
|
|
import utils
|
|
real_mkstemp = utils.tempfile.mkstemp
|
|
|
|
def spying_mkstemp(*args, **kwargs):
|
|
fd, path = real_mkstemp(*args, **kwargs)
|
|
tmp_paths_seen.append(path)
|
|
return fd, path
|
|
|
|
monkeypatch.setattr(utils.tempfile, "mkstemp", spying_mkstemp)
|
|
|
|
shell_hooks.save_allowlist({"approvals": [{"event": "a", "command": "x"}]})
|
|
shell_hooks.save_allowlist({"approvals": [{"event": "b", "command": "y"}]})
|
|
|
|
assert len(tmp_paths_seen) == 2
|
|
assert tmp_paths_seen[0] != tmp_paths_seen[1]
|
|
|
|
|
|
# ── fail_closed parsing ───────────────────────────────────────────────────
|
|
|
|
|
|
class TestFailClosedParsing:
|
|
def test_fail_closed_parsed(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"pre_tool_call": [
|
|
{"command": "/tmp/h.sh", "fail_closed": True},
|
|
],
|
|
})
|
|
assert len(specs) == 1
|
|
assert specs[0].fail_closed is True
|
|
|
|
def test_fail_closed_defaults_false(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"pre_tool_call": [{"command": "/tmp/h.sh"}],
|
|
})
|
|
assert specs[0].fail_closed is False
|
|
|
|
def test_failclosed_camel_alias(self):
|
|
"""Cursor/Claude Code configs spell it failClosed."""
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"pre_tool_call": [
|
|
{"command": "/tmp/h.sh", "failClosed": True},
|
|
],
|
|
})
|
|
assert specs[0].fail_closed is True
|
|
|
|
def test_canonical_wins_over_alias(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"pre_tool_call": [
|
|
{"command": "/tmp/h.sh", "fail_closed": False,
|
|
"failClosed": True},
|
|
],
|
|
})
|
|
assert specs[0].fail_closed is False
|
|
|
|
def test_non_bool_warns_and_defaults_false(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"pre_tool_call": [
|
|
{"command": "/tmp/h.sh", "fail_closed": "yes"},
|
|
],
|
|
})
|
|
assert specs[0].fail_closed is False
|
|
|
|
def test_fail_closed_on_non_blocking_event_warns_and_ignores(self):
|
|
specs = shell_hooks._parse_hooks_block({
|
|
"on_session_start": [
|
|
{"command": "/tmp/h.sh", "fail_closed": True},
|
|
],
|
|
})
|
|
assert specs[0].fail_closed is False
|
|
|
|
|
|
# ── _evaluate_result semantics ────────────────────────────────────────────
|
|
|
|
|
|
def _spawn_result(**overrides):
|
|
base = {
|
|
"returncode": 0,
|
|
"stdout": "",
|
|
"stderr": "",
|
|
"timed_out": False,
|
|
"elapsed_seconds": 0.1,
|
|
"error": None,
|
|
}
|
|
base.update(overrides)
|
|
return base
|
|
|
|
|
|
class TestEvaluateResult:
|
|
def _spec(self, event="pre_tool_call", fail_closed=False):
|
|
return shell_hooks.ShellHookSpec(
|
|
event=event, command="/tmp/h.sh", fail_closed=fail_closed,
|
|
)
|
|
|
|
# -- exit code 2 = block --------------------------------------------
|
|
|
|
def test_exit_2_blocks_with_stderr_message(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(),
|
|
_spawn_result(returncode=2, stderr="policy violation\n"),
|
|
)
|
|
assert r == {"action": "block", "message": "policy violation"}
|
|
|
|
def test_exit_2_blocks_with_default_message(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(), _spawn_result(returncode=2),
|
|
)
|
|
assert r == {
|
|
"action": "block",
|
|
"message": shell_hooks._DEFAULT_BLOCK_MESSAGE,
|
|
}
|
|
|
|
def test_exit_2_stdout_block_json_wins(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(),
|
|
_spawn_result(
|
|
returncode=2,
|
|
stdout='{"decision": "block", "reason": "from stdout"}',
|
|
stderr="from stderr",
|
|
),
|
|
)
|
|
assert r == {"action": "block", "message": "from stdout"}
|
|
|
|
def test_exit_2_on_non_blocking_event_does_not_block(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(event="on_session_start"),
|
|
_spawn_result(returncode=2, stderr="boom"),
|
|
)
|
|
assert r is None
|
|
|
|
def test_other_nonzero_exit_still_parses_stdout(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(),
|
|
_spawn_result(
|
|
returncode=1,
|
|
stdout='{"decision": "block", "reason": "nope"}',
|
|
),
|
|
)
|
|
assert r == {"action": "block", "message": "nope"}
|
|
|
|
def test_nonzero_exit_without_directive_is_none(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(), _spawn_result(returncode=1, stderr="oops"),
|
|
)
|
|
assert r is None
|
|
|
|
# -- fail_closed ------------------------------------------------------
|
|
|
|
def test_spawn_error_fails_open_by_default(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(), _spawn_result(error="No such file"),
|
|
)
|
|
assert r is None
|
|
|
|
def test_spawn_error_fail_closed_blocks(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(fail_closed=True), _spawn_result(error="No such file"),
|
|
)
|
|
assert r["action"] == "block"
|
|
assert "failed closed" in r["message"]
|
|
assert "No such file" in r["message"]
|
|
|
|
def test_timeout_fails_open_by_default(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(), _spawn_result(timed_out=True),
|
|
)
|
|
assert r is None
|
|
|
|
def test_timeout_fail_closed_blocks(self):
|
|
spec = self._spec(fail_closed=True)
|
|
r = shell_hooks._evaluate_result(spec, _spawn_result(timed_out=True))
|
|
assert r["action"] == "block"
|
|
assert f"timed out after {spec.timeout}s" in r["message"]
|
|
|
|
def test_unparseable_stdout_fail_closed_blocks(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(fail_closed=True),
|
|
_spawn_result(stdout="Traceback (most recent call last): ..."),
|
|
)
|
|
assert r["action"] == "block"
|
|
assert "unparseable stdout" in r["message"]
|
|
|
|
def test_unparseable_stdout_fails_open_by_default(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(),
|
|
_spawn_result(stdout="Traceback (most recent call last): ..."),
|
|
)
|
|
assert r is None
|
|
|
|
def test_valid_noop_json_passes_fail_closed(self):
|
|
"""A clean {} no-op must NOT be blocked by fail_closed."""
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(fail_closed=True), _spawn_result(stdout="{}"),
|
|
)
|
|
assert r is None
|
|
|
|
def test_empty_stdout_passes_fail_closed(self):
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(fail_closed=True), _spawn_result(stdout=""),
|
|
)
|
|
assert r is None
|
|
|
|
def test_fail_closed_on_non_blocking_event_still_fails_open(self):
|
|
"""Defense in depth: even if a spec sneaks past parsing with
|
|
fail_closed on a non-blocking event, runtime fails open."""
|
|
r = shell_hooks._evaluate_result(
|
|
self._spec(event="on_session_start", fail_closed=True),
|
|
_spawn_result(error="boom"),
|
|
)
|
|
assert r is None
|
|
|
|
|
|
# ── exit-2 / fail_closed end-to-end ──────────────────────────────────────
|
|
|
|
|
|
class TestFailSemanticsEndToEnd:
|
|
@pytest.mark.platforms("linux")
|
|
def test_exit_2_script_blocks(self, tmp_path):
|
|
script = _write_script(
|
|
tmp_path, "exit2.sh",
|
|
"#!/usr/bin/env bash\n"
|
|
'echo "rm -rf is not permitted" >&2\n'
|
|
"exit 2\n",
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command=str(script),
|
|
)
|
|
cb = shell_hooks._make_callback(spec)
|
|
result = cb(tool_name="terminal", args={"command": "rm -rf /"})
|
|
assert result == {
|
|
"action": "block", "message": "rm -rf is not permitted",
|
|
}
|
|
|
|
def test_fail_closed_missing_command_blocks(self, tmp_path):
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call",
|
|
command=str(tmp_path / "does-not-exist.sh"),
|
|
fail_closed=True,
|
|
)
|
|
cb = shell_hooks._make_callback(spec)
|
|
result = cb(tool_name="terminal", args={"command": "ls"})
|
|
assert result is not None and result["action"] == "block"
|
|
assert "failed closed" in result["message"]
|
|
|
|
@pytest.mark.platforms("linux")
|
|
def test_run_once_reflects_exit_2_block(self, tmp_path):
|
|
"""hermes hooks test must mirror production semantics."""
|
|
script = _write_script(
|
|
tmp_path, "exit2.sh",
|
|
"#!/usr/bin/env bash\n"
|
|
'echo "denied" >&2\n'
|
|
"exit 2\n",
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command=str(script),
|
|
)
|
|
result = shell_hooks.run_once(
|
|
spec, {"tool_name": "terminal", "args": {"command": "ls"}},
|
|
)
|
|
assert result["returncode"] == 2
|
|
assert result["parsed"] == {"action": "block", "message": "denied"}
|
|
|
|
@pytest.mark.platforms("linux")
|
|
def test_run_once_reflects_fail_closed_timeout(self, tmp_path):
|
|
script = _write_script(
|
|
tmp_path, "sleepy.sh",
|
|
"#!/usr/bin/env bash\nsleep 5\n",
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(
|
|
event="pre_tool_call", command=str(script),
|
|
timeout=1, fail_closed=True,
|
|
)
|
|
result = shell_hooks.run_once(
|
|
spec, {"tool_name": "terminal", "args": {"command": "ls"}},
|
|
)
|
|
assert result["timed_out"] is True
|
|
assert result["parsed"]["action"] == "block"
|
|
assert "failed closed" in result["parsed"]["message"]
|
|
|
|
|
|
# ── multiplexed profiles ──────────────────────────────────────────────────
|
|
|
|
|
|
class TestRoutedProfileEnv:
|
|
@pytest.mark.platforms("linux")
|
|
def test_hook_child_sees_routed_profile_home_and_no_default_secrets(self, tmp_path, monkeypatch):
|
|
"""Under multiplexing the child gets the ROUTED HERMES_HOME, the default profile's secrets
|
|
stay out of its env, and the payload names the firing profile."""
|
|
from hermes_constants import reset_hermes_home_override, set_hermes_home_override
|
|
|
|
launch, routed = tmp_path / "launch", tmp_path / "routed"
|
|
launch.mkdir(); routed.mkdir()
|
|
monkeypatch.setenv("HERMES_HOME", str(launch))
|
|
monkeypatch.setenv("OPENAI_API_KEY", "sk-default-profile")
|
|
monkeypatch.setattr("agent.secret_scope.is_multiplex_active", lambda: True)
|
|
script = _write_script(
|
|
tmp_path, "env_dump.sh",
|
|
"#!/usr/bin/env bash\ncat > /dev/null\n"
|
|
'printf \'{"home": "%s", "key": "%s"}\\n\' "$HERMES_HOME" "${OPENAI_API_KEY:-}"\n',
|
|
)
|
|
spec = shell_hooks.ShellHookSpec(event="pre_tool_call", command=str(script))
|
|
token = set_hermes_home_override(str(routed))
|
|
try:
|
|
result = shell_hooks._spawn(spec, shell_hooks._serialize_payload("pre_tool_call", {"tool_name": "terminal"}))
|
|
payload = json.loads(shell_hooks._serialize_payload("pre_tool_call", {"tool_name": "terminal"}))
|
|
finally:
|
|
reset_hermes_home_override(token)
|
|
seen = json.loads(result["stdout"])
|
|
assert seen["home"] == str(routed)
|
|
assert seen["key"] == ""
|
|
assert "profile" in payload
|
|
|
|
|
|
# ── bare script paths on native Windows ─────────────────────────────────
|
|
# Real subprocesses, no mocked spawn: the failure being guarded is CreateProcess rejecting a text
|
|
# file, which only exists on the host it happens on. Marked per the root AGENTS.md rule against
|
|
# faking ``sys.platform``.
|
|
|
|
|
|
@pytest.mark.platforms("windows")
|
|
def test_bare_script_hook_path_executes_on_windows(tmp_path):
|
|
"""A hook whose command is a bare script path — the shape every example in
|
|
``website/docs/user-guide/features/hooks.md`` uses — must run. POSIX gets there through the
|
|
kernel's shebang handling; CreateProcess has no equivalent, so the same config failed on
|
|
Windows while working everywhere else. A path that is not a file must still be reported as
|
|
missing rather than laundered through an interpreter."""
|
|
script = _write_script(tmp_path, "hook.sh", '#!/usr/bin/env bash\necho "ran" >&2\nexit 7\n')
|
|
|
|
def spec(command):
|
|
return shell_hooks.ShellHookSpec(event="pre_tool_call", command=command)
|
|
|
|
result = shell_hooks._spawn(spec(str(script)), "{}")
|
|
assert result["error"] is None, result["error"]
|
|
assert result["returncode"] == 7, "the script's own exit code must reach a fail_closed gate"
|
|
assert "ran" in result["stderr"]
|
|
|
|
missing = shell_hooks._spawn(spec(str(tmp_path / "gone.sh")), "{}")
|
|
assert missing["error"] == "command not found"
|
|
|
|
|
|
@pytest.mark.platforms("windows")
|
|
def test_unroutable_script_hook_names_the_remediation(tmp_path):
|
|
"""A suffix we deliberately do not route still fails, but the diagnostic has to say what to do:
|
|
the raw WinError text is localized, so a non-English Windows install could not act on it."""
|
|
script = _write_script(tmp_path, "hook.zsh", "#!/bin/zsh\necho hi\n")
|
|
spec = shell_hooks.ShellHookSpec(event="pre_tool_call", command=str(script))
|
|
|
|
result = shell_hooks._spawn(spec, "{}")
|
|
|
|
assert result["returncode"] is None
|
|
assert "interpreter" in result["error"] and "bash" in result["error"]
|