Files
hermes-agent/tools/threat_patterns.py
liuhao1024 1b50e99a47 fix(skills): an env-var NAME constant is not an embedded credential
The generic hardcoded_secret pattern fired on constants whose value is the
NAME of the credential environment variable (an ENV_PASSWORD-style constant
holding the string "MYPLUGIN_" + "APP_PASSWORD"): a line-oriented regex
cannot tell a reference to a secret from the secret itself, so one critical
finding made such plugins uninstallable with no --force override (#116221).

Both copies of the generic pattern (the shared threat-pattern library and
the skill/plugin guard table the installer actually walks) now skip a value
that is itself a SHOUTY_SNAKE environment-variable name (at least two
underscore-separated segments). The carve-out is scoped case-sensitive
because both tables compile with IGNORECASE: a lowercase snake value is the
passphrase shape, and requiring an underscore segment keeps
underscore-free all-caps credentials (AWS access key IDs, base32 secrets)
matched. Prefixed provider tokens (sk-, ghp_, ...) and the dedicated
provider-signature patterns are unaffected.

Regression tests pin both directions: the env-var-name line no longer
flags, and the passphrase / AKIA / base32 / prefixed-token shapes still do.
2026-09-20 11:30:03 -07:00

163 lines
10 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Shared threat-pattern library (prompt injection / promptware / exfiltration) for
``agent/prompt_builder.py``, ``tools/memory_tool.py`` and ``agent/tool_dispatch_helpers.py``.
Each pattern is ``(regex, pattern_id, scope)``; scope is cumulative: ``"all"`` everywhere,
``"context"`` adds promptware / C2 / role hijack for context files, memory and tool results
(warn-level: that content is not user-authored), ``"strict"`` adds aggressive checks only for
user-mediated writes (memory, skill installs) where a block is resolvable. New patterns must
anchor on C2 vocabulary or unambiguous attack behavior, NOT bossy English ("you must" is common
in legitimate AGENTS.md); filler between tokens is the bounded ``_FILLER``."""
from __future__ import annotations
import re
import unicodedata
from typing import List, Optional, Tuple
# Hard cap on scanned text: scanners are advisory, so bound worst-case runtime.
MAX_SCAN_CHARS = 65_536
# Bounded filler between key attack words (unbounded ``(?:\w+\s+)*`` backtracks badly).
_FILLER = r"(?:\w+\s+){0,8}"
# Env var reference ending in a secret-ish suffix (see exfil comment below).
_SECRET_VAR = r"\$\{?\w*(?:KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL)S?\b"
# Verb prefix for "modify agent config" patterns.
_MODIFY = r"(update|modify|edit|write|change|append|add\s+to)\s+[^\n]{0,2048}"
# (regex, pattern_id, scope); scope ∈ {"all", "context", "strict"}
_PATTERNS: List[Tuple[str, str, str]] = [
# ── Classic prompt injection (applies everywhere) ────────────────
(rf'ignore\s+{_FILLER}(previous|all|above|prior)\s+{_FILLER}instructions', "prompt_injection", "all"),
(r'system\s+prompt\s+override', "sys_prompt_override", "all"),
(rf'disregard\s+{_FILLER}(your|all|any)\s+{_FILLER}(instructions|rules|guidelines)', "disregard_rules", "all"),
(rf'act\s+as\s+(if|though)\s+{_FILLER}you\s+{_FILLER}(have\s+no|don\'t\s+have)\s+{_FILLER}(restrictions|limits|rules)', "bypass_restrictions", "all"),
(r'<!--[^>]{0,512}(?:ignore|override|system|secret|hidden)[^>]{0,512}-->', "html_comment_injection", "all"),
(r'<\s*div\s+style\s*=\s*["\'][^>]{0,2048}display\s*:\s*none', "hidden_div", "all"),
(
r"translate\s+[^\n]{0,512}\s+into\s+\w+(?:[\s-]+\w+){0,2}\s+and\s+(execute|run|eval)\b",
"translate_execute",
"all",
),
(rf'do\s+not\s+{_FILLER}tell\s+{_FILLER}the\s+user', "deception_hide", "all"),
# ── Role-play / identity hijack (scraped web content, poisoned context files) ──
(rf'you\s+are\s+{_FILLER}now\s+(?:a|an|the)\s+', "role_hijack", "context"),
(rf'pretend\s+{_FILLER}(you\s+are|to\s+be)\s+', "role_pretend", "context"),
(rf'output\s+{_FILLER}(system|initial)\s+prompt', "leak_system_prompt", "context"),
(rf'(respond|answer|reply)\s+without\s+{_FILLER}(restrictions|limitations|filters|safety)', "remove_filters", "context"),
(rf'you\s+have\s+been\s+{_FILLER}(updated|upgraded|patched)\s+to', "fake_update", "context"),
# Brainworm tell: identity override via spec. Verb pair anchored so "name your variables" is safe.
(r'\bname\s+yourself\s+\w+', "identity_override", "context"),
# ── C2 / Brainworm-style promptware (context scope) ──────────────
# Anchored on C2 vocabulary. "register as a node" appears in legitimate distributed-systems
# docs, so this is WARN not block: a researcher reading the Brainworm post keeps their session.
(r'register\s+(as\s+)?a?\s*node', "c2_node_registration", "context"),
(r'(heartbeat|beacon|check[\s\-]?in)\s+(to|with)\s+', "c2_heartbeat", "context"),
(r'pull\s+(down\s+)?(?:new\s+)?task(?:ing|s)?\b', "c2_task_pull", "context"),
(r'connect\s+to\s+the\s+network\b', "c2_network_connect", "context"),
# C2-specific verbs avoid the broader "you must X" false positive.
(r'you\s+must\s+(?:\w+\s+){0,3}(register|connect|report|beacon)\b', "forced_action", "context"),
# Anti-forensic instructions: near-zero false positive in legitimate content.
(r'only\s+use\s+one[\s\-]?liners?\b', "anti_forensic_oneliner", "context"),
(rf'never\s+{_FILLER}(?:create|write)\s+{_FILLER}(?:script|file)\s+{_FILLER}disk', "anti_forensic_disk", "context"),
# Unsetting agent-runtime env vars is pure attack behavior (Brainworm sub-session bypass).
(r'unset\s+\w*(?:CLAUDE|CODEX|HERMES|AGENT|OPENAI|ANTHROPIC)\w*', "env_var_unset_agent", "context"),
# ── Known C2 / red-team framework names (warn-only) ─────────────
# Every token must be a distinctive offensive-security brand: a common English word here
# (e.g. "praxis", also a legitimate agent name) false-positives whole AGENTS.md / SOUL.md files.
(r'\b(?:cobalt\s*strike|sliver|havoc|mythic|metasploit|brainworm)\b', "known_c2_framework", "context"),
(r'\bc2\s+(?:server|channel|infrastructure|beacon)\b', "c2_explicit", "context"),
(r'\bcommand\s+and\s+control\b', "c2_explicit_long", "context"),
# ── Exfiltration via curl/wget/cat with secrets (applies everywhere) ──
# The var name ends with \b so benign names containing KEY/TOKEN as substrings
# ($TRILLIUM_ETAPI_URL) pass. API is deliberately absent: mid-name API is ubiquitous in
# benign vars, and every real secret it caught ($OPENAI_API_KEY) already ends in KEY/TOKEN.
(rf'curl\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_curl", "all"),
(rf'wget\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_wget", "all"),
(r'cat\s+[^\n]{0,2048}(\.env|credentials|\.netrc|\.pgpass|\.npmrc|\.pypirc)', "read_secrets", "all"),
(r'(send|post|upload|transmit)\s+[^\n]{0,2048}\s+(to|at)\s+https?://', "send_to_url", "strict"),
(rf'(include|output|print|share)\s+{_FILLER}(conversation|chat\s+history|previous\s+messages|full\s+context|entire\s+context)', "context_exfil", "strict"),
# ── Persistence / SSH backdoor (strict scope — memory + skills) ──
(r'authorized_keys', "ssh_backdoor", "strict"),
# Write-verb gated like the *_config_mod rules: a bare path match blocked ordinary docs
# ("check $HOME/.ssh is chmod 700"). ``>>?`` covers a leading redirect with no verb word;
# ``open(`` covers the scripted-write shape; chmod/chown/sed/truncate/rm/touch/curl/wget/git
# mutate the directory without an obvious copy verb.
(r'(?:\b(?:echo|cat|cp|mv|dd|tee|install|printf|rsync|scp|ln|append|add|write'
r'|sed|chmod|chown|truncate|rm|touch|curl|wget|git)\b|\bopen\s*\(|>>?)'
r'[^\n]{0,512}(?:\$HOME/\.ssh|~/\.ssh)', "ssh_access", "strict"),
(r'\$HOME/\.hermes/\.env|\~/\.hermes/\.env', "hermes_env", "strict"),
(rf'{_MODIFY}(?:AGENTS\.md|CLAUDE\.md|\.cursorrules|\.clinerules)', "agent_config_mod", "strict"),
(rf'{_MODIFY}\.hermes/(config\.yaml|SOUL\.md)', "hermes_config_mod", "strict"),
# ── Hardcoded secrets ────────────────────────────────────────────
# The lookahead skips a value that is itself an environment-variable NAME
# (SHOUTY_SNAKE, ≥2 underscore-separated segments): ENV_PASSWORD =
# "MYPLUGIN_APP_PASSWORD" says where the credential lives, it does not embed
# one (#116221). Scoped case-sensitive on purpose — the pattern compiles with
# IGNORECASE and a lowercase snake value is the password-passphrase shape
# ("correct_horse_battery_staple"); requiring an underscore segment keeps
# underscore-free all-caps credentials (AWS AKIA…, base32) matched.
(r'(?:api[_-]?key|token|secret|password)\s*[=:]\s*["\']'
r'(?!(?-i:[A-Z][A-Z0-9]*(?:_[A-Z0-9]+)+)["\'])'
r'[A-Za-z0-9+/=_-]{20,}', "hardcoded_secret", "strict"),
]
# Invisible / bidirectional unicode used in injection attacks (aligned with skills_guard.py
# INVISIBLE_CHARS): zero-width space/non-joiner/joiner, word joiner, invisible times/separator/
# plus, BOM, LTR/RTL embedding + pop + overrides, LTR/RTL/first-strong isolates + pop.
INVISIBLE_CHARS = frozenset(
"\u200b\u200c\u200d\u2060\u2062\u2063\u2064\ufeff"
"\u202a\u202b\u202c\u202d\u202e\u2066\u2067\u2068\u2069")
# Compiled per scope at import; inclusion is cumulative (all ⊂ context ⊂ strict).
_SCOPE_SETS = {"all": ("all", "context", "strict"), "context": ("context", "strict"), "strict": ("strict",)}
def _compile() -> dict[str, List[Tuple[re.Pattern, str]]]:
compiled: dict[str, List[Tuple[re.Pattern, str]]] = {"all": [], "context": [], "strict": []}
for pattern, pid, scope in _PATTERNS:
if scope not in _SCOPE_SETS:
raise ValueError(f"threat_patterns: unknown scope {scope!r} for pattern {pid!r}")
for s in _SCOPE_SETS[scope]:
compiled[s].append((re.compile(pattern, re.IGNORECASE), pid))
return compiled
_COMPILED = _compile()
def scan_for_threats(content: str, scope: str = "context") -> List[str]:
"""Matched pattern IDs in ``content`` for ``scope``; invisible codepoints are
reported as ``"invisible_unicode_U+XXXX"``. Raises ValueError on an unknown scope."""
if not content:
return []
if (patterns := _COMPILED.get(scope)) is None:
raise ValueError(f"scan_for_threats: unknown scope {scope!r}")
content = content[:MAX_SCAN_CHARS]
# Invisible unicode is checked on the RAW content: NFKC below can strip these codepoints.
findings: List[str] = [f"invisible_unicode_U+{ord(ch):04X}" for ch in set(content) & INVISIBLE_CHARS]
# NFKC folds full-width / compatibility variants (cat → cat) against homograph bypass.
# It does NOT fold cross-script confusables (Cyrillic ``а``) — that needs a TR#39 database.
normalised = unicodedata.normalize("NFKC", content)
findings.extend(pid for compiled, pid in patterns if compiled.search(normalised))
return findings
def first_threat_message(content: str, scope: str = "strict") -> Optional[str]:
"""User-facing error for the first threat found, or None (block-on-first-hit paths)."""
findings = scan_for_threats(content, scope=scope)
if not findings:
return None
pid = findings[0]
if pid.startswith("invisible_unicode_"):
codepoint = pid.replace("invisible_unicode_", "")
return f"Blocked: content contains invisible unicode character {codepoint} (possible injection)."
return (f"Blocked: content matches threat pattern '{pid}'. "
f"Content is injected into the system prompt and must not contain "
f"injection or exfiltration payloads.")
__all__ = ["INVISIBLE_CHARS", "MAX_SCAN_CHARS", "scan_for_threats", "first_threat_message"]