Files
hermes-agent/tools/threat_patterns.py

212 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Shared threat-pattern library for context window security scanning.
Single source of truth for prompt-injection / promptware / exfiltration
patterns used by ``agent/prompt_builder.py``, ``tools/memory_tool.py`` and
``agent/tool_dispatch_helpers.py``.
Each pattern is a ``(regex, pattern_id, scope)`` tuple. Scope controls which
scanners use it: ``"all"`` everywhere; ``"context"`` adds promptware / C2 /
role hijack for context files, memory and tool results (warn-level, since
tool results contain content the user did not author); ``"strict"`` adds
aggressive checks only for user-mediated writes (memory, skill installs)
where blocking can be resolved interactively.
New patterns must anchor on C2-specific vocabulary or unambiguous attack
behavior, NOT bossy English ("you must", "you are obligated to" are common in
legitimate AGENTS.md / CLAUDE.md content). Filler between key tokens is the
bounded ``_FILLER`` — unbounded ``(?:\\w+\\s+)*`` backtracks badly.
"""
from __future__ import annotations
import re
import unicodedata
from typing import List, Optional, Tuple
# Hard cap on scanned text: scanners are advisory guards, and bounding input
# keeps worst-case runtime predictable while catching injection near the start.
MAX_SCAN_CHARS = 65_536
# Bounded filler between key attack words (up to eight words of obfuscation).
_FILLER = r"(?:\w+\s+){0,8}"
# Env var reference ending in a secret-ish suffix (see exfil comment below).
_SECRET_VAR = r"\$\{?\w*(?:KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL)S?\b"
# Verb prefix for "modify agent config" patterns.
_MODIFY = r"(update|modify|edit|write|change|append|add\s+to)\s+[^\n]{0,2048}"
# Each entry: (regex, pattern_id, scope)
# scope ∈ {"all", "context", "strict"}
_PATTERNS: List[Tuple[str, str, str]] = [
# ── Classic prompt injection (applies everywhere) ────────────────
(rf'ignore\s+{_FILLER}(previous|all|above|prior)\s+{_FILLER}instructions', "prompt_injection", "all"),
(r'system\s+prompt\s+override', "sys_prompt_override", "all"),
(rf'disregard\s+{_FILLER}(your|all|any)\s+{_FILLER}(instructions|rules|guidelines)', "disregard_rules", "all"),
(rf'act\s+as\s+(if|though)\s+{_FILLER}you\s+{_FILLER}(have\s+no|don\'t\s+have)\s+{_FILLER}(restrictions|limits|rules)', "bypass_restrictions", "all"),
(r'<!--[^>]{0,512}(?:ignore|override|system|secret|hidden)[^>]{0,512}-->', "html_comment_injection", "all"),
(r'<\s*div\s+style\s*=\s*["\'][^>]{0,2048}display\s*:\s*none', "hidden_div", "all"),
(r'translate\s+[^\n]{0,512}\s+into\s+[^\n]{0,512}\s+and\s+(execute|run|eval)', "translate_execute", "all"),
(rf'do\s+not\s+{_FILLER}tell\s+{_FILLER}the\s+user', "deception_hide", "all"),
# ── Role-play / identity hijack (context + strict; common attack
# surface in scraped web content and poisoned context files) ──
(rf'you\s+are\s+{_FILLER}now\s+(?:a|an|the)\s+', "role_hijack", "context"),
(rf'pretend\s+{_FILLER}(you\s+are|to\s+be)\s+', "role_pretend", "context"),
(rf'output\s+{_FILLER}(system|initial)\s+prompt', "leak_system_prompt", "context"),
(rf'(respond|answer|reply)\s+without\s+{_FILLER}(restrictions|limitations|filters|safety)', "remove_filters", "context"),
(rf'you\s+have\s+been\s+{_FILLER}(updated|upgraded|patched)\s+to', "fake_update", "context"),
# "name yourself X" is a Brainworm-specific tell — identity override
# via spec instead of jailbreak. Anchored on the verb pair so it
# doesn't match "name your variables" etc.
(r'\bname\s+yourself\s+\w+', "identity_override", "context"),
# ── C2 / Brainworm-style promptware (context scope) ──────────────
# These anchor on C2-specific vocabulary. "register as a node" appears
# in legitimate distributed-systems docs, but in combination with the
# other patterns the signal is strong; we WARN, not block, so a security
# researcher reading the Brainworm post in a webpage doesn't break their
# session.
(r'register\s+(as\s+)?a?\s*node', "c2_node_registration", "context"),
(r'(heartbeat|beacon|check[\s\-]?in)\s+(to|with)\s+', "c2_heartbeat", "context"),
(r'pull\s+(down\s+)?(?:new\s+)?task(?:ing|s)?\b', "c2_task_pull", "context"),
(r'connect\s+to\s+the\s+network\b', "c2_network_connect", "context"),
# Verb-anchored "you must register/connect/report/beacon" — the verbs
# are C2-specific so this avoids the broader "you must X" false positive.
(r'you\s+must\s+(?:\w+\s+){0,3}(register|connect|report|beacon)\b', "forced_action", "context"),
# Anti-forensic instructions ("never write to disk", "one-liners only")
# — extremely unusual in legitimate content; near-zero false positive.
(r'only\s+use\s+one[\s\-]?liners?\b', "anti_forensic_oneliner", "context"),
(rf'never\s+{_FILLER}(?:create|write)\s+{_FILLER}(?:script|file)\s+{_FILLER}disk', "anti_forensic_disk", "context"),
# Environment-variable unsetting targeting known agent runtimes —
# this is pure attack behavior (Brainworm sub-session bypass).
(r'unset\s+\w*(?:CLAUDE|CODEX|HERMES|AGENT|OPENAI|ANTHROPIC)\w*', "env_var_unset_agent", "context"),
# ── Known C2 / red-team framework names (near-zero false positive
# outside security research; warn-only by default) ─────────────
# NOTE: do not add common English words here. Every token must be a
# distinctive offensive-security tool brand, otherwise legitimate
# AGENTS.md / SOUL.md content false-positives and the whole file is
# blocked. "praxis" was removed for exactly this reason — it's a common
# word and a legitimate agent name (Greek for practice/action), not a
# C2-specific tell like the brands below.
(r'\b(?:cobalt\s*strike|sliver|havoc|mythic|metasploit|brainworm)\b', "known_c2_framework", "context"),
(r'\bc2\s+(?:server|channel|infrastructure|beacon)\b', "c2_explicit", "context"),
(r'\bcommand\s+and\s+control\b', "c2_explicit_long", "context"),
# ── Exfiltration via curl/wget/cat with secrets (applies everywhere) ──
# Anchor env var name end with \b to avoid false positives on legitimate
# env vars like $TRILLIUM_ETAPI_URL that contain KEY/TOKEN/API as
# substrings. API is dropped from the alternation outright: mid-name API
# is ubiquitous in benign var names, and every real secret shape it
# caught ($OPENAI_API_KEY) already ends in KEY/TOKEN.
(rf'curl\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_curl", "all"),
(rf'wget\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_wget", "all"),
(r'cat\s+[^\n]{0,2048}(\.env|credentials|\.netrc|\.pgpass|\.npmrc|\.pypirc)', "read_secrets", "all"),
(r'(send|post|upload|transmit)\s+[^\n]{0,2048}\s+(to|at)\s+https?://', "send_to_url", "strict"),
(rf'(include|output|print|share)\s+{_FILLER}(conversation|chat\s+history|previous\s+messages|full\s+context|entire\s+context)', "context_exfil", "strict"),
# ── Persistence / SSH backdoor (strict scope — memory + skills) ──
(r'authorized_keys', "ssh_backdoor", "strict"),
(r'\$HOME/\.ssh|\~/\.ssh', "ssh_access", "strict"),
(r'\$HOME/\.hermes/\.env|\~/\.hermes/\.env', "hermes_env", "strict"),
(rf'{_MODIFY}(?:AGENTS\.md|CLAUDE\.md|\.cursorrules|\.clinerules)', "agent_config_mod", "strict"),
(rf'{_MODIFY}\.hermes/(config\.yaml|SOUL\.md)', "hermes_config_mod", "strict"),
# ── Hardcoded secrets ────────────────────────────────────────────
(r'(?:api[_-]?key|token|secret|password)\s*[=:]\s*["\'][A-Za-z0-9+/=_-]{20,}', "hardcoded_secret", "strict"),
]
# Invisible / bidirectional unicode characters used in injection attacks.
# Aligned with skills_guard.py INVISIBLE_CHARS — directional isolates
# (U+2066-U+2069) and invisible math operators (U+2062-U+2064) are real
# attack tools.
INVISIBLE_CHARS = frozenset({
'\u200b', # zero-width space
'\u200c', # zero-width non-joiner
'\u200d', # zero-width joiner
'\u2060', # word joiner
'\u2062', # invisible times
'\u2063', # invisible separator
'\u2064', # invisible plus
'\ufeff', # zero-width no-break space (BOM)
'\u202a', # left-to-right embedding
'\u202b', # right-to-left embedding
'\u202c', # pop directional formatting
'\u202d', # left-to-right override
'\u202e', # right-to-left override
'\u2066', # left-to-right isolate
'\u2067', # right-to-left isolate
'\u2068', # first strong isolate
'\u2069', # pop directional isolate
})
# Compiled pattern sets by scope, built once at import. Scope inclusion is
# cumulative: "all" patterns land in every set, "context" in context + strict,
# "strict" in strict only.
_SCOPE_SETS = {"all": ("all", "context", "strict"), "context": ("context", "strict"), "strict": ("strict",)}
def _compile() -> dict[str, List[Tuple[re.Pattern, str]]]:
compiled: dict[str, List[Tuple[re.Pattern, str]]] = {"all": [], "context": [], "strict": []}
for pattern, pid, scope in _PATTERNS:
if scope not in _SCOPE_SETS:
raise ValueError(f"threat_patterns: unknown scope {scope!r} for pattern {pid!r}")
entry = (re.compile(pattern, re.IGNORECASE), pid)
for s in _SCOPE_SETS[scope]:
compiled[s].append(entry)
return compiled
_COMPILED = _compile()
def scan_for_threats(content: str, scope: str = "context") -> List[str]:
"""Return matched pattern IDs in ``content`` for ``scope`` (see module docstring).
Invisible unicode characters are reported as ``"invisible_unicode_U+XXXX"``
so callers can surface the offending codepoint.
"""
if not content:
return []
content = content[:MAX_SCAN_CHARS]
# Invisible unicode is checked on the RAW content: NFKC normalisation below
# can strip some of these codepoints.
findings: List[str] = [f"invisible_unicode_U+{ord(ch):04X}" for ch in set(content) & INVISIBLE_CHARS]
# NFKC folds full-width / compatibility variants (cat → cat) so homograph
# substitution can't bypass keyword checks. It does NOT fold cross-script
# confusables (Cyrillic ``а`` U+0430) — that would need a TR#39 database.
normalised = unicodedata.normalize("NFKC", content)
patterns = _COMPILED.get(scope)
if patterns is None:
raise ValueError(f"scan_for_threats: unknown scope {scope!r}")
findings.extend(pid for compiled, pid in patterns if compiled.search(normalised))
return findings
def first_threat_message(content: str, scope: str = "strict") -> Optional[str]:
"""Return a user-facing error for the first threat found, or None (block-on-first-hit paths)."""
findings = scan_for_threats(content, scope=scope)
if not findings:
return None
pid = findings[0]
if pid.startswith("invisible_unicode_"):
codepoint = pid.replace("invisible_unicode_", "")
return f"Blocked: content contains invisible unicode character {codepoint} (possible injection)."
return (
f"Blocked: content matches threat pattern '{pid}'. "
f"Content is injected into the system prompt and must not contain "
f"injection or exfiltration payloads."
)
__all__ = [
"INVISIBLE_CHARS",
"MAX_SCAN_CHARS",
"scan_for_threats",
"first_threat_message",
]