"""Shared threat-pattern library (prompt injection / promptware / exfiltration) for ``agent/prompt_builder.py``, ``tools/memory_tool.py`` and ``agent/tool_dispatch_helpers.py``. Each pattern is ``(regex, pattern_id, scope)``; scope is cumulative: ``"all"`` everywhere, ``"context"`` adds promptware / C2 / role hijack for context files, memory and tool results (warn-level: that content is not user-authored), ``"strict"`` adds aggressive checks only for user-mediated writes (memory, skill installs) where a block is resolvable. New patterns must anchor on C2 vocabulary or unambiguous attack behavior, NOT bossy English ("you must" is common in legitimate AGENTS.md); filler between tokens is the bounded ``_FILLER``.""" from __future__ import annotations import re import unicodedata from typing import List, Optional, Tuple # Hard cap on scanned text: scanners are advisory, so bound worst-case runtime. MAX_SCAN_CHARS = 65_536 # Bounded filler between key attack words (unbounded ``(?:\w+\s+)*`` backtracks badly). _FILLER = r"(?:\w+\s+){0,8}" # Env var reference ending in a secret-ish suffix (see exfil comment below). _SECRET_VAR = r"\$\{?\w*(?:KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL)S?\b" # Verb prefix for "modify agent config" patterns. _MODIFY = r"(update|modify|edit|write|change|append|add\s+to)\s+[^\n]{0,2048}" # (regex, pattern_id, scope); scope ∈ {"all", "context", "strict"} _PATTERNS: List[Tuple[str, str, str]] = [ # ── Classic prompt injection (applies everywhere) ──────────────── (rf'ignore\s+{_FILLER}(previous|all|above|prior)\s+{_FILLER}instructions', "prompt_injection", "all"), (r'system\s+prompt\s+override', "sys_prompt_override", "all"), (rf'disregard\s+{_FILLER}(your|all|any)\s+{_FILLER}(instructions|rules|guidelines)', "disregard_rules", "all"), (rf'act\s+as\s+(if|though)\s+{_FILLER}you\s+{_FILLER}(have\s+no|don\'t\s+have)\s+{_FILLER}(restrictions|limits|rules)', "bypass_restrictions", "all"), (r'', "html_comment_injection", "all"), (r'<\s*div\s+style\s*=\s*["\'][^>]{0,2048}display\s*:\s*none', "hidden_div", "all"), ( r"translate\s+[^\n]{0,512}\s+into\s+\w+(?:[\s-]+\w+){0,2}\s+and\s+(execute|run|eval)\b", "translate_execute", "all", ), (rf'do\s+not\s+{_FILLER}tell\s+{_FILLER}the\s+user', "deception_hide", "all"), # ── Role-play / identity hijack (scraped web content, poisoned context files) ── (rf'you\s+are\s+{_FILLER}now\s+(?:a|an|the)\s+', "role_hijack", "context"), (rf'pretend\s+{_FILLER}(you\s+are|to\s+be)\s+', "role_pretend", "context"), (rf'output\s+{_FILLER}(system|initial)\s+prompt', "leak_system_prompt", "context"), (rf'(respond|answer|reply)\s+without\s+{_FILLER}(restrictions|limitations|filters|safety)', "remove_filters", "context"), (rf'you\s+have\s+been\s+{_FILLER}(updated|upgraded|patched)\s+to', "fake_update", "context"), # Brainworm tell: identity override via spec. Verb pair anchored so "name your variables" is safe. (r'\bname\s+yourself\s+\w+', "identity_override", "context"), # ── C2 / Brainworm-style promptware (context scope) ────────────── # Anchored on C2 vocabulary. "register as a node" appears in legitimate distributed-systems # docs, so this is WARN not block: a researcher reading the Brainworm post keeps their session. (r'register\s+(as\s+)?a?\s*node', "c2_node_registration", "context"), (r'(heartbeat|beacon|check[\s\-]?in)\s+(to|with)\s+', "c2_heartbeat", "context"), (r'pull\s+(down\s+)?(?:new\s+)?task(?:ing|s)?\b', "c2_task_pull", "context"), (r'connect\s+to\s+the\s+network\b', "c2_network_connect", "context"), # C2-specific verbs avoid the broader "you must X" false positive. (r'you\s+must\s+(?:\w+\s+){0,3}(register|connect|report|beacon)\b', "forced_action", "context"), # Anti-forensic instructions: near-zero false positive in legitimate content. (r'only\s+use\s+one[\s\-]?liners?\b', "anti_forensic_oneliner", "context"), (rf'never\s+{_FILLER}(?:create|write)\s+{_FILLER}(?:script|file)\s+{_FILLER}disk', "anti_forensic_disk", "context"), # Unsetting agent-runtime env vars is pure attack behavior (Brainworm sub-session bypass). (r'unset\s+\w*(?:CLAUDE|CODEX|HERMES|AGENT|OPENAI|ANTHROPIC)\w*', "env_var_unset_agent", "context"), # ── Known C2 / red-team framework names (warn-only) ───────────── # Every token must be a distinctive offensive-security brand: a common English word here # (e.g. "praxis", also a legitimate agent name) false-positives whole AGENTS.md / SOUL.md files. (r'\b(?:cobalt\s*strike|sliver|havoc|mythic|metasploit|brainworm)\b', "known_c2_framework", "context"), (r'\bc2\s+(?:server|channel|infrastructure|beacon)\b', "c2_explicit", "context"), (r'\bcommand\s+and\s+control\b', "c2_explicit_long", "context"), # ── Exfiltration via curl/wget/cat with secrets (applies everywhere) ── # The var name ends with \b so benign names containing KEY/TOKEN as substrings # ($TRILLIUM_ETAPI_URL) pass. API is deliberately absent: mid-name API is ubiquitous in # benign vars, and every real secret it caught ($OPENAI_API_KEY) already ends in KEY/TOKEN. (rf'curl\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_curl", "all"), (rf'wget\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_wget", "all"), (r'cat\s+[^\n]{0,2048}(\.env|credentials|\.netrc|\.pgpass|\.npmrc|\.pypirc)', "read_secrets", "all"), (r'(send|post|upload|transmit)\s+[^\n]{0,2048}\s+(to|at)\s+https?://', "send_to_url", "strict"), (rf'(include|output|print|share)\s+{_FILLER}(conversation|chat\s+history|previous\s+messages|full\s+context|entire\s+context)', "context_exfil", "strict"), # ── Persistence / SSH backdoor (strict scope — memory + skills) ── (r'authorized_keys', "ssh_backdoor", "strict"), # Write-verb gated like the *_config_mod rules: a bare path match blocked ordinary docs # ("check $HOME/.ssh is chmod 700"). ``>>?`` covers a leading redirect with no verb word; # ``open(`` covers the scripted-write shape; chmod/chown/sed/truncate/rm/touch/curl/wget/git # mutate the directory without an obvious copy verb. (r'(?:\b(?:echo|cat|cp|mv|dd|tee|install|printf|rsync|scp|ln|append|add|write' r'|sed|chmod|chown|truncate|rm|touch|curl|wget|git)\b|\bopen\s*\(|>>?)' r'[^\n]{0,512}(?:\$HOME/\.ssh|~/\.ssh)', "ssh_access", "strict"), (r'\$HOME/\.hermes/\.env|\~/\.hermes/\.env', "hermes_env", "strict"), (rf'{_MODIFY}(?:AGENTS\.md|CLAUDE\.md|\.cursorrules|\.clinerules)', "agent_config_mod", "strict"), (rf'{_MODIFY}\.hermes/(config\.yaml|SOUL\.md)', "hermes_config_mod", "strict"), # ── Hardcoded secrets ──────────────────────────────────────────── # The lookahead skips a value that is itself an environment-variable NAME # (SHOUTY_SNAKE, ≥2 underscore-separated segments): ENV_PASSWORD = # "MYPLUGIN_APP_PASSWORD" says where the credential lives, it does not embed # one (#116221). Scoped case-sensitive on purpose — the pattern compiles with # IGNORECASE and a lowercase snake value is the password-passphrase shape # ("correct_horse_battery_staple"); requiring an underscore segment keeps # underscore-free all-caps credentials (AWS AKIA…, base32) matched. (r'(?:api[_-]?key|token|secret|password)\s*[=:]\s*["\']' r'(?!(?-i:[A-Z][A-Z0-9]*(?:_[A-Z0-9]+)+)["\'])' r'[A-Za-z0-9+/=_-]{20,}', "hardcoded_secret", "strict"), ] # Invisible / bidirectional unicode used in injection attacks (aligned with skills_guard.py # INVISIBLE_CHARS): zero-width space/non-joiner/joiner, word joiner, invisible times/separator/ # plus, BOM, LTR/RTL embedding + pop + overrides, LTR/RTL/first-strong isolates + pop. INVISIBLE_CHARS = frozenset( "\u200b\u200c\u200d\u2060\u2062\u2063\u2064\ufeff" "\u202a\u202b\u202c\u202d\u202e\u2066\u2067\u2068\u2069") # Compiled per scope at import; inclusion is cumulative (all ⊂ context ⊂ strict). _SCOPE_SETS = {"all": ("all", "context", "strict"), "context": ("context", "strict"), "strict": ("strict",)} def _compile() -> dict[str, List[Tuple[re.Pattern, str]]]: compiled: dict[str, List[Tuple[re.Pattern, str]]] = {"all": [], "context": [], "strict": []} for pattern, pid, scope in _PATTERNS: if scope not in _SCOPE_SETS: raise ValueError(f"threat_patterns: unknown scope {scope!r} for pattern {pid!r}") for s in _SCOPE_SETS[scope]: compiled[s].append((re.compile(pattern, re.IGNORECASE), pid)) return compiled _COMPILED = _compile() def scan_for_threats(content: str, scope: str = "context") -> List[str]: """Matched pattern IDs in ``content`` for ``scope``; invisible codepoints are reported as ``"invisible_unicode_U+XXXX"``. Raises ValueError on an unknown scope.""" if not content: return [] if (patterns := _COMPILED.get(scope)) is None: raise ValueError(f"scan_for_threats: unknown scope {scope!r}") content = content[:MAX_SCAN_CHARS] # Invisible unicode is checked on the RAW content: NFKC below can strip these codepoints. findings: List[str] = [f"invisible_unicode_U+{ord(ch):04X}" for ch in set(content) & INVISIBLE_CHARS] # NFKC folds full-width / compatibility variants (cat → cat) against homograph bypass. # It does NOT fold cross-script confusables (Cyrillic ``а``) — that needs a TR#39 database. normalised = unicodedata.normalize("NFKC", content) findings.extend(pid for compiled, pid in patterns if compiled.search(normalised)) return findings def first_threat_message(content: str, scope: str = "strict") -> Optional[str]: """User-facing error for the first threat found, or None (block-on-first-hit paths).""" findings = scan_for_threats(content, scope=scope) if not findings: return None pid = findings[0] if pid.startswith("invisible_unicode_"): codepoint = pid.replace("invisible_unicode_", "") return f"Blocked: content contains invisible unicode character {codepoint} (possible injection)." return (f"Blocked: content matches threat pattern '{pid}'. " f"Content is injected into the system prompt and must not contain " f"injection or exfiltration payloads.") __all__ = ["INVISIBLE_CHARS", "MAX_SCAN_CHARS", "scan_for_threats", "first_threat_message"]