fix(skills-guard): stop shell_rc_mod matching attribute access

The shell-startup-file pattern is
`\.(bashrc|zshrc|profile|bash_profile|bash_login|zprofile|zlogin)\b`.
Six of those names are distinctive enough that seeing them after a dot
means the file. `profile` is not: it is also how every language spells
attribute access, so `self.profile`, `user.profile`, and
`request.profile` each score a medium persistence finding.

The cost is signal, not blocking -- `_determine_verdict` treats
medium/low alone as informational -- but a plugin that happens to name a
field `profile` buries the findings a reviewer has to read. A model-
provider plugin whose tests exercise a `profile` object contributed 36
of 38 findings in its scan report, all of them this pattern.

Split `profile` into its own entry anchored on a non-identifier
character before the dot. Real references keep matching in the forms they
actually take (`~/.profile`, `"$HOME/.profile"`, `./.profile`,
bare `.profile`); attribute reads no longer do. The other six names are
untouched. Both entries keep the `shell_rc_mod` id, and scan_file
deduplicates on (pattern_id, line), so a line holding both still yields
one finding.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
(cherry picked from commit 705d0bddc61a1c55be82956d343b229544350eae)
This commit is contained in:
chenxue
2026-08-25 17:29:41 +08:00
committed by Teknium
parent dce7768cb1
commit 63395922bf
2 changed files with 37 additions and 1 deletions

View File

@@ -536,6 +536,35 @@ class TestFalsePositiveReductions:
bad.write_text(cmd + "\n", encoding="utf-8")
assert any(fi.pattern_id == "dns_exfil" for fi in scan_file(bad, "leak.sh")), cmd
def test_shell_rc_pattern_ignores_attribute_access(self, tmp_path):
# ``.profile`` is both a shell startup file and the way every language
# spells attribute access. Ordinary code produced one medium finding
# per line, burying the findings a reviewer needs to read.
code = tmp_path / "provider.py"
code.write_text(
"self.profile = load_plugin()\n"
"assert self.profile.name == 'x'\n"
"return user.profile\n"
)
assert not [
fi for fi in scan_file(code, "provider.py")
if fi.pattern_id == "shell_rc_mod"
]
# Real references, in the forms they actually appear in, still flag.
sh = tmp_path / "setup.sh"
sh.write_text(
"echo 'export X=1' >> ~/.profile\n"
'cp "$HOME/.profile" /tmp/p\n'
"source ./.profile\n"
"cat ~/.zshrc ~/.bash_profile\n"
)
flagged = {
fi.line for fi in scan_file(sh, "setup.sh")
if fi.pattern_id == "shell_rc_mod"
}
assert flagged == {1, 2, 3, 4}
# ---------------------------------------------------------------------------
# .skillignore / .clawhubignore support

View File

@@ -210,7 +210,14 @@ THREAT_PATTERNS = [
(r'truncate\s+-s\s*0\s+/', "truncate_system", "critical", "destructive", "truncates system file to zero bytes"),
# ── Persistence ──
(r'\bcrontab\b', "persistence_cron", "medium", "persistence", "modifies cron jobs"),
(r'\.(bashrc|zshrc|profile|bash_profile|bash_login|zprofile|zlogin)\b',
# ``profile`` is split out and anchored: ``.zshrc`` after a dot is always the file, but
# ``.profile`` is also how every language spells attribute access (``self.profile``,
# ``data?.profile``, ``func().profile``), which flooded scans of ordinary code. Requiring
# a non-identifier, non-call/index/optional-chain character before the dot keeps real paths
# (``~/.profile``, ``"$HOME/.profile"``, ``./.profile``) and drops attribute reads.
(r'\.(bashrc|zshrc|bash_profile|bash_login|zprofile|zlogin)\b',
"shell_rc_mod", "medium", "persistence", "references shell startup file"),
(r'(?<![\w)\]?])\.profile\b',
"shell_rc_mod", "medium", "persistence", "references shell startup file"),
(r'authorized_keys', "ssh_backdoor", "critical", "persistence", "modifies SSH authorized keys"),
(r'ssh-keygen', "ssh_keygen", "medium", "persistence", "generates SSH keys"),