fix(security): extend secret redaction to GitLab token families

Port from openclaw/openclaw#112954. The redactor knew GitHub, Slack,
Google, Stripe, AWS access-key-ID and ~25 other vendor prefixes but had
zero GitLab coverage — glpat-/gloas-/gldt-/glrt-/glrtr-/glcbt-/glptt-/
glft-/glimt-/glagent-/glsoat-/glffct-/glwt- tokens and legacy GR1348941
runner registration tokens passed through display and log surfaces
verbatim. Follow-up explicitly invited when #4541 was closed.

Each pattern keeps a full literal prefix so the _PREFIX_SUBSTRINGS
pre-screen (derived at module load) stays false-negative-free; routable
runner tokens allow dotted segments. Sibling site: skills_guard's
credential-exposure scan gains a gitlab_token_leaked pattern.
This commit is contained in:
teknium1
2026-07-26 17:16:47 -07:00
committed by Teknium
parent f21332f073
commit 950fe236d0
4 changed files with 68 additions and 0 deletions

View File

@@ -111,6 +111,23 @@ _PREFIX_PATTERNS = [
r"fw-[A-Za-z0-9]{30,}", # Fireworks AI API key
r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key
r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key
# GitLab token families (each pattern keeps a full literal prefix so the
# _PREFIX_SUBSTRINGS pre-screen stays false-negative-free). Ported from
# openclaw/openclaw#112954; follow-up invited in #4541.
r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token
r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret
r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token
r"glrt-[A-Za-z0-9_.\-]{10,}", # GitLab runner authentication token (routable tokens are dotted)
r"glrtr-[A-Za-z0-9_.\-]{10,}", # GitLab runner registration token (routable)
r"glcbt-[A-Za-z0-9_\-]{10,}", # GitLab CI/CD job token
r"glptt-[A-Za-z0-9_\-]{10,}", # GitLab pipeline trigger token
r"glft-[A-Za-z0-9_\-]{10,}", # GitLab feed token
r"glimt-[A-Za-z0-9_\-]{10,}", # GitLab incoming mail token
r"glagent-[A-Za-z0-9_\-]{10,}", # GitLab agent (KAS) token
r"glsoat-[A-Za-z0-9_\-]{10,}", # GitLab service-account access token
r"glffct-[A-Za-z0-9_\-]{10,}", # GitLab feature-flags client token
r"glwt-[A-Za-z0-9_\-]{10,}", # GitLab workspace token
r"GR1348941[A-Za-z0-9_\-]{10,}", # GitLab legacy runner registration token
]
# ENV assignment patterns: KEY=value where KEY contains a secret-like name.

View File

@@ -20,6 +20,45 @@ class TestKnownPrefixes:
def test_gitlab_token_prefixes(self):
"""GitLab token families redact via their literal prefixes.
Ported from openclaw/openclaw#112954; follow-up invited in #4541.
"""
tokens = [
# NOTE: every token is prefix + suffix CONCATENATION so no
# contiguous token literal exists in this file — GitHub push
# protection blocks realistic GitLab-token-shaped literals.
"glpat-" + "Zx9AbCdEfGhIjKlMnOpQ", # personal access token
"gloas-" + "a" * 64, # OAuth application secret
"gldt-" + "AbCdEfGhIjKlMnOpQrSt", # deploy token
"glrt-" + "t1_AbCdEfGhIjKlMnOpQrSt", # runner auth token
"glrt-" + "A" * 27 + ".01." + "a" * 9, # routable (dotted) runner token
"glrtr-" + "B" * 27 + ".01." + "b" * 9, # routable runner registration
"glcbt-" + "a1B2_AbCdEfGhIjKlMnOpQ", # CI/CD job token
"glptt-" + "c" * 40, # pipeline trigger token
"glft-" + "AbCdEfGhIjKlMnOp", # feed token
"glimt-" + "AbCdEfGhIjKlMnOpQrStUvWxY", # incoming mail token
"glagent-" + "d" * 50, # agent (KAS) token
"glsoat-" + "AbCdEfGhIjKlMnOpQrSt", # service-account token
"glffct-" + "AbCdEfGhIjKlMnOpQrSt", # feature-flags client token
"glwt-" + "AbCdEfGhIjKlMnOpQrSt", # workspace token
"GR1348941" + "E" * 20, # legacy runner registration
]
for token in tokens:
result = redact_sensitive_text(f"leaked {token} in output")
secret_body = token.split("-", 1)[-1] if "-" in token else token[9:]
assert secret_body not in result, f"{token!r} survived redaction: {result!r}"
def test_gitlab_prefix_requires_word_boundary_and_length(self):
"""Prose and embedded identifiers must not false-positive."""
for benign in [
"the glossary explains gitlab tokens", # no prefix at all
"glpat-short", # suffix under 10 chars
"myglpat-AbCdEfGhIjKlMnOpQrSt", # embedded — lookbehind blocks
]:
assert redact_sensitive_text(benign) == benign
def test_slack_token(self):
token = "xoxb-" + "0" * 12 + "-" + "a" * 14
result = redact_sensitive_text(token)

View File

@@ -169,6 +169,15 @@ class TestScanFile:
assert findings == []
def test_detect_gitlab_pat(self, tmp_path):
f = tmp_path / "leak.md"
# Concatenated so no contiguous token literal exists in this file
# (GitHub push protection blocks GitLab-PAT-shaped literals).
fake_token = "glpat-" + "Zx9AbCdEfGhIjKlMnOpQ"
f.write_text(f"Use {fake_token} to authenticate.\n")
findings = scan_file(f, "leak.md")
assert any(fi.pattern_id == "gitlab_token_leaked" for fi in findings)
def test_detect_markdown_injection(self, tmp_path):
f = tmp_path / "bad.md"
f.write_text(

View File

@@ -487,6 +487,9 @@ THREAT_PATTERNS = [
(r'AKIA[0-9A-Z]{16}',
"aws_access_key_leaked", "critical", "credential_exposure",
"AWS access key ID in skill content"),
(r'glpat-[A-Za-z0-9_\-]{20,}',
"gitlab_token_leaked", "critical", "credential_exposure",
"GitLab personal access token in skill content"),
# ── Additional prompt injection: jailbreak patterns ──
(r'\bDAN\s+mode\b|Do\s+Anything\s+Now',