From 950fe236d0e08d0739d6a2bd90290c30ebcf5272 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sun, 26 Jul 2026 17:16:47 -0700 Subject: [PATCH] fix(security): extend secret redaction to GitLab token families MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port from openclaw/openclaw#112954. The redactor knew GitHub, Slack, Google, Stripe, AWS access-key-ID and ~25 other vendor prefixes but had zero GitLab coverage — glpat-/gloas-/gldt-/glrt-/glrtr-/glcbt-/glptt-/ glft-/glimt-/glagent-/glsoat-/glffct-/glwt- tokens and legacy GR1348941 runner registration tokens passed through display and log surfaces verbatim. Follow-up explicitly invited when #4541 was closed. Each pattern keeps a full literal prefix so the _PREFIX_SUBSTRINGS pre-screen (derived at module load) stays false-negative-free; routable runner tokens allow dotted segments. Sibling site: skills_guard's credential-exposure scan gains a gitlab_token_leaked pattern. --- agent/redact.py | 17 ++++++++++++++ tests/agent/test_redact.py | 39 ++++++++++++++++++++++++++++++++ tests/tools/test_skills_guard.py | 9 ++++++++ tools/skills_guard.py | 3 +++ 4 files changed, 68 insertions(+) diff --git a/agent/redact.py b/agent/redact.py index b104ad4c22..b8b9be8912 100644 --- a/agent/redact.py +++ b/agent/redact.py @@ -111,6 +111,23 @@ _PREFIX_PATTERNS = [ r"fw-[A-Za-z0-9]{30,}", # Fireworks AI API key r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key + # GitLab token families (each pattern keeps a full literal prefix so the + # _PREFIX_SUBSTRINGS pre-screen stays false-negative-free). Ported from + # openclaw/openclaw#112954; follow-up invited in #4541. + r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token + r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret + r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token + r"glrt-[A-Za-z0-9_.\-]{10,}", # GitLab runner authentication token (routable tokens are dotted) + r"glrtr-[A-Za-z0-9_.\-]{10,}", # GitLab runner registration token (routable) + r"glcbt-[A-Za-z0-9_\-]{10,}", # GitLab CI/CD job token + r"glptt-[A-Za-z0-9_\-]{10,}", # GitLab pipeline trigger token + r"glft-[A-Za-z0-9_\-]{10,}", # GitLab feed token + r"glimt-[A-Za-z0-9_\-]{10,}", # GitLab incoming mail token + r"glagent-[A-Za-z0-9_\-]{10,}", # GitLab agent (KAS) token + r"glsoat-[A-Za-z0-9_\-]{10,}", # GitLab service-account access token + r"glffct-[A-Za-z0-9_\-]{10,}", # GitLab feature-flags client token + r"glwt-[A-Za-z0-9_\-]{10,}", # GitLab workspace token + r"GR1348941[A-Za-z0-9_\-]{10,}", # GitLab legacy runner registration token ] # ENV assignment patterns: KEY=value where KEY contains a secret-like name. diff --git a/tests/agent/test_redact.py b/tests/agent/test_redact.py index c138a7473c..2ea405f051 100644 --- a/tests/agent/test_redact.py +++ b/tests/agent/test_redact.py @@ -20,6 +20,45 @@ class TestKnownPrefixes: + def test_gitlab_token_prefixes(self): + """GitLab token families redact via their literal prefixes. + + Ported from openclaw/openclaw#112954; follow-up invited in #4541. + """ + tokens = [ + # NOTE: every token is prefix + suffix CONCATENATION so no + # contiguous token literal exists in this file — GitHub push + # protection blocks realistic GitLab-token-shaped literals. + "glpat-" + "Zx9AbCdEfGhIjKlMnOpQ", # personal access token + "gloas-" + "a" * 64, # OAuth application secret + "gldt-" + "AbCdEfGhIjKlMnOpQrSt", # deploy token + "glrt-" + "t1_AbCdEfGhIjKlMnOpQrSt", # runner auth token + "glrt-" + "A" * 27 + ".01." + "a" * 9, # routable (dotted) runner token + "glrtr-" + "B" * 27 + ".01." + "b" * 9, # routable runner registration + "glcbt-" + "a1B2_AbCdEfGhIjKlMnOpQ", # CI/CD job token + "glptt-" + "c" * 40, # pipeline trigger token + "glft-" + "AbCdEfGhIjKlMnOp", # feed token + "glimt-" + "AbCdEfGhIjKlMnOpQrStUvWxY", # incoming mail token + "glagent-" + "d" * 50, # agent (KAS) token + "glsoat-" + "AbCdEfGhIjKlMnOpQrSt", # service-account token + "glffct-" + "AbCdEfGhIjKlMnOpQrSt", # feature-flags client token + "glwt-" + "AbCdEfGhIjKlMnOpQrSt", # workspace token + "GR1348941" + "E" * 20, # legacy runner registration + ] + for token in tokens: + result = redact_sensitive_text(f"leaked {token} in output") + secret_body = token.split("-", 1)[-1] if "-" in token else token[9:] + assert secret_body not in result, f"{token!r} survived redaction: {result!r}" + + def test_gitlab_prefix_requires_word_boundary_and_length(self): + """Prose and embedded identifiers must not false-positive.""" + for benign in [ + "the glossary explains gitlab tokens", # no prefix at all + "glpat-short", # suffix under 10 chars + "myglpat-AbCdEfGhIjKlMnOpQrSt", # embedded — lookbehind blocks + ]: + assert redact_sensitive_text(benign) == benign + def test_slack_token(self): token = "xoxb-" + "0" * 12 + "-" + "a" * 14 result = redact_sensitive_text(token) diff --git a/tests/tools/test_skills_guard.py b/tests/tools/test_skills_guard.py index f740bbd3a8..b7f08256d7 100644 --- a/tests/tools/test_skills_guard.py +++ b/tests/tools/test_skills_guard.py @@ -169,6 +169,15 @@ class TestScanFile: assert findings == [] + def test_detect_gitlab_pat(self, tmp_path): + f = tmp_path / "leak.md" + # Concatenated so no contiguous token literal exists in this file + # (GitHub push protection blocks GitLab-PAT-shaped literals). + fake_token = "glpat-" + "Zx9AbCdEfGhIjKlMnOpQ" + f.write_text(f"Use {fake_token} to authenticate.\n") + findings = scan_file(f, "leak.md") + assert any(fi.pattern_id == "gitlab_token_leaked" for fi in findings) + def test_detect_markdown_injection(self, tmp_path): f = tmp_path / "bad.md" f.write_text( diff --git a/tools/skills_guard.py b/tools/skills_guard.py index 47f200ab88..2ba266e4fd 100644 --- a/tools/skills_guard.py +++ b/tools/skills_guard.py @@ -487,6 +487,9 @@ THREAT_PATTERNS = [ (r'AKIA[0-9A-Z]{16}', "aws_access_key_leaked", "critical", "credential_exposure", "AWS access key ID in skill content"), + (r'glpat-[A-Za-z0-9_\-]{20,}', + "gitlab_token_leaked", "critical", "credential_exposure", + "GitLab personal access token in skill content"), # ── Additional prompt injection: jailbreak patterns ── (r'\bDAN\s+mode\b|Do\s+Anything\s+Now',