diff --git a/agent/redact.py b/agent/redact.py index b104ad4c22..b8b9be8912 100644 --- a/agent/redact.py +++ b/agent/redact.py @@ -111,6 +111,23 @@ _PREFIX_PATTERNS = [ r"fw-[A-Za-z0-9]{30,}", # Fireworks AI API key r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key + # GitLab token families (each pattern keeps a full literal prefix so the + # _PREFIX_SUBSTRINGS pre-screen stays false-negative-free). Ported from + # openclaw/openclaw#112954; follow-up invited in #4541. + r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token + r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret + r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token + r"glrt-[A-Za-z0-9_.\-]{10,}", # GitLab runner authentication token (routable tokens are dotted) + r"glrtr-[A-Za-z0-9_.\-]{10,}", # GitLab runner registration token (routable) + r"glcbt-[A-Za-z0-9_\-]{10,}", # GitLab CI/CD job token + r"glptt-[A-Za-z0-9_\-]{10,}", # GitLab pipeline trigger token + r"glft-[A-Za-z0-9_\-]{10,}", # GitLab feed token + r"glimt-[A-Za-z0-9_\-]{10,}", # GitLab incoming mail token + r"glagent-[A-Za-z0-9_\-]{10,}", # GitLab agent (KAS) token + r"glsoat-[A-Za-z0-9_\-]{10,}", # GitLab service-account access token + r"glffct-[A-Za-z0-9_\-]{10,}", # GitLab feature-flags client token + r"glwt-[A-Za-z0-9_\-]{10,}", # GitLab workspace token + r"GR1348941[A-Za-z0-9_\-]{10,}", # GitLab legacy runner registration token ] # ENV assignment patterns: KEY=value where KEY contains a secret-like name. diff --git a/tests/agent/test_redact.py b/tests/agent/test_redact.py index c138a7473c..2ea405f051 100644 --- a/tests/agent/test_redact.py +++ b/tests/agent/test_redact.py @@ -20,6 +20,45 @@ class TestKnownPrefixes: + def test_gitlab_token_prefixes(self): + """GitLab token families redact via their literal prefixes. + + Ported from openclaw/openclaw#112954; follow-up invited in #4541. + """ + tokens = [ + # NOTE: every token is prefix + suffix CONCATENATION so no + # contiguous token literal exists in this file — GitHub push + # protection blocks realistic GitLab-token-shaped literals. + "glpat-" + "Zx9AbCdEfGhIjKlMnOpQ", # personal access token + "gloas-" + "a" * 64, # OAuth application secret + "gldt-" + "AbCdEfGhIjKlMnOpQrSt", # deploy token + "glrt-" + "t1_AbCdEfGhIjKlMnOpQrSt", # runner auth token + "glrt-" + "A" * 27 + ".01." + "a" * 9, # routable (dotted) runner token + "glrtr-" + "B" * 27 + ".01." + "b" * 9, # routable runner registration + "glcbt-" + "a1B2_AbCdEfGhIjKlMnOpQ", # CI/CD job token + "glptt-" + "c" * 40, # pipeline trigger token + "glft-" + "AbCdEfGhIjKlMnOp", # feed token + "glimt-" + "AbCdEfGhIjKlMnOpQrStUvWxY", # incoming mail token + "glagent-" + "d" * 50, # agent (KAS) token + "glsoat-" + "AbCdEfGhIjKlMnOpQrSt", # service-account token + "glffct-" + "AbCdEfGhIjKlMnOpQrSt", # feature-flags client token + "glwt-" + "AbCdEfGhIjKlMnOpQrSt", # workspace token + "GR1348941" + "E" * 20, # legacy runner registration + ] + for token in tokens: + result = redact_sensitive_text(f"leaked {token} in output") + secret_body = token.split("-", 1)[-1] if "-" in token else token[9:] + assert secret_body not in result, f"{token!r} survived redaction: {result!r}" + + def test_gitlab_prefix_requires_word_boundary_and_length(self): + """Prose and embedded identifiers must not false-positive.""" + for benign in [ + "the glossary explains gitlab tokens", # no prefix at all + "glpat-short", # suffix under 10 chars + "myglpat-AbCdEfGhIjKlMnOpQrSt", # embedded — lookbehind blocks + ]: + assert redact_sensitive_text(benign) == benign + def test_slack_token(self): token = "xoxb-" + "0" * 12 + "-" + "a" * 14 result = redact_sensitive_text(token) diff --git a/tests/tools/test_skills_guard.py b/tests/tools/test_skills_guard.py index f740bbd3a8..b7f08256d7 100644 --- a/tests/tools/test_skills_guard.py +++ b/tests/tools/test_skills_guard.py @@ -169,6 +169,15 @@ class TestScanFile: assert findings == [] + def test_detect_gitlab_pat(self, tmp_path): + f = tmp_path / "leak.md" + # Concatenated so no contiguous token literal exists in this file + # (GitHub push protection blocks GitLab-PAT-shaped literals). + fake_token = "glpat-" + "Zx9AbCdEfGhIjKlMnOpQ" + f.write_text(f"Use {fake_token} to authenticate.\n") + findings = scan_file(f, "leak.md") + assert any(fi.pattern_id == "gitlab_token_leaked" for fi in findings) + def test_detect_markdown_injection(self, tmp_path): f = tmp_path / "bad.md" f.write_text( diff --git a/tools/skills_guard.py b/tools/skills_guard.py index 47f200ab88..2ba266e4fd 100644 --- a/tools/skills_guard.py +++ b/tools/skills_guard.py @@ -487,6 +487,9 @@ THREAT_PATTERNS = [ (r'AKIA[0-9A-Z]{16}', "aws_access_key_leaked", "critical", "credential_exposure", "AWS access key ID in skill content"), + (r'glpat-[A-Za-z0-9_\-]{20,}', + "gitlab_token_leaked", "critical", "credential_exposure", + "GitLab personal access token in skill content"), # ── Additional prompt injection: jailbreak patterns ── (r'\bDAN\s+mode\b|Do\s+Anything\s+Now',