diff --git a/plugins/teams_pipeline/meetings.py b/plugins/teams_pipeline/meetings.py index f0772c423d..994894ac64 100644 --- a/plugins/teams_pipeline/meetings.py +++ b/plugins/teams_pipeline/meetings.py @@ -2,6 +2,8 @@ from __future__ import annotations +import base64 +import binascii import re import tempfile from pathlib import Path @@ -88,7 +90,32 @@ def looks_like_transcript_id(value: str, *, odata_type: str | None = None) -> bo if "calltranscript" in str(odata_type or "").lower(): return True - return "transcript" in str(value or "").lower() + text = str(value or "") + if "transcript" in text.lower(): + return True + return "transcript" in _decoded_id_hint(text) + + +def _decoded_id_hint(value: str) -> str: + """Best-effort base64 decode of a Graph id for artifact-marker sniffing. + + Graph transcript ids are base64url blobs whose *decoded* payload carries a + ``...-TranscriptV2`` suffix while the encoded form contains no readable + marker (this is exactly the id shape from getAllTranscripts + ``resourceData.id``). Returns lowercase decoded text, or "" when the value + does not decode. + """ + + stripped = value.strip() + if len(stripped) < 16: + return "" + padded = stripped + "=" * (-len(stripped) % 4) + for decoder in (base64.urlsafe_b64decode, base64.b64decode): + try: + return decoder(padded).decode("utf-8", "ignore").lower() + except (binascii.Error, ValueError): + continue + return "" def _meeting_path(meeting_ref: TeamsMeetingRef | str) -> str: diff --git a/tests/plugins/test_teams_pipeline_plugin.py b/tests/plugins/test_teams_pipeline_plugin.py index 6723bd66e9..42b298ccda 100644 --- a/tests/plugins/test_teams_pipeline_plugin.py +++ b/tests/plugins/test_teams_pipeline_plugin.py @@ -461,6 +461,24 @@ def test_looks_like_transcript_id_detects_graph_call_transcripts(): assert not looks_like_transcript_id("MSo-meeting") +def test_looks_like_transcript_id_detects_base64_encoded_marker(): + # Real-world getAllTranscripts resourceData.id shape: base64url blob whose + # DECODED payload ends in "-TranscriptV2" while the encoded form has no + # readable marker (from a field report, Aug 2026). + import base64 + + encoded = base64.urlsafe_b64encode( + b"...meeting_ABC@thread.v2#55d6e479-2db7-4297-8590-98f6011a9613-1787004872-TranscriptV2" + ).decode().rstrip("=") + "=" + assert "transcript" not in encoded.lower() + assert looks_like_transcript_id(encoded) + # Meeting ids (base64 of "0#...#0**19:meeting_...@thread.v2") must NOT match. + meeting_encoded = base64.b64encode( + b"0#af8b854e-e592-4d5a-b04c-6e63a3b502e7#0**19:meeting_ABC@thread.v2" + ).decode() + assert not looks_like_transcript_id(meeting_encoded) + + def test_create_job_from_get_all_transcripts_uses_meeting_id_not_transcript_id(tmp_path): store = TeamsPipelineStore(tmp_path / "teams-store.json") pipeline = TeamsMeetingPipeline(graph_client=FakeGraphClient(), store=store)