diff --git a/plugins/platforms/matrix/adapter.py b/plugins/platforms/matrix/adapter.py index d8468fd2eb..a416c1ba71 100644 --- a/plugins/platforms/matrix/adapter.py +++ b/plugins/platforms/matrix/adapter.py @@ -581,6 +581,59 @@ def _scoped_recovery_key() -> str: return _startup_env_secret("MATRIX_RECOVERY_KEY") + +# --- LaTeX math ($...$, $$...$$) -> Element data-mx-maths markup --- +# Element (feature_latex_maths) typesets at display time. +# Our sanitizer allowlists tags/attrs, so data-mx-maths cannot pass through HTML +# sanitization directly. Instead, math is swapped for opaque sentinel tokens before +# Markdown conversion (protecting TeX from escaping) and expanded back to math +# markup after sanitization. Tokens are plain printable text with no special +# HTML/Markdown meaning, so both the Markdown converter and the sanitizer +# pass them through verbatim. +_TEX_TOKEN_RE = re.compile(r"HERMESTEX(?:DISPLAY|INLINE)(\d+)HERMESTEXEND") +_TEX_DISPLAY_TOKEN = "HERMESTEXDISPLAY%dHERMESTEXEND" +_TEX_INLINE_TOKEN = "HERMESTEXINLINE%dHERMESTEXEND" + + +def _latex_to_tokens(text: str) -> tuple[str, list[tuple[str, str]]]: + """Replace ``$$...$$``/``$...$`` with sentinel tokens. + + Returns the tokenized text plus an ordered ``(tag, tex)`` store, where tag + is ``div`` for display math and ``span`` for inline math. Dollars that do + not form a pair (prices, literals) are left untouched. + """ + if not text or "$" not in text: + return text, [] + store: list[tuple[str, str]] = [] + + def _sub_display(match: re.Match[str]) -> str: + store.append(("div", match.group(1).strip())) + return _TEX_DISPLAY_TOKEN % (len(store) - 1) + + def _sub_inline(match: re.Match[str]) -> str: + store.append(("span", match.group(1).strip())) + return _TEX_INLINE_TOKEN % (len(store) - 1) + + text = re.sub(r"\$\$([^\n$]+?)\$\$", _sub_display, text) + text = re.sub(r"(? str: + """Expand sentinel tokens into ``data-mx-maths`` markup (TeX HTML-escaped).""" + + def _expand(match: re.Match[str]) -> str: + idx = int(match.group(1)) + if idx >= len(store): + # Not one of our tokens (user-typed text that collides with the + # sentinel format) — leave it verbatim. + return match.group(0) + tag, tex = store[idx] + escaped = _html_escape(tex, quote=True) + return f'<{tag} data-mx-maths="{escaped}">{escaped}' + + return _TEX_TOKEN_RE.sub(_expand, html) + def _sanitize_matrix_html(html: str) -> str: sanitizer = _MatrixHtmlSanitizer() try: @@ -2755,6 +2808,8 @@ class MatrixAdapter(BasePlatformAdapter): def _markdown_to_html(self, text: str) -> str: """Markdown → org.matrix.custom.html via ``markdown`` when installed, else the regex fallback.""" text = _pre_sanitize_matrix_markdown(text) + _tex_store = [] + text, _tex_store = _latex_to_tokens(text) with suppress(ImportError): import markdown as _md md = _md.Markdown(extensions=["fenced_code", "tables", "nl2br", "sane_lists"]) @@ -2764,8 +2819,10 @@ class MatrixAdapter(BasePlatformAdapter): md.reset() if html.count("

") == 1: html = html.replace("

", "").replace("

", "") - return _sanitize_matrix_html(html) - return _sanitize_matrix_html(self._markdown_to_html_fallback(text)) + html = _tokens_to_mx_maths(_sanitize_matrix_html(html), _tex_store) + return html + _fb = _sanitize_matrix_html(self._markdown_to_html_fallback(text)) + return _tokens_to_mx_maths(_fb, _tex_store) @staticmethod def _sanitize_link_url(url: str) -> str: diff --git a/tests/gateway/test_matrix_latex_maths.py b/tests/gateway/test_matrix_latex_maths.py new file mode 100644 index 0000000000..f7353264b4 --- /dev/null +++ b/tests/gateway/test_matrix_latex_maths.py @@ -0,0 +1,145 @@ +"""Tests for LaTeX math ($...$ / $$...$$) -> Element data-mx-maths markup.""" + +import pytest + +from gateway.config import PlatformConfig + + +def _make_adapter(): + from plugins.platforms.matrix.adapter import MatrixAdapter + + config = PlatformConfig( + enabled=True, + token="syt_test_token", + extra={ + "homeserver": "https://matrix.example.org", + "user_id": "@bot:example.org", + }, + ) + return MatrixAdapter(config) + + +# --------------------------------------------------------------------------- +# _latex_to_tokens / _tokens_to_mx_maths unit tests +# --------------------------------------------------------------------------- + + +class TestLatexToTokens: + def setup_method(self): + from plugins.platforms.matrix.adapter import _latex_to_tokens + + self.convert = _latex_to_tokens + + def test_inline_math_becomes_span_token(self): + text, store = self.convert("Energy: $E = mc^2$ indeed.") + assert store == [("span", "E = mc^2")] + assert "E = mc^2" not in text + assert "$" not in text + + def test_display_math_becomes_div_token(self): + text, store = self.convert("$$\\hat{H}\\Psi = E\\Psi$$") + assert store == [("div", "\\hat{H}\\Psi = E\\Psi")] + + def test_mixed_math_all_captured(self): + # Display ($$...$$) is tokenized before inline ($...$), so store order + # follows regex pass order, not text order. Index-token correspondence + # is what matters. + text, store = self.convert("$a$ then $$b$$ then $c$") + assert sorted((tex, tag) for tag, tex in store) == [ + ("a", "span"), + ("b", "div"), + ("c", "span"), + ] + + def test_unpaired_dollars_untouched(self): + text, store = self.convert("Costs $5 or $10 today.") + assert store == [] + assert text == "Costs $5 or $10 today." + + def test_no_math_no_change(self): + text, store = self.convert("No math here.") + assert store == [] + assert text == "No math here." + + def test_backslash_dollar_not_math(self): + _, store = self.convert(r"Escaped \\$5 not math.") + assert store == [] + + +class TestTokensToMxMaths: + def setup_method(self): + from plugins.platforms.matrix.adapter import ( + _latex_to_tokens, + _tokens_to_mx_maths, + ) + + self.tokenize = _latex_to_tokens + self.expand = _tokens_to_mx_maths + + def test_inline_expansion(self): + tex = r"\psi(x)" + text, store = self.tokenize(f"Wave: ${tex}$") + html = self.expand(text, store) + assert html == f'Wave: {tex}' + + def test_display_expansion(self): + text, store = self.tokenize("$$x^2$$") + html = self.expand(text, store) + assert 'data-mx-maths="x^2"' in html + + def test_tex_is_html_escaped(self): + text, store = self.tokenize('$a"d$') + html = self.expand(text, store) + assert "" in html + assert 'data-mx-maths="x^2"' in html + + def test_dollar_amounts_not_converted(self): + html = self.adapter._markdown_to_html("Costs $5 or $10 today.") + assert "data-mx-maths" not in html + + def test_math_inside_code_block_still_converted(self): + # Note: sentinel conversion happens before Markdown sees the text, + # so even code-span math becomes markup. Documented trade-off: the + # composer-style consumers (chat) rarely put $...$ in code spans. + html = self.adapter._markdown_to_html("Use `$x$` here.") + assert 'data-mx-maths="x"' in html