feat(matrix): render LaTeX math via Element data-mx-maths markup

Element (feature_latex_maths) typesets <div|span data-mx-maths="TEX">
elements at display time, but the outbound HTML sanitizer allowlists tags
and attributes, so data-mx-maths markup sent by the gateway never reaches
Element intact - messages containing $...$ render as raw dollars.

Convert $...$ (inline) and $$...$$ (display) to opaque sentinel tokens
before Markdown conversion and expand them to data-mx-maths markup after
sanitization. Tokens are printable text with no HTML/Markdown meaning, so
neither the converter nor the sanitizer touches the TeX. Unpaired dollars
(prices, literals) are untouched, and adversarial text colliding with the
sentinel format passes through verbatim (index-checked expansion).

(cherry picked from commit eb73aafe0e08502c54e63dbd03e99796c3b7fee3)
This commit is contained in:
romanovzky
2026-09-09 09:56:57 +00:00
committed by Teknium
parent 1a500a43a7
commit 199f4d5b96
2 changed files with 204 additions and 2 deletions

View File

@@ -581,6 +581,59 @@ def _scoped_recovery_key() -> str:
return _startup_env_secret("MATRIX_RECOVERY_KEY")
# --- LaTeX math ($...$, $$...$$) -> Element data-mx-maths markup ---
# Element (feature_latex_maths) typesets <div|span data-mx-maths="TEX"> at display time.
# Our sanitizer allowlists tags/attrs, so data-mx-maths cannot pass through HTML
# sanitization directly. Instead, math is swapped for opaque sentinel tokens before
# Markdown conversion (protecting TeX from escaping) and expanded back to math
# markup after sanitization. Tokens are plain printable text with no special
# HTML/Markdown meaning, so both the Markdown converter and the sanitizer
# pass them through verbatim.
_TEX_TOKEN_RE = re.compile(r"HERMESTEX(?:DISPLAY|INLINE)(\d+)HERMESTEXEND")
_TEX_DISPLAY_TOKEN = "HERMESTEXDISPLAY%dHERMESTEXEND"
_TEX_INLINE_TOKEN = "HERMESTEXINLINE%dHERMESTEXEND"
def _latex_to_tokens(text: str) -> tuple[str, list[tuple[str, str]]]:
"""Replace ``$$...$$``/``$...$`` with sentinel tokens.
Returns the tokenized text plus an ordered ``(tag, tex)`` store, where tag
is ``div`` for display math and ``span`` for inline math. Dollars that do
not form a pair (prices, literals) are left untouched.
"""
if not text or "$" not in text:
return text, []
store: list[tuple[str, str]] = []
def _sub_display(match: re.Match[str]) -> str:
store.append(("div", match.group(1).strip()))
return _TEX_DISPLAY_TOKEN % (len(store) - 1)
def _sub_inline(match: re.Match[str]) -> str:
store.append(("span", match.group(1).strip()))
return _TEX_INLINE_TOKEN % (len(store) - 1)
text = re.sub(r"\$\$([^\n$]+?)\$\$", _sub_display, text)
text = re.sub(r"(?<![\\$\w])\$([^\n$]+?)\$(?!\w)", _sub_inline, text)
return text, store
def _tokens_to_mx_maths(html: str, store: list[tuple[str, str]]) -> str:
"""Expand sentinel tokens into ``data-mx-maths`` markup (TeX HTML-escaped)."""
def _expand(match: re.Match[str]) -> str:
idx = int(match.group(1))
if idx >= len(store):
# Not one of our tokens (user-typed text that collides with the
# sentinel format) — leave it verbatim.
return match.group(0)
tag, tex = store[idx]
escaped = _html_escape(tex, quote=True)
return f'<{tag} data-mx-maths="{escaped}">{escaped}</{tag}>'
return _TEX_TOKEN_RE.sub(_expand, html)
def _sanitize_matrix_html(html: str) -> str:
sanitizer = _MatrixHtmlSanitizer()
try:
@@ -2755,6 +2808,8 @@ class MatrixAdapter(BasePlatformAdapter):
def _markdown_to_html(self, text: str) -> str:
"""Markdown → org.matrix.custom.html via ``markdown`` when installed, else the regex fallback."""
text = _pre_sanitize_matrix_markdown(text)
_tex_store = []
text, _tex_store = _latex_to_tokens(text)
with suppress(ImportError):
import markdown as _md
md = _md.Markdown(extensions=["fenced_code", "tables", "nl2br", "sane_lists"])
@@ -2764,8 +2819,10 @@ class MatrixAdapter(BasePlatformAdapter):
md.reset()
if html.count("<p>") == 1:
html = html.replace("<p>", "").replace("</p>", "")
return _sanitize_matrix_html(html)
return _sanitize_matrix_html(self._markdown_to_html_fallback(text))
html = _tokens_to_mx_maths(_sanitize_matrix_html(html), _tex_store)
return html
_fb = _sanitize_matrix_html(self._markdown_to_html_fallback(text))
return _tokens_to_mx_maths(_fb, _tex_store)
@staticmethod
def _sanitize_link_url(url: str) -> str:

View File

@@ -0,0 +1,145 @@
"""Tests for LaTeX math ($...$ / $$...$$) -> Element data-mx-maths markup."""
import pytest
from gateway.config import PlatformConfig
def _make_adapter():
from plugins.platforms.matrix.adapter import MatrixAdapter
config = PlatformConfig(
enabled=True,
token="syt_test_token",
extra={
"homeserver": "https://matrix.example.org",
"user_id": "@bot:example.org",
},
)
return MatrixAdapter(config)
# ---------------------------------------------------------------------------
# _latex_to_tokens / _tokens_to_mx_maths unit tests
# ---------------------------------------------------------------------------
class TestLatexToTokens:
def setup_method(self):
from plugins.platforms.matrix.adapter import _latex_to_tokens
self.convert = _latex_to_tokens
def test_inline_math_becomes_span_token(self):
text, store = self.convert("Energy: $E = mc^2$ indeed.")
assert store == [("span", "E = mc^2")]
assert "E = mc^2" not in text
assert "$" not in text
def test_display_math_becomes_div_token(self):
text, store = self.convert("$$\\hat{H}\\Psi = E\\Psi$$")
assert store == [("div", "\\hat{H}\\Psi = E\\Psi")]
def test_mixed_math_all_captured(self):
# Display ($$...$$) is tokenized before inline ($...$), so store order
# follows regex pass order, not text order. Index-token correspondence
# is what matters.
text, store = self.convert("$a$ then $$b$$ then $c$")
assert sorted((tex, tag) for tag, tex in store) == [
("a", "span"),
("b", "div"),
("c", "span"),
]
def test_unpaired_dollars_untouched(self):
text, store = self.convert("Costs $5 or $10 today.")
assert store == []
assert text == "Costs $5 or $10 today."
def test_no_math_no_change(self):
text, store = self.convert("No math here.")
assert store == []
assert text == "No math here."
def test_backslash_dollar_not_math(self):
_, store = self.convert(r"Escaped \\$5 not math.")
assert store == []
class TestTokensToMxMaths:
def setup_method(self):
from plugins.platforms.matrix.adapter import (
_latex_to_tokens,
_tokens_to_mx_maths,
)
self.tokenize = _latex_to_tokens
self.expand = _tokens_to_mx_maths
def test_inline_expansion(self):
tex = r"\psi(x)"
text, store = self.tokenize(f"Wave: ${tex}$")
html = self.expand(text, store)
assert html == f'Wave: <span data-mx-maths="{tex}">{tex}</span>'
def test_display_expansion(self):
text, store = self.tokenize("$$x^2$$")
html = self.expand(text, store)
assert 'data-mx-maths="x^2"' in html
def test_tex_is_html_escaped(self):
text, store = self.tokenize('$a<b & c>"d$')
html = self.expand(text, store)
assert "<b &" not in html
assert "data-mx-maths=" in html
def test_no_token_left_behind(self):
text, store = self.tokenize("$a$ $$b$$")
html = self.expand(text, store)
assert "HERMESTEX" not in html
def test_user_text_colliding_with_sentinel_is_safe(self):
# Adversarial literal that matches the sentinel format but has no
# matching store entry must pass through unchanged, not raise.
html = self.expand("HERMESTEXDISPLAY99HERMESTEXEND", [])
assert html == "HERMESTEXDISPLAY99HERMESTEXEND"
def test_mismatched_store_is_safe(self):
# Expansion with an empty store leaves unknown tokens alone
html = self.expand("HERMESTEXDISPLAY99HERMESTEXEND", [])
assert "data-mx-maths" not in html
# ---------------------------------------------------------------------------
# Integration through _markdown_to_html (the outbound pipeline)
# ---------------------------------------------------------------------------
class TestMarkdownToHtmlLatex:
def setup_method(self):
self.adapter = _make_adapter()
def test_inline_math_reaches_formatted_html(self):
html = self.adapter._markdown_to_html("Wave: $\\psi(x) = e^{ikx}$ ok")
assert 'data-mx-maths="\\psi(x) = e^{ikx}"' in html
assert "HERMESTEX" not in html
def test_display_math_reaches_formatted_html(self):
html = self.adapter._markdown_to_html("$$\\hat{H}\\Psi = E\\Psi$$")
assert 'data-mx-maths="\\hat{H}\\Psi = E\\Psi"' in html
def test_markdown_formatting_still_applied(self):
html = self.adapter._markdown_to_html("**bold** with $x^2$")
assert "<strong>" in html
assert 'data-mx-maths="x^2"' in html
def test_dollar_amounts_not_converted(self):
html = self.adapter._markdown_to_html("Costs $5 or $10 today.")
assert "data-mx-maths" not in html
def test_math_inside_code_block_still_converted(self):
# Note: sentinel conversion happens before Markdown sees the text,
# so even code-span math becomes markup. Documented trade-off: the
# composer-style consumers (chat) rarely put $...$ in code spans.
html = self.adapter._markdown_to_html("Use `$x$` here.")
assert 'data-mx-maths="x"' in html