Files
hermes-agent/apps/desktop/src/lib/markdown-html-depth.test.ts
Brooklyn Nicholson 1641512c94 fix(desktop): bound raw HTML nesting before it reaches rehype-raw
Streamdown parses assistant markdown with allowDangerousHtml, so every
`<tag>` run in a message goes to parse5 and then through
hast-util-from-parse5, which recurses once per level of unclosed
nesting. Past roughly 1,750 consecutive unclosed tags that overflows the
call stack and throws RangeError out of the middle of a React render.

Nemotron-3-ultra degenerates into exactly that: thousands of `<unk>`
tokens emitted as reasoning, every one of them an element parse5 opens
and never closes. The payload is persisted to the session, so the throw
comes back on every reload.

Clamp the depth of unclosed elements in the prose path and escape the
opening `<` past the cap, leaving the text visible as the literal
`<unk>` it always was. The bound is on depth, not size: 20,000 balanced
`<b>x</b>` pairs and 20,000 void `<br>` tags parse fine because neither
drives the tree deeper, so only unclosed elements are counted and normal
markup is returned by identity.
2026-08-13 13:36:54 -05:00

70 lines
2.5 KiB
TypeScript

import { describe, expect, it } from 'vitest'
import { clampHtmlNestingDepth } from './markdown-html-depth'
// The clamp exists to keep parse5 / hast-util-from-parse5 recursion off the
// call stack. Its contract is about DEPTH of unclosed elements, so the tests
// assert that relationship rather than any particular output string.
describe('clampHtmlNestingDepth', () => {
it('returns text without tags unchanged, by identity', () => {
const text = 'C:\\Windows\\System32 is 12 GB, and 5 < 7 is true'
expect(clampHtmlNestingDepth(text)).toBe(text)
})
it('leaves ordinary markup untouched, by identity', () => {
const text = '<div class="a"><p>A <b>bold</b> claim</p></div>'
expect(clampHtmlNestingDepth(text)).toBe(text)
})
it('leaves a wide but shallow tag run untouched — count is not the risk', () => {
const text = '<b>x</b>'.repeat(20_000)
expect(clampHtmlNestingDepth(text)).toBe(text)
})
it('leaves void elements untouched — they never add depth', () => {
const text = '<br>'.repeat(20_000)
expect(clampHtmlNestingDepth(text)).toBe(text)
})
it('leaves self-closing tags untouched — they never add depth', () => {
const text = '<custom-el />'.repeat(20_000)
expect(clampHtmlNestingDepth(text)).toBe(text)
})
it('escapes an unbounded run of unclosed unknown tags', () => {
const clamped = clampHtmlNestingDepth('Let' + '<unk>'.repeat(16_383))
// Nothing is dropped: the text is still all there, just literal. Only the
// opening `<` needs escaping — that alone stops parse5 opening an element.
expect(clamped).toContain('&lt;unk>')
expect(clamped.startsWith('Let<unk>')).toBe(true)
})
it('escapes an unbounded run of unclosed known tags', () => {
const clamped = clampHtmlNestingDepth('<div>'.repeat(5_000))
expect(clamped).toContain('&lt;div>')
})
it('keeps the prefix below the cap parseable and only escapes past it', () => {
const clamped = clampHtmlNestingDepth('<unk>'.repeat(5_000))
const parseable = clamped.slice(0, clamped.indexOf('&lt;'))
// Everything before the overflow point is untouched real markup...
expect(parseable).toBe('<unk>'.repeat(parseable.length / '<unk>'.length))
// ...and it is bounded, which is the entire point.
expect(parseable.length).toBeLessThan('<unk>'.repeat(5_000).length)
})
it('does not accumulate depth when tags close, however many there are', () => {
const text = '<div>x</div>'.repeat(10_000)
expect(clampHtmlNestingDepth(text)).toBe(text)
})
})