fix(desktop): decode fetched link metadata by charset

This commit is contained in:
Hukla
2026-09-23 17:35:38 +03:00
committed by brooklyn!
parent f1247d2e01
commit f7a861379f
5 changed files with 184 additions and 23 deletions

View File

@@ -0,0 +1,38 @@
import assert from 'node:assert/strict'
import { describe, test } from 'vitest'
import { parseCurlTitleResponse } from './link-title-curl'
const BIG5_TITLE = Buffer.from([
...Buffer.from('<title>'),
0xb4,
0xa3,
0xa5,
0xdc,
0xab,
0x48,
0xae,
0xa7,
...Buffer.from('</title>')
])
const TRAILER = Buffer.from(
'\nhermes-content-type:text/html; charset=big5\nhermes-url-effective:https://example.test/final'
)
describe('parseCurlTitleResponse', () => {
test('decodes a legacy page from curl content-type metadata', () => {
assert.deepEqual(parseCurlTitleResponse(Buffer.concat([BIG5_TITLE, TRAILER]), Buffer.alloc(0)), {
effectiveUrl: 'https://example.test/final',
html: '<title>提示信息</title>'
})
})
test('reads the trailer from the retained tail after the body budget is exhausted', () => {
assert.deepEqual(parseCurlTitleResponse(BIG5_TITLE, TRAILER), {
effectiveUrl: 'https://example.test/final',
html: '<title>提示信息</title>'
})
})
})

View File

@@ -0,0 +1,17 @@
import { decodeWebText } from './web-text-decoder'
const CONTENT_TYPE_MARK = 'hermes-content-type:'
const URL_EFFECTIVE_MARK = 'hermes-url-effective:'
export const CURL_TITLE_WRITE_OUT = `\n${CONTENT_TYPE_MARK}%{content_type}\n${URL_EFFECTIVE_MARK}%{url_effective}`
export function parseCurlTitleResponse(bodyWithTrailer: Buffer, tail: Buffer): { effectiveUrl: string; html: string } {
const contentTypeMarker = Buffer.from(`\n${CONTENT_TYPE_MARK}`)
const at = bodyWithTrailer.lastIndexOf(contentTypeMarker)
const trailer = (at >= 0 ? bodyWithTrailer.subarray(at) : tail).toString('utf8')
const contentType = trailer.match(new RegExp(`(?:^|\\n)${CONTENT_TYPE_MARK}([^\\n]*)`))?.[1]?.trim() ?? ''
const effectiveUrl = trailer.match(new RegExp(`(?:^|\\n)${URL_EFFECTIVE_MARK}([^\\n]*)`))?.[1]?.trim() ?? ''
const body = at >= 0 ? bodyWithTrailer.subarray(0, at) : bodyWithTrailer
return { effectiveUrl, html: decodeWebText(body, contentType) }
}

View File

@@ -294,6 +294,7 @@ import { resolveHudWindowing } from './hud-windowing'
import { INSTALL_STAMP, installShape } from './install-stamp'
import type { InstallStamp } from './install-stamp'
import { createIntroRevealWindowController } from './intro-reveal-window'
import { CURL_TITLE_WRITE_OUT, parseCurlTitleResponse } from './link-title-curl'
import { isAuthWall, resolveLinkTitle } from './link-title-wall'
import { createLinkTitleWindow, guardLinkTitleSession, readLinkTitleWindowTitle } from './link-title-window'
import { CHROMIUM_LOG_FILENAME, enableLinuxCrashDiagnostics, linuxCrashDiagnostics } from './linux-crash-diagnostics'
@@ -523,6 +524,7 @@ import { createStoreStrategy } from './updater/store-client'
import { isHermesOwnedVenvDaemon } from './venv-holder-select'
import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace'
import { createWakeIndicatorWindowController } from './wake-indicator-window'
import { decodeWebText } from './web-text-decoder'
import { windowAcceleratorAction } from './window-accelerator'
import { enumerateWindowsFrontToBack, enumerationFailed, readWindowBelow } from './window-below'
import { bindWindowChromeEvents } from './window-chrome-events'
@@ -5172,23 +5174,8 @@ function parseHtmlTitle(html) {
return raw ? decodeHtmlEntities(raw).replace(/\s+/g, ' ').trim() : ''
}
// `--write-out` trailer: `\n<mark><url_effective>` after the body.
const URL_EFFECTIVE_MARK = 'hermes-url-effective:'
const URL_EFFECTIVE_TAIL_BYTES = 4096
function splitUrlEffective(stdout: string): { effectiveUrl: string; html: string } {
const at = stdout.lastIndexOf(`\n${URL_EFFECTIVE_MARK}`)
if (at < 0) {
return { effectiveUrl: '', html: stdout }
}
return {
effectiveUrl: stdout.slice(at + 1 + URL_EFFECTIVE_MARK.length).trim(),
html: stdout.slice(0, at)
}
}
function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; title: string }> {
return new Promise(resolve => {
const url = String(rawUrl || '').trim()
@@ -5219,7 +5206,7 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti
// Arrival URL after redirects, on its own line after the body: a sign-in
// wall is proven from where curl landed even when the page has no markup id.
'--write-out',
`\n${URL_EFFECTIVE_MARK}%{url_effective}`,
CURL_TITLE_WRITE_OUT,
url
]
@@ -5250,12 +5237,10 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti
return resolve({ authWall: false, title: '' })
}
const body = Buffer.concat(chunks)
// The trailer is inside `body` unless the budget cut it off; then it is in `tail`.
const { effectiveUrl, html } = splitUrlEffective(
(bytes >= TITLE_BYTE_BUDGET ? Buffer.concat([body, tail]) : body).toString('utf8')
)
// The trailer is inside `bodyWithTrailer` unless the budget cut it off;
// then it is still present in the separately retained tail.
const bodyWithTrailer = Buffer.concat(chunks)
const { effectiveUrl, html } = parseCurlTitleResponse(bodyWithTrailer, tail)
const title = parseHtmlTitle(html)
@@ -5524,7 +5509,13 @@ const faviconIo: FaviconIo = {
fetchText: async url => {
const response = await faviconFetch(url, 'text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5')
return response.ok ? (await response.text()).slice(0, TITLE_BYTE_BUDGET * 2) : ''
if (!response.ok) {
return ''
}
const bytes = new Uint8Array(await response.arrayBuffer()).subarray(0, TITLE_BYTE_BUDGET * 2)
return decodeWebText(bytes, response.headers.get('content-type') ?? '')
}
}

View File

@@ -0,0 +1,64 @@
import assert from 'node:assert/strict'
import { describe, test } from 'vitest'
import { decodeWebText } from './web-text-decoder'
describe('decodeWebText', () => {
test('honours a quoted Big5 charset in the response header', () => {
const bytes = new Uint8Array([
...Buffer.from('<title>'),
0xb4,
0xa3,
0xa5,
0xdc,
0xab,
0x48,
0xae,
0xa7,
...Buffer.from('</title>')
])
assert.equal(decodeWebText(bytes, 'text/html; charset="big5"'), '<title>提示信息</title>')
})
test('sniffs a Shift-JIS meta charset before decoding the document', () => {
const bytes = new Uint8Array([
...Buffer.from('<meta charset=shift_jis><title>'),
0x93,
0xfa,
0x96,
0x7b,
0x8c,
0xea,
...Buffer.from('</title>')
])
assert.equal(decodeWebText(bytes), '<meta charset=shift_jis><title>日本語</title>')
})
test('sniffs charset from an http-equiv content attribute', () => {
const bytes = new Uint8Array([
...Buffer.from('<meta http-equiv="Content-Type" content="text/html; charset=gbk"><title>'),
0xd6,
0xd0,
0xce,
0xc4,
...Buffer.from('</title>')
])
assert.ok(decodeWebText(bytes).includes('<title>中文</title>'))
})
test('falls back to UTF-8 when a server declares an unknown charset', () => {
const bytes = new TextEncoder().encode('<title>Résumé</title>')
assert.equal(decodeWebText(bytes, 'text/html; charset=not-a-real-encoding'), '<title>Résumé</title>')
})
test('the HTTP charset takes precedence over a conflicting meta declaration', () => {
const bytes = new Uint8Array([...Buffer.from('<meta charset=utf-8><title>'), 0xd6, 0xd0, 0xce, 0xc4])
assert.equal(decodeWebText(bytes, 'text/html; charset=gbk'), '<meta charset=utf-8><title>中文')
})
})

View File

@@ -0,0 +1,51 @@
const META_SCAN_BYTES = 8192
function charsetFromContentType(contentType: string): string {
const match = contentType.match(/(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i)
return (match?.slice(1).find(Boolean) ?? '').trim().slice(0, 64)
}
function charsetFromHtml(bytes: Uint8Array): string {
// Charset declarations are ASCII even when the document is not. Decode a
// small prefix as a single-byte encoding so invalid UTF-8 cannot erase the
// declaration before we know which decoder to use.
const head = new TextDecoder('windows-1252').decode(bytes.subarray(0, META_SCAN_BYTES))
for (const tag of head.match(/<meta\b[^>]*>/gi) ?? []) {
const direct = tag.match(/\bcharset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^\s"'/>;]+))/i)
const content = tag.match(/\bcontent\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+))/i)
const nested = (content?.slice(1).find(Boolean) ?? '').match(
/(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i
)
const label = direct?.slice(1).find(Boolean) ?? nested?.slice(1).find(Boolean)
if (label) {
return label.trim().slice(0, 64)
}
}
return ''
}
/** Decode fetched page/manifest bytes the same way a browser would choose a
* character encoding: HTTP header, then an HTML meta declaration, then UTF-8. */
export function decodeWebText(bytes: Uint8Array, contentType = ''): string {
const labels = [charsetFromContentType(contentType), charsetFromHtml(bytes), 'utf-8']
for (const [index, label] of labels.entries()) {
if (!label || labels.indexOf(label) !== index) {
continue
}
try {
return new TextDecoder(label).decode(bytes)
} catch {
// Unknown/malformed server labels fall through to the next source.
}
}
return new TextDecoder().decode(bytes)
}