fix(desktop): decode fetched link metadata by charset
This commit is contained in:
38
apps/desktop/electron/link-title-curl.test.ts
Normal file
38
apps/desktop/electron/link-title-curl.test.ts
Normal file
@@ -0,0 +1,38 @@
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { describe, test } from 'vitest'
|
||||
|
||||
import { parseCurlTitleResponse } from './link-title-curl'
|
||||
|
||||
const BIG5_TITLE = Buffer.from([
|
||||
...Buffer.from('<title>'),
|
||||
0xb4,
|
||||
0xa3,
|
||||
0xa5,
|
||||
0xdc,
|
||||
0xab,
|
||||
0x48,
|
||||
0xae,
|
||||
0xa7,
|
||||
...Buffer.from('</title>')
|
||||
])
|
||||
|
||||
const TRAILER = Buffer.from(
|
||||
'\nhermes-content-type:text/html; charset=big5\nhermes-url-effective:https://example.test/final'
|
||||
)
|
||||
|
||||
describe('parseCurlTitleResponse', () => {
|
||||
test('decodes a legacy page from curl content-type metadata', () => {
|
||||
assert.deepEqual(parseCurlTitleResponse(Buffer.concat([BIG5_TITLE, TRAILER]), Buffer.alloc(0)), {
|
||||
effectiveUrl: 'https://example.test/final',
|
||||
html: '<title>提示信息</title>'
|
||||
})
|
||||
})
|
||||
|
||||
test('reads the trailer from the retained tail after the body budget is exhausted', () => {
|
||||
assert.deepEqual(parseCurlTitleResponse(BIG5_TITLE, TRAILER), {
|
||||
effectiveUrl: 'https://example.test/final',
|
||||
html: '<title>提示信息</title>'
|
||||
})
|
||||
})
|
||||
})
|
||||
17
apps/desktop/electron/link-title-curl.ts
Normal file
17
apps/desktop/electron/link-title-curl.ts
Normal file
@@ -0,0 +1,17 @@
|
||||
import { decodeWebText } from './web-text-decoder'
|
||||
|
||||
const CONTENT_TYPE_MARK = 'hermes-content-type:'
|
||||
const URL_EFFECTIVE_MARK = 'hermes-url-effective:'
|
||||
|
||||
export const CURL_TITLE_WRITE_OUT = `\n${CONTENT_TYPE_MARK}%{content_type}\n${URL_EFFECTIVE_MARK}%{url_effective}`
|
||||
|
||||
export function parseCurlTitleResponse(bodyWithTrailer: Buffer, tail: Buffer): { effectiveUrl: string; html: string } {
|
||||
const contentTypeMarker = Buffer.from(`\n${CONTENT_TYPE_MARK}`)
|
||||
const at = bodyWithTrailer.lastIndexOf(contentTypeMarker)
|
||||
const trailer = (at >= 0 ? bodyWithTrailer.subarray(at) : tail).toString('utf8')
|
||||
const contentType = trailer.match(new RegExp(`(?:^|\\n)${CONTENT_TYPE_MARK}([^\\n]*)`))?.[1]?.trim() ?? ''
|
||||
const effectiveUrl = trailer.match(new RegExp(`(?:^|\\n)${URL_EFFECTIVE_MARK}([^\\n]*)`))?.[1]?.trim() ?? ''
|
||||
const body = at >= 0 ? bodyWithTrailer.subarray(0, at) : bodyWithTrailer
|
||||
|
||||
return { effectiveUrl, html: decodeWebText(body, contentType) }
|
||||
}
|
||||
@@ -294,6 +294,7 @@ import { resolveHudWindowing } from './hud-windowing'
|
||||
import { INSTALL_STAMP, installShape } from './install-stamp'
|
||||
import type { InstallStamp } from './install-stamp'
|
||||
import { createIntroRevealWindowController } from './intro-reveal-window'
|
||||
import { CURL_TITLE_WRITE_OUT, parseCurlTitleResponse } from './link-title-curl'
|
||||
import { isAuthWall, resolveLinkTitle } from './link-title-wall'
|
||||
import { createLinkTitleWindow, guardLinkTitleSession, readLinkTitleWindowTitle } from './link-title-window'
|
||||
import { CHROMIUM_LOG_FILENAME, enableLinuxCrashDiagnostics, linuxCrashDiagnostics } from './linux-crash-diagnostics'
|
||||
@@ -523,6 +524,7 @@ import { createStoreStrategy } from './updater/store-client'
|
||||
import { isHermesOwnedVenvDaemon } from './venv-holder-select'
|
||||
import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace'
|
||||
import { createWakeIndicatorWindowController } from './wake-indicator-window'
|
||||
import { decodeWebText } from './web-text-decoder'
|
||||
import { windowAcceleratorAction } from './window-accelerator'
|
||||
import { enumerateWindowsFrontToBack, enumerationFailed, readWindowBelow } from './window-below'
|
||||
import { bindWindowChromeEvents } from './window-chrome-events'
|
||||
@@ -5172,23 +5174,8 @@ function parseHtmlTitle(html) {
|
||||
return raw ? decodeHtmlEntities(raw).replace(/\s+/g, ' ').trim() : ''
|
||||
}
|
||||
|
||||
// `--write-out` trailer: `\n<mark><url_effective>` after the body.
|
||||
const URL_EFFECTIVE_MARK = 'hermes-url-effective:'
|
||||
const URL_EFFECTIVE_TAIL_BYTES = 4096
|
||||
|
||||
function splitUrlEffective(stdout: string): { effectiveUrl: string; html: string } {
|
||||
const at = stdout.lastIndexOf(`\n${URL_EFFECTIVE_MARK}`)
|
||||
|
||||
if (at < 0) {
|
||||
return { effectiveUrl: '', html: stdout }
|
||||
}
|
||||
|
||||
return {
|
||||
effectiveUrl: stdout.slice(at + 1 + URL_EFFECTIVE_MARK.length).trim(),
|
||||
html: stdout.slice(0, at)
|
||||
}
|
||||
}
|
||||
|
||||
function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; title: string }> {
|
||||
return new Promise(resolve => {
|
||||
const url = String(rawUrl || '').trim()
|
||||
@@ -5219,7 +5206,7 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti
|
||||
// Arrival URL after redirects, on its own line after the body: a sign-in
|
||||
// wall is proven from where curl landed even when the page has no markup id.
|
||||
'--write-out',
|
||||
`\n${URL_EFFECTIVE_MARK}%{url_effective}`,
|
||||
CURL_TITLE_WRITE_OUT,
|
||||
url
|
||||
]
|
||||
|
||||
@@ -5250,12 +5237,10 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti
|
||||
return resolve({ authWall: false, title: '' })
|
||||
}
|
||||
|
||||
const body = Buffer.concat(chunks)
|
||||
|
||||
// The trailer is inside `body` unless the budget cut it off; then it is in `tail`.
|
||||
const { effectiveUrl, html } = splitUrlEffective(
|
||||
(bytes >= TITLE_BYTE_BUDGET ? Buffer.concat([body, tail]) : body).toString('utf8')
|
||||
)
|
||||
// The trailer is inside `bodyWithTrailer` unless the budget cut it off;
|
||||
// then it is still present in the separately retained tail.
|
||||
const bodyWithTrailer = Buffer.concat(chunks)
|
||||
const { effectiveUrl, html } = parseCurlTitleResponse(bodyWithTrailer, tail)
|
||||
|
||||
const title = parseHtmlTitle(html)
|
||||
|
||||
@@ -5524,7 +5509,13 @@ const faviconIo: FaviconIo = {
|
||||
fetchText: async url => {
|
||||
const response = await faviconFetch(url, 'text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5')
|
||||
|
||||
return response.ok ? (await response.text()).slice(0, TITLE_BYTE_BUDGET * 2) : ''
|
||||
if (!response.ok) {
|
||||
return ''
|
||||
}
|
||||
|
||||
const bytes = new Uint8Array(await response.arrayBuffer()).subarray(0, TITLE_BYTE_BUDGET * 2)
|
||||
|
||||
return decodeWebText(bytes, response.headers.get('content-type') ?? '')
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
64
apps/desktop/electron/web-text-decoder.test.ts
Normal file
64
apps/desktop/electron/web-text-decoder.test.ts
Normal file
@@ -0,0 +1,64 @@
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { describe, test } from 'vitest'
|
||||
|
||||
import { decodeWebText } from './web-text-decoder'
|
||||
|
||||
describe('decodeWebText', () => {
|
||||
test('honours a quoted Big5 charset in the response header', () => {
|
||||
const bytes = new Uint8Array([
|
||||
...Buffer.from('<title>'),
|
||||
0xb4,
|
||||
0xa3,
|
||||
0xa5,
|
||||
0xdc,
|
||||
0xab,
|
||||
0x48,
|
||||
0xae,
|
||||
0xa7,
|
||||
...Buffer.from('</title>')
|
||||
])
|
||||
|
||||
assert.equal(decodeWebText(bytes, 'text/html; charset="big5"'), '<title>提示信息</title>')
|
||||
})
|
||||
|
||||
test('sniffs a Shift-JIS meta charset before decoding the document', () => {
|
||||
const bytes = new Uint8Array([
|
||||
...Buffer.from('<meta charset=shift_jis><title>'),
|
||||
0x93,
|
||||
0xfa,
|
||||
0x96,
|
||||
0x7b,
|
||||
0x8c,
|
||||
0xea,
|
||||
...Buffer.from('</title>')
|
||||
])
|
||||
|
||||
assert.equal(decodeWebText(bytes), '<meta charset=shift_jis><title>日本語</title>')
|
||||
})
|
||||
|
||||
test('sniffs charset from an http-equiv content attribute', () => {
|
||||
const bytes = new Uint8Array([
|
||||
...Buffer.from('<meta http-equiv="Content-Type" content="text/html; charset=gbk"><title>'),
|
||||
0xd6,
|
||||
0xd0,
|
||||
0xce,
|
||||
0xc4,
|
||||
...Buffer.from('</title>')
|
||||
])
|
||||
|
||||
assert.ok(decodeWebText(bytes).includes('<title>中文</title>'))
|
||||
})
|
||||
|
||||
test('falls back to UTF-8 when a server declares an unknown charset', () => {
|
||||
const bytes = new TextEncoder().encode('<title>Résumé</title>')
|
||||
|
||||
assert.equal(decodeWebText(bytes, 'text/html; charset=not-a-real-encoding'), '<title>Résumé</title>')
|
||||
})
|
||||
|
||||
test('the HTTP charset takes precedence over a conflicting meta declaration', () => {
|
||||
const bytes = new Uint8Array([...Buffer.from('<meta charset=utf-8><title>'), 0xd6, 0xd0, 0xce, 0xc4])
|
||||
|
||||
assert.equal(decodeWebText(bytes, 'text/html; charset=gbk'), '<meta charset=utf-8><title>中文')
|
||||
})
|
||||
})
|
||||
51
apps/desktop/electron/web-text-decoder.ts
Normal file
51
apps/desktop/electron/web-text-decoder.ts
Normal file
@@ -0,0 +1,51 @@
|
||||
const META_SCAN_BYTES = 8192
|
||||
|
||||
function charsetFromContentType(contentType: string): string {
|
||||
const match = contentType.match(/(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i)
|
||||
|
||||
return (match?.slice(1).find(Boolean) ?? '').trim().slice(0, 64)
|
||||
}
|
||||
|
||||
function charsetFromHtml(bytes: Uint8Array): string {
|
||||
// Charset declarations are ASCII even when the document is not. Decode a
|
||||
// small prefix as a single-byte encoding so invalid UTF-8 cannot erase the
|
||||
// declaration before we know which decoder to use.
|
||||
const head = new TextDecoder('windows-1252').decode(bytes.subarray(0, META_SCAN_BYTES))
|
||||
|
||||
for (const tag of head.match(/<meta\b[^>]*>/gi) ?? []) {
|
||||
const direct = tag.match(/\bcharset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^\s"'/>;]+))/i)
|
||||
const content = tag.match(/\bcontent\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+))/i)
|
||||
|
||||
const nested = (content?.slice(1).find(Boolean) ?? '').match(
|
||||
/(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i
|
||||
)
|
||||
|
||||
const label = direct?.slice(1).find(Boolean) ?? nested?.slice(1).find(Boolean)
|
||||
|
||||
if (label) {
|
||||
return label.trim().slice(0, 64)
|
||||
}
|
||||
}
|
||||
|
||||
return ''
|
||||
}
|
||||
|
||||
/** Decode fetched page/manifest bytes the same way a browser would choose a
|
||||
* character encoding: HTTP header, then an HTML meta declaration, then UTF-8. */
|
||||
export function decodeWebText(bytes: Uint8Array, contentType = ''): string {
|
||||
const labels = [charsetFromContentType(contentType), charsetFromHtml(bytes), 'utf-8']
|
||||
|
||||
for (const [index, label] of labels.entries()) {
|
||||
if (!label || labels.indexOf(label) !== index) {
|
||||
continue
|
||||
}
|
||||
|
||||
try {
|
||||
return new TextDecoder(label).decode(bytes)
|
||||
} catch {
|
||||
// Unknown/malformed server labels fall through to the next source.
|
||||
}
|
||||
}
|
||||
|
||||
return new TextDecoder().decode(bytes)
|
||||
}
|
||||
Reference in New Issue
Block a user