diff --git a/apps/desktop/electron/link-title-curl.test.ts b/apps/desktop/electron/link-title-curl.test.ts new file mode 100644 index 0000000000..7657fc072b --- /dev/null +++ b/apps/desktop/electron/link-title-curl.test.ts @@ -0,0 +1,38 @@ +import assert from 'node:assert/strict' + +import { describe, test } from 'vitest' + +import { parseCurlTitleResponse } from './link-title-curl' + +const BIG5_TITLE = Buffer.from([ + ...Buffer.from(''), + 0xb4, + 0xa3, + 0xa5, + 0xdc, + 0xab, + 0x48, + 0xae, + 0xa7, + ...Buffer.from('') +]) + +const TRAILER = Buffer.from( + '\nhermes-content-type:text/html; charset=big5\nhermes-url-effective:https://example.test/final' +) + +describe('parseCurlTitleResponse', () => { + test('decodes a legacy page from curl content-type metadata', () => { + assert.deepEqual(parseCurlTitleResponse(Buffer.concat([BIG5_TITLE, TRAILER]), Buffer.alloc(0)), { + effectiveUrl: 'https://example.test/final', + html: '提示信息' + }) + }) + + test('reads the trailer from the retained tail after the body budget is exhausted', () => { + assert.deepEqual(parseCurlTitleResponse(BIG5_TITLE, TRAILER), { + effectiveUrl: 'https://example.test/final', + html: '提示信息' + }) + }) +}) diff --git a/apps/desktop/electron/link-title-curl.ts b/apps/desktop/electron/link-title-curl.ts new file mode 100644 index 0000000000..4512fcf077 --- /dev/null +++ b/apps/desktop/electron/link-title-curl.ts @@ -0,0 +1,17 @@ +import { decodeWebText } from './web-text-decoder' + +const CONTENT_TYPE_MARK = 'hermes-content-type:' +const URL_EFFECTIVE_MARK = 'hermes-url-effective:' + +export const CURL_TITLE_WRITE_OUT = `\n${CONTENT_TYPE_MARK}%{content_type}\n${URL_EFFECTIVE_MARK}%{url_effective}` + +export function parseCurlTitleResponse(bodyWithTrailer: Buffer, tail: Buffer): { effectiveUrl: string; html: string } { + const contentTypeMarker = Buffer.from(`\n${CONTENT_TYPE_MARK}`) + const at = bodyWithTrailer.lastIndexOf(contentTypeMarker) + const trailer = (at >= 0 ? bodyWithTrailer.subarray(at) : tail).toString('utf8') + const contentType = trailer.match(new RegExp(`(?:^|\\n)${CONTENT_TYPE_MARK}([^\\n]*)`))?.[1]?.trim() ?? '' + const effectiveUrl = trailer.match(new RegExp(`(?:^|\\n)${URL_EFFECTIVE_MARK}([^\\n]*)`))?.[1]?.trim() ?? '' + const body = at >= 0 ? bodyWithTrailer.subarray(0, at) : bodyWithTrailer + + return { effectiveUrl, html: decodeWebText(body, contentType) } +} diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 79d7135fef..070ecb4737 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -294,6 +294,7 @@ import { resolveHudWindowing } from './hud-windowing' import { INSTALL_STAMP, installShape } from './install-stamp' import type { InstallStamp } from './install-stamp' import { createIntroRevealWindowController } from './intro-reveal-window' +import { CURL_TITLE_WRITE_OUT, parseCurlTitleResponse } from './link-title-curl' import { isAuthWall, resolveLinkTitle } from './link-title-wall' import { createLinkTitleWindow, guardLinkTitleSession, readLinkTitleWindowTitle } from './link-title-window' import { CHROMIUM_LOG_FILENAME, enableLinuxCrashDiagnostics, linuxCrashDiagnostics } from './linux-crash-diagnostics' @@ -523,6 +524,7 @@ import { createStoreStrategy } from './updater/store-client' import { isHermesOwnedVenvDaemon } from './venv-holder-select' import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace' import { createWakeIndicatorWindowController } from './wake-indicator-window' +import { decodeWebText } from './web-text-decoder' import { windowAcceleratorAction } from './window-accelerator' import { enumerateWindowsFrontToBack, enumerationFailed, readWindowBelow } from './window-below' import { bindWindowChromeEvents } from './window-chrome-events' @@ -5172,23 +5174,8 @@ function parseHtmlTitle(html) { return raw ? decodeHtmlEntities(raw).replace(/\s+/g, ' ').trim() : '' } -// `--write-out` trailer: `\n` after the body. -const URL_EFFECTIVE_MARK = 'hermes-url-effective:' const URL_EFFECTIVE_TAIL_BYTES = 4096 -function splitUrlEffective(stdout: string): { effectiveUrl: string; html: string } { - const at = stdout.lastIndexOf(`\n${URL_EFFECTIVE_MARK}`) - - if (at < 0) { - return { effectiveUrl: '', html: stdout } - } - - return { - effectiveUrl: stdout.slice(at + 1 + URL_EFFECTIVE_MARK.length).trim(), - html: stdout.slice(0, at) - } -} - function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; title: string }> { return new Promise(resolve => { const url = String(rawUrl || '').trim() @@ -5219,7 +5206,7 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti // Arrival URL after redirects, on its own line after the body: a sign-in // wall is proven from where curl landed even when the page has no markup id. '--write-out', - `\n${URL_EFFECTIVE_MARK}%{url_effective}`, + CURL_TITLE_WRITE_OUT, url ] @@ -5250,12 +5237,10 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti return resolve({ authWall: false, title: '' }) } - const body = Buffer.concat(chunks) - - // The trailer is inside `body` unless the budget cut it off; then it is in `tail`. - const { effectiveUrl, html } = splitUrlEffective( - (bytes >= TITLE_BYTE_BUDGET ? Buffer.concat([body, tail]) : body).toString('utf8') - ) + // The trailer is inside `bodyWithTrailer` unless the budget cut it off; + // then it is still present in the separately retained tail. + const bodyWithTrailer = Buffer.concat(chunks) + const { effectiveUrl, html } = parseCurlTitleResponse(bodyWithTrailer, tail) const title = parseHtmlTitle(html) @@ -5524,7 +5509,13 @@ const faviconIo: FaviconIo = { fetchText: async url => { const response = await faviconFetch(url, 'text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5') - return response.ok ? (await response.text()).slice(0, TITLE_BYTE_BUDGET * 2) : '' + if (!response.ok) { + return '' + } + + const bytes = new Uint8Array(await response.arrayBuffer()).subarray(0, TITLE_BYTE_BUDGET * 2) + + return decodeWebText(bytes, response.headers.get('content-type') ?? '') } } diff --git a/apps/desktop/electron/web-text-decoder.test.ts b/apps/desktop/electron/web-text-decoder.test.ts new file mode 100644 index 0000000000..07cc185810 --- /dev/null +++ b/apps/desktop/electron/web-text-decoder.test.ts @@ -0,0 +1,64 @@ +import assert from 'node:assert/strict' + +import { describe, test } from 'vitest' + +import { decodeWebText } from './web-text-decoder' + +describe('decodeWebText', () => { + test('honours a quoted Big5 charset in the response header', () => { + const bytes = new Uint8Array([ + ...Buffer.from(''), + 0xb4, + 0xa3, + 0xa5, + 0xdc, + 0xab, + 0x48, + 0xae, + 0xa7, + ...Buffer.from('') + ]) + + assert.equal(decodeWebText(bytes, 'text/html; charset="big5"'), '提示信息') + }) + + test('sniffs a Shift-JIS meta charset before decoding the document', () => { + const bytes = new Uint8Array([ + ...Buffer.from(''), + 0x93, + 0xfa, + 0x96, + 0x7b, + 0x8c, + 0xea, + ...Buffer.from('') + ]) + + assert.equal(decodeWebText(bytes), '日本語') + }) + + test('sniffs charset from an http-equiv content attribute', () => { + const bytes = new Uint8Array([ + ...Buffer.from(''), + 0xd6, + 0xd0, + 0xce, + 0xc4, + ...Buffer.from('') + ]) + + assert.ok(decodeWebText(bytes).includes('中文')) + }) + + test('falls back to UTF-8 when a server declares an unknown charset', () => { + const bytes = new TextEncoder().encode('Résumé') + + assert.equal(decodeWebText(bytes, 'text/html; charset=not-a-real-encoding'), 'Résumé') + }) + + test('the HTTP charset takes precedence over a conflicting meta declaration', () => { + const bytes = new Uint8Array([...Buffer.from(''), 0xd6, 0xd0, 0xce, 0xc4]) + + assert.equal(decodeWebText(bytes, 'text/html; charset=gbk'), '<meta charset=utf-8><title>中文') + }) +}) diff --git a/apps/desktop/electron/web-text-decoder.ts b/apps/desktop/electron/web-text-decoder.ts new file mode 100644 index 0000000000..5545622661 --- /dev/null +++ b/apps/desktop/electron/web-text-decoder.ts @@ -0,0 +1,51 @@ +const META_SCAN_BYTES = 8192 + +function charsetFromContentType(contentType: string): string { + const match = contentType.match(/(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i) + + return (match?.slice(1).find(Boolean) ?? '').trim().slice(0, 64) +} + +function charsetFromHtml(bytes: Uint8Array): string { + // Charset declarations are ASCII even when the document is not. Decode a + // small prefix as a single-byte encoding so invalid UTF-8 cannot erase the + // declaration before we know which decoder to use. + const head = new TextDecoder('windows-1252').decode(bytes.subarray(0, META_SCAN_BYTES)) + + for (const tag of head.match(/<meta\b[^>]*>/gi) ?? []) { + const direct = tag.match(/\bcharset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^\s"'/>;]+))/i) + const content = tag.match(/\bcontent\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+))/i) + + const nested = (content?.slice(1).find(Boolean) ?? '').match( + /(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i + ) + + const label = direct?.slice(1).find(Boolean) ?? nested?.slice(1).find(Boolean) + + if (label) { + return label.trim().slice(0, 64) + } + } + + return '' +} + +/** Decode fetched page/manifest bytes the same way a browser would choose a + * character encoding: HTTP header, then an HTML meta declaration, then UTF-8. */ +export function decodeWebText(bytes: Uint8Array, contentType = ''): string { + const labels = [charsetFromContentType(contentType), charsetFromHtml(bytes), 'utf-8'] + + for (const [index, label] of labels.entries()) { + if (!label || labels.indexOf(label) !== index) { + continue + } + + try { + return new TextDecoder(label).decode(bytes) + } catch { + // Unknown/malformed server labels fall through to the next source. + } + } + + return new TextDecoder().decode(bytes) +}