From f7a861379f5ea06fdbf1e561c666958fc606602b Mon Sep 17 00:00:00 2001
From: Hukla <129692708+huklaa@users.noreply.github.com>
Date: Wed, 23 Sep 2026 17:35:38 +0300
Subject: [PATCH] fix(desktop): decode fetched link metadata by charset
---
apps/desktop/electron/link-title-curl.test.ts | 38 +++++++++++
apps/desktop/electron/link-title-curl.ts | 17 +++++
apps/desktop/electron/main.ts | 37 ++++-------
.../desktop/electron/web-text-decoder.test.ts | 64 +++++++++++++++++++
apps/desktop/electron/web-text-decoder.ts | 51 +++++++++++++++
5 files changed, 184 insertions(+), 23 deletions(-)
create mode 100644 apps/desktop/electron/link-title-curl.test.ts
create mode 100644 apps/desktop/electron/link-title-curl.ts
create mode 100644 apps/desktop/electron/web-text-decoder.test.ts
create mode 100644 apps/desktop/electron/web-text-decoder.ts
diff --git a/apps/desktop/electron/link-title-curl.test.ts b/apps/desktop/electron/link-title-curl.test.ts
new file mode 100644
index 0000000000..7657fc072b
--- /dev/null
+++ b/apps/desktop/electron/link-title-curl.test.ts
@@ -0,0 +1,38 @@
+import assert from 'node:assert/strict'
+
+import { describe, test } from 'vitest'
+
+import { parseCurlTitleResponse } from './link-title-curl'
+
+const BIG5_TITLE = Buffer.from([
+ ...Buffer.from('
'),
+ 0xb4,
+ 0xa3,
+ 0xa5,
+ 0xdc,
+ 0xab,
+ 0x48,
+ 0xae,
+ 0xa7,
+ ...Buffer.from('')
+])
+
+const TRAILER = Buffer.from(
+ '\nhermes-content-type:text/html; charset=big5\nhermes-url-effective:https://example.test/final'
+)
+
+describe('parseCurlTitleResponse', () => {
+ test('decodes a legacy page from curl content-type metadata', () => {
+ assert.deepEqual(parseCurlTitleResponse(Buffer.concat([BIG5_TITLE, TRAILER]), Buffer.alloc(0)), {
+ effectiveUrl: 'https://example.test/final',
+ html: '提示信息'
+ })
+ })
+
+ test('reads the trailer from the retained tail after the body budget is exhausted', () => {
+ assert.deepEqual(parseCurlTitleResponse(BIG5_TITLE, TRAILER), {
+ effectiveUrl: 'https://example.test/final',
+ html: '提示信息'
+ })
+ })
+})
diff --git a/apps/desktop/electron/link-title-curl.ts b/apps/desktop/electron/link-title-curl.ts
new file mode 100644
index 0000000000..4512fcf077
--- /dev/null
+++ b/apps/desktop/electron/link-title-curl.ts
@@ -0,0 +1,17 @@
+import { decodeWebText } from './web-text-decoder'
+
+const CONTENT_TYPE_MARK = 'hermes-content-type:'
+const URL_EFFECTIVE_MARK = 'hermes-url-effective:'
+
+export const CURL_TITLE_WRITE_OUT = `\n${CONTENT_TYPE_MARK}%{content_type}\n${URL_EFFECTIVE_MARK}%{url_effective}`
+
+export function parseCurlTitleResponse(bodyWithTrailer: Buffer, tail: Buffer): { effectiveUrl: string; html: string } {
+ const contentTypeMarker = Buffer.from(`\n${CONTENT_TYPE_MARK}`)
+ const at = bodyWithTrailer.lastIndexOf(contentTypeMarker)
+ const trailer = (at >= 0 ? bodyWithTrailer.subarray(at) : tail).toString('utf8')
+ const contentType = trailer.match(new RegExp(`(?:^|\\n)${CONTENT_TYPE_MARK}([^\\n]*)`))?.[1]?.trim() ?? ''
+ const effectiveUrl = trailer.match(new RegExp(`(?:^|\\n)${URL_EFFECTIVE_MARK}([^\\n]*)`))?.[1]?.trim() ?? ''
+ const body = at >= 0 ? bodyWithTrailer.subarray(0, at) : bodyWithTrailer
+
+ return { effectiveUrl, html: decodeWebText(body, contentType) }
+}
diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts
index 79d7135fef..070ecb4737 100644
--- a/apps/desktop/electron/main.ts
+++ b/apps/desktop/electron/main.ts
@@ -294,6 +294,7 @@ import { resolveHudWindowing } from './hud-windowing'
import { INSTALL_STAMP, installShape } from './install-stamp'
import type { InstallStamp } from './install-stamp'
import { createIntroRevealWindowController } from './intro-reveal-window'
+import { CURL_TITLE_WRITE_OUT, parseCurlTitleResponse } from './link-title-curl'
import { isAuthWall, resolveLinkTitle } from './link-title-wall'
import { createLinkTitleWindow, guardLinkTitleSession, readLinkTitleWindowTitle } from './link-title-window'
import { CHROMIUM_LOG_FILENAME, enableLinuxCrashDiagnostics, linuxCrashDiagnostics } from './linux-crash-diagnostics'
@@ -523,6 +524,7 @@ import { createStoreStrategy } from './updater/store-client'
import { isHermesOwnedVenvDaemon } from './venv-holder-select'
import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace'
import { createWakeIndicatorWindowController } from './wake-indicator-window'
+import { decodeWebText } from './web-text-decoder'
import { windowAcceleratorAction } from './window-accelerator'
import { enumerateWindowsFrontToBack, enumerationFailed, readWindowBelow } from './window-below'
import { bindWindowChromeEvents } from './window-chrome-events'
@@ -5172,23 +5174,8 @@ function parseHtmlTitle(html) {
return raw ? decodeHtmlEntities(raw).replace(/\s+/g, ' ').trim() : ''
}
-// `--write-out` trailer: `\n` after the body.
-const URL_EFFECTIVE_MARK = 'hermes-url-effective:'
const URL_EFFECTIVE_TAIL_BYTES = 4096
-function splitUrlEffective(stdout: string): { effectiveUrl: string; html: string } {
- const at = stdout.lastIndexOf(`\n${URL_EFFECTIVE_MARK}`)
-
- if (at < 0) {
- return { effectiveUrl: '', html: stdout }
- }
-
- return {
- effectiveUrl: stdout.slice(at + 1 + URL_EFFECTIVE_MARK.length).trim(),
- html: stdout.slice(0, at)
- }
-}
-
function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; title: string }> {
return new Promise(resolve => {
const url = String(rawUrl || '').trim()
@@ -5219,7 +5206,7 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti
// Arrival URL after redirects, on its own line after the body: a sign-in
// wall is proven from where curl landed even when the page has no markup id.
'--write-out',
- `\n${URL_EFFECTIVE_MARK}%{url_effective}`,
+ CURL_TITLE_WRITE_OUT,
url
]
@@ -5250,12 +5237,10 @@ function fetchHtmlTitleWithCurl(rawUrl: string): Promise<{ authWall: boolean; ti
return resolve({ authWall: false, title: '' })
}
- const body = Buffer.concat(chunks)
-
- // The trailer is inside `body` unless the budget cut it off; then it is in `tail`.
- const { effectiveUrl, html } = splitUrlEffective(
- (bytes >= TITLE_BYTE_BUDGET ? Buffer.concat([body, tail]) : body).toString('utf8')
- )
+ // The trailer is inside `bodyWithTrailer` unless the budget cut it off;
+ // then it is still present in the separately retained tail.
+ const bodyWithTrailer = Buffer.concat(chunks)
+ const { effectiveUrl, html } = parseCurlTitleResponse(bodyWithTrailer, tail)
const title = parseHtmlTitle(html)
@@ -5524,7 +5509,13 @@ const faviconIo: FaviconIo = {
fetchText: async url => {
const response = await faviconFetch(url, 'text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5')
- return response.ok ? (await response.text()).slice(0, TITLE_BYTE_BUDGET * 2) : ''
+ if (!response.ok) {
+ return ''
+ }
+
+ const bytes = new Uint8Array(await response.arrayBuffer()).subarray(0, TITLE_BYTE_BUDGET * 2)
+
+ return decodeWebText(bytes, response.headers.get('content-type') ?? '')
}
}
diff --git a/apps/desktop/electron/web-text-decoder.test.ts b/apps/desktop/electron/web-text-decoder.test.ts
new file mode 100644
index 0000000000..07cc185810
--- /dev/null
+++ b/apps/desktop/electron/web-text-decoder.test.ts
@@ -0,0 +1,64 @@
+import assert from 'node:assert/strict'
+
+import { describe, test } from 'vitest'
+
+import { decodeWebText } from './web-text-decoder'
+
+describe('decodeWebText', () => {
+ test('honours a quoted Big5 charset in the response header', () => {
+ const bytes = new Uint8Array([
+ ...Buffer.from(''),
+ 0xb4,
+ 0xa3,
+ 0xa5,
+ 0xdc,
+ 0xab,
+ 0x48,
+ 0xae,
+ 0xa7,
+ ...Buffer.from('')
+ ])
+
+ assert.equal(decodeWebText(bytes, 'text/html; charset="big5"'), '提示信息')
+ })
+
+ test('sniffs a Shift-JIS meta charset before decoding the document', () => {
+ const bytes = new Uint8Array([
+ ...Buffer.from(''),
+ 0x93,
+ 0xfa,
+ 0x96,
+ 0x7b,
+ 0x8c,
+ 0xea,
+ ...Buffer.from('')
+ ])
+
+ assert.equal(decodeWebText(bytes), '日本語')
+ })
+
+ test('sniffs charset from an http-equiv content attribute', () => {
+ const bytes = new Uint8Array([
+ ...Buffer.from(''),
+ 0xd6,
+ 0xd0,
+ 0xce,
+ 0xc4,
+ ...Buffer.from('')
+ ])
+
+ assert.ok(decodeWebText(bytes).includes('中文'))
+ })
+
+ test('falls back to UTF-8 when a server declares an unknown charset', () => {
+ const bytes = new TextEncoder().encode('Résumé')
+
+ assert.equal(decodeWebText(bytes, 'text/html; charset=not-a-real-encoding'), 'Résumé')
+ })
+
+ test('the HTTP charset takes precedence over a conflicting meta declaration', () => {
+ const bytes = new Uint8Array([...Buffer.from(''), 0xd6, 0xd0, 0xce, 0xc4])
+
+ assert.equal(decodeWebText(bytes, 'text/html; charset=gbk'), '中文')
+ })
+})
diff --git a/apps/desktop/electron/web-text-decoder.ts b/apps/desktop/electron/web-text-decoder.ts
new file mode 100644
index 0000000000..5545622661
--- /dev/null
+++ b/apps/desktop/electron/web-text-decoder.ts
@@ -0,0 +1,51 @@
+const META_SCAN_BYTES = 8192
+
+function charsetFromContentType(contentType: string): string {
+ const match = contentType.match(/(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i)
+
+ return (match?.slice(1).find(Boolean) ?? '').trim().slice(0, 64)
+}
+
+function charsetFromHtml(bytes: Uint8Array): string {
+ // Charset declarations are ASCII even when the document is not. Decode a
+ // small prefix as a single-byte encoding so invalid UTF-8 cannot erase the
+ // declaration before we know which decoder to use.
+ const head = new TextDecoder('windows-1252').decode(bytes.subarray(0, META_SCAN_BYTES))
+
+ for (const tag of head.match(/]*>/gi) ?? []) {
+ const direct = tag.match(/\bcharset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^\s"'/>;]+))/i)
+ const content = tag.match(/\bcontent\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+))/i)
+
+ const nested = (content?.slice(1).find(Boolean) ?? '').match(
+ /(?:^|;)\s*charset\s*=\s*(?:"([^"]+)"|'([^']+)'|([^;\s]+))/i
+ )
+
+ const label = direct?.slice(1).find(Boolean) ?? nested?.slice(1).find(Boolean)
+
+ if (label) {
+ return label.trim().slice(0, 64)
+ }
+ }
+
+ return ''
+}
+
+/** Decode fetched page/manifest bytes the same way a browser would choose a
+ * character encoding: HTTP header, then an HTML meta declaration, then UTF-8. */
+export function decodeWebText(bytes: Uint8Array, contentType = ''): string {
+ const labels = [charsetFromContentType(contentType), charsetFromHtml(bytes), 'utf-8']
+
+ for (const [index, label] of labels.entries()) {
+ if (!label || labels.indexOf(label) !== index) {
+ continue
+ }
+
+ try {
+ return new TextDecoder(label).decode(bytes)
+ } catch {
+ // Unknown/malformed server labels fall through to the next source.
+ }
+ }
+
+ return new TextDecoder().decode(bytes)
+}