diff --git a/.github/workflows/install-e2e-windows-run.yml b/.github/workflows/install-e2e-windows-run.yml index b6bf850680..223ae9f6ab 100644 --- a/.github/workflows/install-e2e-windows-run.yml +++ b/.github/workflows/install-e2e-windows-run.yml @@ -137,9 +137,26 @@ jobs: run: powershell -NoProfile -ExecutionPolicy Bypass -File tests\install\windows-e2e.ps1 -Phase install -InstallMethod "${{ inputs.install-method }}" -Route "${{ inputs.update-method }}" -InstallRef "${{ inputs.install-ref }}" -SetupExeUrl ${{ inputs.setup-exe-url }} - name: Update ${{ inputs.install-ref }} -> HEAD (${{ inputs.update-method }}) + id: update shell: powershell run: powershell -NoProfile -ExecutionPolicy Bypass -File tests\install\windows-e2e.ps1 -Phase update -InstallMethod "${{ inputs.install-method }}" -Route "${{ inputs.update-method }}" -InstallRef "${{ inputs.install-ref }}" -SetupExeUrl ${{ inputs.setup-exe-url }} + - name: Stage known-failure receipt + if: steps.update.outputs.known_failure != '' + shell: pwsh + run: | + New-Item -ItemType Directory -Path gui-e2e-proof -Force | Out-Null + Copy-Item -LiteralPath (Join-Path $env:HERMES_E2E_WORKROOT 'known-failure.json') -Destination gui-e2e-proof/known-failure.json + + - name: Upload known-failure receipt + if: steps.update.outputs.known_failure != '' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: install-e2e-known-${{ steps.update.outputs.known_failure }}--${{ inputs.leg-id }} + path: gui-e2e-proof/known-failure.json + if-no-files-found: error + retention-days: 14 + - name: Stop screen recording if: always() uses: ./.github/actions/e2e-screen-record diff --git a/.github/workflows/install-e2e.yml b/.github/workflows/install-e2e.yml index 26154a4bef..84817d4dbb 100644 --- a/.github/workflows/install-e2e.yml +++ b/.github/workflows/install-e2e.yml @@ -57,6 +57,11 @@ on: required: false type: string default: '3' + install-ref: + description: 'Optional exact release tag for a focused reproduction; overrides tag-count.' + required: false + type: string + default: '' schedule: # Every 12 hours, off the hour to avoid the top-of-hour runner crunch. - cron: '20 7,19 * * *' @@ -102,10 +107,17 @@ jobs: # applies to each per-OS matrix separately; at 10 tags the largest # is windows at 180 (first over the cap at 15 tags = 270). TAG_COUNT: ${{ inputs.tag-count || 2 }} + INSTALL_REF: ${{ inputs.install-ref }} run: | set -euo pipefail [[ "$TAG_COUNT" =~ ^(10|[1-9])$ ]] || { echo "tag-count must be 1-10, got: $TAG_COUNT" >&2; exit 1; } - tags="$(scripts/sandbox/pick-release-tags.sh --count "$TAG_COUNT")" + if [ -n "$INSTALL_REF" ]; then + [[ "$INSTALL_REF" =~ ^v[0-9]+\.[0-9]+\.[0-9]+(\.[0-9]+)?$ ]] || { echo 'install-ref must be an exact release tag' >&2; exit 1; } + git rev-parse --verify "refs/tags/$INSTALL_REF^{commit}" >/dev/null + tags="$(jq -cn --arg ref "$INSTALL_REF" '[$ref]')" + else + tags="$(scripts/sandbox/pick-release-tags.sh --count "$TAG_COUNT")" + fi echo "Testing updates from: $tags" # Annotate each tag with what its own tree supports, so run # workflows can natively skip surfaces the starting version does @@ -241,7 +253,9 @@ jobs: steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 with: - sparse-checkout: scripts/sandbox/generate-e2e-matrix.mjs + sparse-checkout: | + scripts/sandbox/generate-e2e-matrix.mjs + tests/install/e2e-assets/known-failures.json sparse-checkout-cone-mode: false - env: GH_TOKEN: ${{ github.token }} @@ -256,7 +270,7 @@ jobs: --paginate --jq '.jobs[] | {name, conclusion}' > /tmp/e2e-jobs.ndjson gh api "repos/${{ github.repository }}/actions/runs/${{ github.run_id }}/artifacts?per_page=100" \ --paginate --jq '.artifacts[] | {name, id}' > /tmp/e2e-artifacts.ndjson - echo 'Legend: ✅ ran green · ❌ ran red · pre-desktop / TODO = why a leg skipped · 📼 opens the leg player (recording + synced logs)' + echo 'Legend: ✅ upgrade passed · known [n] = exact historical failure, see footnote · ❌ unexpected failure · pre-desktop / TODO = why a leg skipped · 📼 opens the leg player (recording + synced logs)' echo node scripts/sandbox/generate-e2e-matrix.mjs --format results \ --tags '${{ needs.pick-releases.outputs.tags }}' \ diff --git a/apps/desktop/e2e/onboarding-settings.spec.ts b/apps/desktop/e2e/onboarding-settings.spec.ts new file mode 100644 index 0000000000..6fe1253cdb --- /dev/null +++ b/apps/desktop/e2e/onboarding-settings.spec.ts @@ -0,0 +1,82 @@ +import { readFileSync, unlinkSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import path from 'node:path' + +import { buildAppEnv, createSandbox, launchDesktop, setupNoProvider } from './fixtures' +import { type ElectronApplication, expect, type Page, test } from './test' + +const { prepareWindowForInput } = createRequire(import.meta.url)( + '../../../tests/install/e2e-assets/window-input.cjs', +) as { prepareWindowForInput: (app: ElectronApplication, page: Page) => Promise } + +test('input setup survives a fresh-install zoom restore before onboarding', async () => { + const sandbox = createSandbox('cold-input') + unlinkSync(path.join(sandbox.userDataDir, 'zoom-state.json')) + writeFileSync(path.join(sandbox.hermesHome, 'config.yaml'), '# no provider\n', 'utf8') + let app: ElectronApplication | undefined + + try { + const launched = await launchDesktop(buildAppEnv(sandbox)) + app = launched.app + const page = launched.page + await page.waitForSelector('button', { state: 'attached' }) + await prepareWindowForInput(app, page) + const later = page.getByRole('button', { name: /choose a provider later/i }) + await expect(later).toBeVisible({ timeout: 60_000 }) + const appWindow = await app.browserWindow(page) + await appWindow.evaluate(win => win.emit('focus')) + await expect.poll(() => appWindow.evaluate(win => win.webContents.getZoomFactor())).toBeCloseTo(1) + await later.click({ timeout: 5_000 }) + await expect(later).toBeHidden() + } finally { + await app?.close().catch(() => undefined) + sandbox.cleanup() + } +}) + +// Exercise the install driver's input setup against the real renderer/backend, +// with no installer, update, credentials, or live user data. +for (const lifecycleEvent of ['focus', 'navigation'] as const) { + test(`onboarding input zoom survives ${lifecycleEvent} and opens Settings`, async () => { + const fixture = await setupNoProvider() + const { app, page, sandbox } = fixture + + try { + await prepareWindowForInput(app, page) + const later = page.getByRole('button', { name: /choose a provider later/i }) + await expect(later).toBeVisible({ timeout: 60_000 }) + const zoomFile = path.join(sandbox.userDataDir, 'zoom-state.json') + const savedLevel = () => JSON.parse(readFileSync(zoomFile, 'utf8')).zoomLevel as number + await page.evaluate(() => { + const desktop = (window as unknown as { hermesDesktop: { zoom: { setPercent: (percent: number) => void } } }).hermesDesktop + desktop.zoom.setPercent(90) + }) + await expect.poll(savedLevel).toBeCloseTo(Math.log(0.9) / Math.log(1.2)) + + await prepareWindowForInput(app, page) + const appWindow = await app.browserWindow(page) + + // The same lifecycle callback that fires when another window takes focus + // must restore our input scale, not the original 90% preference. + if (lifecycleEvent === 'focus') { + await appWindow.evaluate(win => win.emit('focus')) + } else { + await page.evaluate(() => { window.location.hash = '#/settings' }) + } + + await expect.poll(() => appWindow.evaluate(win => win.webContents.getZoomFactor())).toBeCloseTo(1) + expect(savedLevel()).toBe(0) + await later.click({ timeout: 5_000 }) + await expect(later).toBeHidden() + + if (lifecycleEvent === 'navigation') { + await page.evaluate(() => { window.location.hash = '#/' }) + } + + await page.getByRole('button', { name: 'Open settings', exact: true }).click({ timeout: 5_000 }) + await expect(page).toHaveURL(/settings/) + } finally { + await fixture.cleanup() + } + }) +} diff --git a/apps/desktop/e2e/window-input.unit.test.ts b/apps/desktop/e2e/window-input.unit.test.ts new file mode 100644 index 0000000000..a15b2b32ba --- /dev/null +++ b/apps/desktop/e2e/window-input.unit.test.ts @@ -0,0 +1,94 @@ +import { createRequire } from 'node:module' + +import { expect, test } from 'vitest' + +const { prepareWindowForInput } = createRequire(import.meta.url)( + '../../../tests/install/e2e-assets/window-input.cjs', +) + +test('does not finish when IPC reports 100% before the window factor settles', async () => { + let observations = 0 + const previous = (globalThis as any).hermesDesktop + ;(globalThis as any).hermesDesktop = { zoom: { + setPercent: () => undefined, + get: async () => ({ percent: 100 }), + } } + const appWindow = { evaluate: async (fn: any) => fn({ webContents: { + getZoomFactor: () => ++observations === 1 ? 0.9 : 1, + } }) } + const page = { + evaluate: async (fn: any) => fn(), + waitForTimeout: async () => undefined, + } + try { + await prepareWindowForInput({ browserWindow: async () => appWindow }, page) + expect(observations).toBeGreaterThan(1) + } finally { + ;(globalThis as any).hermesDesktop = previous + } +}) + +test('reapplies zoom when startup overwrites the first request', async () => { + let requests = 0 + let factor = 0.9 + + const previous = (globalThis as any).hermesDesktop + + ;(globalThis as any).hermesDesktop = { zoom: { + setPercent: () => { requests++; + + if (requests > 1) {factor = 1} }, + get: async () => ({ percent: factor * 100 }), + } } + const window = { evaluate: async (fn: any) => fn({ webContents: { getZoomFactor: () => factor } }) } + + const page = { + evaluate: async (fn: any) => fn(), + waitForTimeout: async () => { + if (requests === 1) {throw new Error('startup overwrote zoom and the driver never reapplied it')} + }, + } + + try { + await prepareWindowForInput({ browserWindow: async () => window }, page) + expect(requests).toBeGreaterThan(1) + } finally { + ;(globalThis as any).hermesDesktop = previous + } +}) + +test('awaits the zoom response instead of accepting a truthy Promise', async () => { + let reads = 0 + let factor = 0.9 + + const zoom = { + setPercent: () => undefined, + get: async () => { + reads++ + + if (reads > 1) {factor = 1} + + return { percent: factor * 100 } + }, + } + + const previous = (globalThis as any).hermesDesktop + + ;(globalThis as any).hermesDesktop = { zoom } + const window = { evaluate: async (fn: any) => fn({ webContents: { getZoomFactor: () => factor } }) } + + const page = { + evaluate: async (fn: any) => fn(), + // Playwright 1.58 accepts the predicate's Promise before it resolves. + waitForFunction: async (fn: any) => { await fn() }, + waitForTimeout: async () => undefined, + } + + try { + await prepareWindowForInput({ browserWindow: async () => window }, page) + expect(reads).toBeGreaterThan(1) + expect(factor).toBe(1) + } finally { + ;(globalThis as any).hermesDesktop = previous + } +}) diff --git a/scripts/sandbox/generate-e2e-matrix.mjs b/scripts/sandbox/generate-e2e-matrix.mjs index 75be6b4836..2db29ac3be 100644 --- a/scripts/sandbox/generate-e2e-matrix.mjs +++ b/scripts/sandbox/generate-e2e-matrix.mjs @@ -8,8 +8,8 @@ * about which combinations CI can drive. Every combination is dispatched to * its OS's run workflow, and THAT workflow natively skips the method pairs * its driver cannot run yet -- capability knowledge lives next to each - * driver (install-e2e-run.yml for linux AND macos, - * install-e2e-windows-run.yml). Correctness here is enforced by the type + * driver (install-e2e-run.yml for linux, install-e2e-macos-run.yml, + * and install-e2e-windows-run.yml). Correctness here is enforced by the type * unions below (checked via `tsc --checkJs`), not by runtime validation -- * anything the types can't catch is self-evident on the next CI run. * @@ -294,6 +294,9 @@ export function renderMarkdownPlan(envs, tags) { * @returns {string} */ export function renderMarkdownResults(jobs, tagAnnotations = [], artifactById = new Map()) { + const knownRules = JSON.parse(fs.readFileSync(new URL('../../tests/install/e2e-assets/known-failures.json', import.meta.url), 'utf8')); + /** @type {Map} */ + const footnotes = new Map(); const LEG = /^(linux|windows|macos): (\S+) -> (\S+) \(([^)]+) -> HEAD\) \//; /** @type {Map} */ const desktopByTag = new Map(tagAnnotations.map((t) => [t.ref, t.desktop])); @@ -315,11 +318,12 @@ export function renderMarkdownResults(jobs, tagAnnotations = [], artifactById = // run workflow may have one inner job per driver arm; exactly one runs // and the others natively skip), so cells merge by significance: a real // outcome always beats a skip, and a bad outcome beats a good one. - const RANK = ['skip', 'TODO', 'pre-desktop', '✅', 'running', 'cancelled', '❌']; + const RANK = ['skip', 'TODO', 'pre-desktop', '✅', 'known', 'running', 'cancelled', '❌']; const SKIPS = ['skip', 'TODO', 'pre-desktop']; // Rendered success/failure cells carry artifact links after the glyph; // rank by the leading token or every such cell would rank as unknown (-1) // and lose to any skip already in the map. + /** @param {string} cell */ const rankOf = (cell) => RANK.findIndex((t) => cell === t || cell.startsWith(`${t} `)); /** @type {Map>} */ const rows = new Map(); @@ -350,7 +354,14 @@ export function renderMarkdownResults(jobs, tagAnnotations = [], artifactById = ? ` [📼](${runBase}/artifacts/${playerId}#zip=${encodeURIComponent(`${runBase}/artifacts/${logsId}`)}) [⬇️](${runBase}/artifacts/${logsId})` : ''; switch (job.conclusion) { - case 'success': return `✅${reel}`; + case 'success': { + // Only the classifier's uploaded receipt turns a successful job into + // a known-failure cell. Tag membership alone never suppresses a red. + const rule = knownRules.find((/** @type {any} */ r) => artifactById.has(`install-e2e-known-${r.id}--${legId2}`)); + if (!rule) return `✅${reel}`; + if (!footnotes.has(rule.id)) footnotes.set(rule.id, { number: footnotes.size + 1, rule }); + return `known [^${footnotes.get(rule.id)?.number}]${reel}`; + } case 'failure': return `❌${reel}`; case 'skipped': return skipLabel(m[2], m[3], tag); case 'cancelled': return 'cancelled'; @@ -368,10 +379,11 @@ export function renderMarkdownResults(jobs, tagAnnotations = [], artifactById = const passed = cells.filter((c) => c.startsWith('✅')).length; const failed = cells.filter((c) => c.startsWith('❌')).length; const skipped = cells.filter((c) => SKIPS.includes(c)).length; + const known = cells.filter((c) => c.startsWith('known ')).length; const lines = [ '### Install & Update E2E results', '', - `${passed} passed, ${failed} failed, ${skipped} skipped (TODO = declared, no driver arm yet; pre-desktop = the starting release predates apps/desktop), ${cells.length} legs total`, + `${passed} passed, ${failed} failed, ${known} known failures, ${skipped} skipped (TODO = declared, no driver arm yet; pre-desktop = the starting release predates apps/desktop), ${cells.length} legs total`, '', `| combination | ${tags.join(' | ')} |`, `|---|${tags.map(() => '---').join('|')}|`, @@ -380,6 +392,9 @@ export function renderMarkdownResults(jobs, tagAnnotations = [], artifactById = lines.push(`| \`${combo}\` | ${tags.map((t) => byTag.get(t) || '-').join(' | ')} |`); } lines.push(''); + for (const { number, rule } of footnotes.values()) { + lines.push(`[^${number}]: **${rule.title}.** ${rule.explanation} [Evidence](${rule.evidence}).`); + } return lines.join('\n'); } diff --git a/tests-js/install-known-failures.test.ts b/tests-js/install-known-failures.test.ts new file mode 100644 index 0000000000..7799f27d7a --- /dev/null +++ b/tests-js/install-known-failures.test.ts @@ -0,0 +1,82 @@ +import { spawnSync } from 'node:child_process' +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import os from 'node:os' +import path from 'node:path' + +import { describe, expect, it } from 'vitest' + +const { matchKnownFailure, rules } = createRequire(import.meta.url)('../tests/install/e2e-assets/known-failures.cjs') +const classifier = path.resolve(import.meta.dirname, '../tests/install/e2e-assets/known-failures.cjs') +const lockedLog = [ + 'error: failed to remove file `C:/install/venv/Lib/site-packages/../../Scripts/hermes.exe`: Access is denied. (os error 5)', + 'File "C:/install/venv/Scripts/hermes.exe/__main__.py", line 10, in ', + "subprocess.CalledProcessError: Command '['uv', 'pip', 'install', '-e', '.', '--quiet']' returned non-zero exit status 2.", +].join('\n') +const base = { + platform: 'windows', phase: 'update', commit: 'a370ab8391ca5f8de7ebbc449f05cb0df36ade7c', + installMethod: 'installer-script', updateMethod: 'hermes-update', + error: 'E2E ASSERTION FAILED: hermes update exited 1 (expected 0)', logs: { update: lockedLog }, +} + +describe('known install failures', () => { + it('recognizes the released launcher self-lock, not generic access denied', () => { + expect(matchKnownFailure(base)?.id).toBe('windows-launcher-self-lock') + expect(matchKnownFailure({ ...base, logs: { update: lockedLog.replaceAll('hermes.exe', 'other.exe') } })).toBeNull() + expect(matchKnownFailure({ ...base, logs: { update: lockedLog.replace('(os error 5)', '(os error 32)') } })).toBeNull() + }) + + it.each([ + { platform: 'linux' }, { phase: 'install' }, { commit: 'v2026.3.12' }, + { commit: 'f'.repeat(40) }, { installMethod: 'desktop-installer@latest' }, + { updateMethod: 'installer-script' }, { error: 'E2E ASSERTION FAILED: update marker cleaned up' }, + { logs: {} }, + ])('rejects a different case or missing evidence: %j', change => { + expect(matchKnownFailure({ ...base, ...change })).toBeNull() + }) + + it('matches manual-only app updates only for the three proven July script cases', () => { + const sample = { + ...base, commit: '7c1a029553d87c43ecff8a3821336bc95872213b', + updateMethod: 'hermes-desktop-app-update', + error: 'E2E ASSERTION FAILED: app driven via captured hermes desktop spec; update completed', + logs: { desktop: '[hermes] [updates] no staged updater; surfacing manual `hermes update` for CLI install at C:/install\n[hermes] [updates] manual: hermes update\n' }, + } + expect(matchKnownFailure(sample)?.id).toBe('windows-july-manual-app-update') + expect(matchKnownFailure({ ...sample, installMethod: 'desktop-installer@latest' })).toBeNull() + expect(matchKnownFailure({ ...sample, error: 'onboarding timed out' })).toBeNull() + expect(matchKnownFailure({ ...sample, logs: { desktop: '[updates] manual: hermes update' } })).toBeNull() + }) + + it('CLI writes a receipt and exits zero only on a confirmed match', () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'known-install-')) + try { + mkdirSync(path.join(root, 'logs')) + writeFileSync(path.join(root, 'shas.json'), '\uFEFF' + JSON.stringify({ old: base.commit, current: 'f'.repeat(40), old_ref: 'v2026.3.12' })) + writeFileSync(path.join(root, 'logs/update.log'), lockedLog) + const args = [classifier, root, base.installMethod, base.updateMethod, base.error] + expect(spawnSync(process.execPath, args).status).toBe(0) + expect(JSON.parse(readFileSync(path.join(root, 'known-failure.json'), 'utf8')).id).toBe('windows-launcher-self-lock') + writeFileSync(path.join(root, 'logs/update.log'), 'an unrelated failure') + expect(spawnSync(process.execPath, args).status).toBe(1) + } finally { + rmSync(root, { recursive: true, force: true }) + } + }) +}) + +it('renders known receipts as footnotes, without suppressing a red job', async () => { + const modulePath = new URL('../scripts/sandbox/generate-e2e-matrix.mjs', import.meta.url).href + const { renderMarkdownResults, legId } = await import(/* @vite-ignore */ modulePath) + const name = 'windows: installer-script -> hermes-update (v2026.3.12 -> HEAD)' + const artifacts = new Map([[`install-e2e-known-${rules[0].id}--${legId(name)}`, 42]]) + const known = renderMarkdownResults([{ name: name + ' / e2e', conclusion: 'success' }], [], artifacts) + expect(known).toContain('0 passed, 0 failed, 1 known failures') + expect(known).toContain('known [^1]') + expect(known).toContain(`[^1]: **${rules[0].title}.**`) + const failed = renderMarkdownResults([{ name: name + ' / e2e', conclusion: 'failure' }], [], artifacts) + expect(failed).toContain('0 passed, 1 failed, 0 known failures') + expect(failed).not.toContain('known [^1]') + const passed = renderMarkdownResults([{ name: name + ' / e2e', conclusion: 'success' }]) + expect(passed).toContain('1 passed, 0 failed, 0 known failures') +}) diff --git a/tests-js/install-process-close.test.ts b/tests-js/install-process-close.test.ts new file mode 100644 index 0000000000..ce8b83c9b9 --- /dev/null +++ b/tests-js/install-process-close.test.ts @@ -0,0 +1,36 @@ +import { EventEmitter } from 'node:events' +import { createRequire } from 'node:module' + +import { expect, it, vi } from 'vitest' + +const { observeProcessClose } = createRequire(import.meta.url)('../tests/install/e2e-assets/process-close.cjs') + +it('waits for native close, not exit, and retains a close observed before hand-off', async () => { + const pipe = { destroy: vi.fn() } + const child = Object.assign(new EventEmitter(), { stdio: [null, pipe], exitCode: null, signalCode: null }) + const waitForClose = observeProcessClose(child) + expect(pipe.destroy).not.toHaveBeenCalled() + let finished = false + const completion = waitForClose().then(() => { finished = true }) + child.emit('exit', 0) + expect(pipe.destroy).toHaveBeenCalledOnce() + await Promise.resolve() + expect(finished).toBe(false) + child.emit('close', 0) + await completion + expect(finished).toBe(true) + await expect(waitForClose()).resolves.toBeUndefined() +}) + +it('fails if the launched process never closes', async () => { + vi.useFakeTimers() + try { + const waitForClose = observeProcessClose(Object.assign(new EventEmitter(), { stdio: [], exitCode: null, signalCode: null })) + const completion = expect(waitForClose(2_000)).rejects.toThrow('Electron process did not close') + await vi.advanceTimersByTimeAsync(2_000) + await completion + expect(vi.getTimerCount()).toBe(0) + } finally { + vi.useRealTimers() + } +}) diff --git a/tests/hermes_cli/test_install_progress_stream.py b/tests/hermes_cli/test_install_progress_stream.py new file mode 100644 index 0000000000..c2a4ee09c2 --- /dev/null +++ b/tests/hermes_cli/test_install_progress_stream.py @@ -0,0 +1,32 @@ +"""Installer progress must use the stream drained by the desktop updater.""" + +import subprocess +import sys + +import pytest + +from hermes_cli.main_install_repair import _run_install_with_heartbeat + + +@pytest.mark.parametrize("exit_code", [0, 7]) +def test_installer_stderr_streams_to_stdout(tmp_path, monkeypatch, capfd, exit_code): + import hermes_cli.main as main + + monkeypatch.setattr(main, "PROJECT_ROOT", tmp_path) + # More than a pipe buffer of progress, from a real child on the native host. + size = 256 * 1024 + cmd = [ + sys.executable, + "-c", + f"import sys; sys.stderr.write('x' * {size}); sys.stderr.flush(); sys.exit({exit_code})", + ] + if exit_code: + with pytest.raises(subprocess.CalledProcessError) as error: + _run_install_with_heartbeat(cmd) + assert error.value.returncode == exit_code + else: + _run_install_with_heartbeat(cmd) + + output = capfd.readouterr() + assert output.out == "x" * size + assert output.err == "" diff --git a/tests/hermes_cli/test_managed_uv.py b/tests/hermes_cli/test_managed_uv.py index a81fd52501..a68ed36cae 100644 --- a/tests/hermes_cli/test_managed_uv.py +++ b/tests/hermes_cli/test_managed_uv.py @@ -16,6 +16,12 @@ import pytest # Helpers # --------------------------------------------------------------------------- +# Host-native managed-uv binary name: managed_uv_path() installs `uv` on +# POSIX and `uv.exe` on Windows. Fixtures must build what the real host +# resolves — no platform fake. +_UV_BINARY_NAME = "uv.exe" if sys.platform == "win32" else "uv" + + def _make_executable(path: Path) -> None: """Create a minimal fake uv binary at *path*.""" path.parent.mkdir(parents=True, exist_ok=True) @@ -158,11 +164,12 @@ class TestMacOSManagedPythonSigning: class TestResolveUv: def test_existing_executable(self, tmp_path): - _make_executable(tmp_path / "bin" / "uv") + uv = tmp_path / "bin" / _UV_BINARY_NAME + _make_executable(uv) with patch("hermes_cli.managed_uv.get_hermes_home", return_value=tmp_path): from hermes_cli.managed_uv import resolve_uv result = resolve_uv() - assert result == str(tmp_path / "bin" / "uv") + assert result == str(uv) def test_non_executable_file_returns_none(self, tmp_path): uv = tmp_path / "bin" / "uv" @@ -182,17 +189,20 @@ class TestResolveUv: class TestEnsureUv: def test_installs_if_missing(self, tmp_path): + uv = tmp_path / "bin" / _UV_BINARY_NAME with patch("hermes_cli.managed_uv.get_hermes_home", return_value=tmp_path), \ patch("hermes_cli.managed_uv.repair_vulnerable_runtime", return_value=_RRR("not-applicable")), \ + patch("hermes_cli.managed_uv._uv_version", return_value="uv 0.1.2"), \ patch("hermes_cli.managed_uv._install_uv") as mock_install: - # Simulate the installer creating the binary + # Simulate the installer creating the binary (host-native name: + # uv.exe on Windows, uv on POSIX). def fake_install(target): _make_executable(target) mock_install.side_effect = fake_install from hermes_cli.managed_uv import ensure_uv path = ensure_uv() - assert path == str(tmp_path / "bin" / "uv") + assert path == str(uv) mock_install.assert_called_once() def test_install_reports_runtime_repair_to_observer(self, tmp_path): @@ -217,13 +227,16 @@ class TestEnsureUv: ), patch( "hermes_cli.managed_uv._install_uv", side_effect=fake_install, + ), patch( + "hermes_cli.managed_uv._uv_version", + return_value="uv 0.1.2", ), patch( "hermes_cli.managed_uv.repair_vulnerable_runtime", return_value=repair, ): path = ensure_uv(repair_observer=observed.append) - assert path == str(tmp_path / "bin" / "uv") + assert path == str(tmp_path / "bin" / _UV_BINARY_NAME) assert observed == [repair] @@ -329,25 +342,30 @@ class TestUpdateManagedUv: - def test_fresh_stamp_skips_network_self_update_but_not_repair(self, tmp_path, monkeypatch): + def test_fresh_stamp_skips_network_self_update_but_not_repair(self, tmp_path): """A recent success stamp must skip `uv self update` entirely while the vulnerable-runtime repair probe still runs (CVE repair is never gated).""" + import time + from hermes_cli.managed_uv import RuntimeRepairResult, update_managed_uv - uv = tmp_path / "bin" / "uv" + uv = tmp_path / "bin" / _UV_BINARY_NAME _make_executable(uv) - # Fresh stamp under the isolated HERMES_HOME. - import hermes_constants - stamp = hermes_constants.get_hermes_home() / "cache" / ".uv_self_update_stamp" + # The stamp reader imports get_hermes_home separately from the binary + # resolver. Give both paths the same explicit test root. + stamp = tmp_path / "cache" / ".uv_self_update_stamp" stamp.parent.mkdir(parents=True, exist_ok=True) stamp.touch() + # File timestamps can lead time.time() briefly on Windows. Stay well + # inside the freshness window instead of racing its age >= 0 boundary. + recent = time.time() - 60 + os.utime(stamp, (recent, recent)) with patch("hermes_cli.managed_uv.get_hermes_home", return_value=tmp_path), \ - patch("hermes_cli.managed_uv.subprocess.run") as mock_run, \ - patch( - "hermes_cli.managed_uv.repair_vulnerable_runtime", - return_value=RuntimeRepairResult("skipped"), - ) as mock_repair: + patch("hermes_cli.managed_uv._uv_self_update_stamp", return_value=stamp), \ + patch("hermes_cli.managed_uv.repair_vulnerable_runtime", + return_value=RuntimeRepairResult("skipped")) as mock_repair, \ + patch("hermes_cli.managed_uv.subprocess.run") as mock_run: result = update_managed_uv() assert result == str(uv) @@ -361,17 +379,19 @@ class TestUpdateManagedUv: from hermes_cli.managed_uv import UV_SELF_UPDATE_INTERVAL_SECONDS, update_managed_uv - uv = tmp_path / "bin" / "uv" + uv = tmp_path / "bin" / _UV_BINARY_NAME _make_executable(uv) - import hermes_constants - stamp = hermes_constants.get_hermes_home() / "cache" / ".uv_self_update_stamp" + # Keep the stamp and binary resolver in the same test root. + stamp = tmp_path / "cache" / ".uv_self_update_stamp" stamp.parent.mkdir(parents=True, exist_ok=True) stamp.touch() old = _time.time() - UV_SELF_UPDATE_INTERVAL_SECONDS - 60 _os.utime(stamp, (old, old)) with patch("hermes_cli.managed_uv.get_hermes_home", return_value=tmp_path), \ + patch("hermes_cli.managed_uv._uv_self_update_stamp", return_value=stamp), \ patch("hermes_cli.managed_uv.repair_vulnerable_runtime", return_value=_RRR("not-applicable")), \ + patch("hermes_cli.managed_uv._uv_version", return_value="uv 0.2.0"), \ patch("hermes_cli.managed_uv.subprocess.run") as mock_run: mock_run.return_value = MagicMock(returncode=0, stdout="uv 0.2.0") update_managed_uv() @@ -464,47 +484,6 @@ class TestRuntimeRepair: assert not (root / ".hermes-runtime").exists() mock_install.assert_not_called() - def test_stage_candidate_sync_keeps_uv_project_config(self, tmp_path): - from hermes_cli.managed_uv import _stage_candidate_venv - - root = tmp_path / "checkout" - root.mkdir() - (root / "uv.lock").write_text("# lock\n", encoding="utf-8") - generation = root / ".hermes-runtime" / "python" / "gen" - python = generation / "bin" / "python" - python.parent.mkdir(parents=True) - python.write_text("py", encoding="utf-8") - - calls = [] - - def fake_run(argv, **kwargs): - calls.append((list(argv), kwargs.get("env"))) - return MagicMock(returncode=0) - - with patch("hermes_cli.managed_uv.subprocess.run", side_effect=fake_run), \ - patch( - "hermes_cli.managed_uv._smoke_candidate_venv", - return_value=(True, "", None), - ): - candidate = _stage_candidate_venv( - "uv", - project_root=root, - generation=generation, - python=python, - ) - - assert candidate is not None - assert len(calls) == 2 - venv_argv, venv_env = calls[0] - sync_argv, sync_env = calls[1] - assert venv_argv[:2] == ["uv", "venv"] - assert "--no-config" in venv_argv - assert venv_env.get("UV_NO_CONFIG") == "1" - assert sync_argv[:2] == ["uv", "sync"] - assert "--locked" in sync_argv - assert "--no-config" not in sync_argv - assert "UV_NO_CONFIG" not in sync_env - def test_failed_candidate_preserves_live_venv(self, tmp_path): from hermes_cli.managed_uv import ( _acquire_repair_lock, @@ -620,6 +599,54 @@ class TestRuntimeRepair: assert leftovers == [], f"no stale markers may remain: {leftovers}" +class TestStageCandidateVenvCrossPlatform: + """Candidate sync preserves project config and streams progress on every host.""" + + def test_sync_keeps_uv_project_config_and_merges_stderr(self, tmp_path): + import subprocess + + from hermes_cli.managed_uv import _stage_candidate_venv + + root = tmp_path / "checkout" + root.mkdir() + (root / "uv.lock").write_text("# lock\n", encoding="utf-8") + generation = root / ".hermes-runtime" / "python" / "gen" + python = generation / "bin" / "python" + python.parent.mkdir(parents=True) + python.write_text("py", encoding="utf-8") + + calls = [] + + def fake_run(argv, **kwargs): + calls.append((list(argv), kwargs)) + return MagicMock(returncode=0) + + with patch("hermes_cli.managed_uv.subprocess.run", side_effect=fake_run), \ + patch( + "hermes_cli.managed_uv._smoke_candidate_venv", + return_value=(True, "", None), + ): + candidate = _stage_candidate_venv( + "uv", + project_root=root, + generation=generation, + python=python, + ) + + assert candidate is not None + assert len(calls) == 2 + venv_argv, venv_kwargs = calls[0] + sync_argv, sync_kwargs = calls[1] + assert venv_argv[:2] == ["uv", "venv"] + assert "--no-config" in venv_argv + assert venv_kwargs["env"].get("UV_NO_CONFIG") == "1" + assert sync_argv[:2] == ["uv", "sync"] + assert "--locked" in sync_argv + assert "--no-config" not in sync_argv + assert "UV_NO_CONFIG" not in sync_kwargs["env"] + assert sync_kwargs["stderr"] == subprocess.STDOUT + + class TestRuntimeCutover: def test_os_lock_blocks_concurrent_repair_and_releases(self, tmp_path): from hermes_cli.managed_uv import _acquire_repair_lock, _release_repair_lock @@ -674,13 +701,23 @@ class TestRuntimeCutover: # --------------------------------------------------------------------------- class TestInstallUvInternals: - def test_posix_sets_uv_unmanaged_install(self, tmp_path): - target = tmp_path / "bin" / "uv" - with patch("hermes_cli.managed_uv._install_uv_posix") as mock_posix: - from hermes_cli.managed_uv import _install_uv - _install_uv(target) - mock_posix.assert_called_once() - call_env = mock_posix.call_args[0][0] + def test_installer_uses_host_branch_and_managed_directory(self, tmp_path): + """The native installer receives the managed directory, not a PATH default.""" + import hermes_cli.managed_uv as managed_uv + + target = tmp_path / "bin" / _UV_BINARY_NAME + with patch("hermes_cli.managed_uv._install_uv_posix") as mock_posix, \ + patch("hermes_cli.managed_uv._install_uv_windows") as mock_windows: + managed_uv._install_uv(target) + + host_installer, other_installer = ( + (mock_windows, mock_posix) if sys.platform == "win32" + else (mock_posix, mock_windows)) + host_installer.assert_called_once() + other_installer.assert_not_called() + call_env = host_installer.call_args[0][0] + assert call_env["UV_INSTALL_DIR"] == str(tmp_path / "bin") + if sys.platform != "win32": assert call_env["UV_UNMANAGED_INSTALL"] == str(tmp_path / "bin") @@ -1252,10 +1289,16 @@ class TestDefaultLiveVenv: root = tmp_path / "checkout" root.mkdir() (root / "pyproject.toml").write_text("[project]\n", encoding="utf-8") + # Host-native venv layout: bin/python on POSIX, Scripts/python.exe on + # Windows — what _venv_python() resolves on the real host. + if sys.platform == "win32": + bin_dir_name, python_name = "Scripts", "python.exe" + else: + bin_dir_name, python_name = "bin", "python" for d in dirs: - bin_dir = root / d / "bin" + bin_dir = root / d / bin_dir_name bin_dir.mkdir(parents=True) - (bin_dir / "python").write_text("py", encoding="utf-8") + (bin_dir / python_name).write_text("py", encoding="utf-8") return root def test_dot_venv_only_is_targeted(self, tmp_path): diff --git a/tests/install/KNOWN_FAILURES.md b/tests/install/KNOWN_FAILURES.md new file mode 100644 index 0000000000..a37d5da81a --- /dev/null +++ b/tests/install/KNOWN_FAILURES.md @@ -0,0 +1,44 @@ +# Confirmed historical upgrade limitations + +These failures cannot be repaired by changing the update target: the failing code is already loaded from the starting release. The workflow still executes each original update path. Only a match on the exact starting commit, method pair, failed assertion, and fresh log signatures produces a non-red known-failure receipt. Other errors still fail. The results table shows matched cases as `known [n]`, with the explanation and evidence in a footnote at the bottom. These cases are counted separately from passed upgrades. + +The machine-readable rules in `e2e-assets/known-failures.json` own the matcher and report footnote text. This document explains their historical evidence. Logs are rotated before each attempt so an earlier failure cannot classify a later one. + +## Windows launcher self-lock + +Classification: **unfixable in the update target for the exact released `hermes.exe update` path**. + +| Starting release | Released commit | Install → update | Verified failing job | +|---|---|---|---| +| `v2026.3.12` | `a370ab8391ca5f8de7ebbc449f05cb0df36ade7c` | `installer-script` → `hermes-update` | [101514756800](https://github.com/ethernet8023/hermes-agent/actions/runs/34043635705/job/101514756800) | +| `v2026.4.8` | `86960cdbb0148145890e2ee90b4e157fa899f6e1` | `installer-script` → `hermes-update` | [101514755527](https://github.com/ethernet8023/hermes-agent/actions/runs/34043635705/job/101514755527) | + +The running console launcher holds `venv/Scripts/hermes.exe` open. The old updater pulls the new checkout, then asks uv to replace that same executable during an editable install. Windows rejects the replacement with `Access is denied. (os error 5)`. The old updater's ZIP fallback repeats the dependency install and encounters the same lock. + +Evidence required: the CLI update phase failed, the traceback identifies the running `hermes.exe/__main__.py`, and uv reports failure to remove that install's `Scripts/hermes.exe` with OS error 5. A generic access-denied error on another file does not match. + +The March call is in the released `hermes_cli/main.py:1678-1683`, with the ZIP fallback at `1571-1576`. April calls `_install_python_dependencies_with_optional_fallback`, whose released body at `3295-3321` runs the installs without launcher quarantine. Those function objects were loaded before the checkout changed. May's sampled CLI update passed; do not classify it from this record. + +Re-running the installer is a separate tested upgrade route. Invoking the old CLI through its venv Python is a possible recovery route, but is not silently substituted for the console-launcher leg. + +## July Windows app offers only a manual update for script installs + +Classification: **unfixable in the update target for the exact released app-button path**. + +Starting release: `v2026.7.1`, commit `7c1a029553d87c43ecff8a3821336bc95872213b`. + +| Install → update | Verified failing job | +|---|---| +| `installer-script` → `hermes-desktop-app-update` | [101514755236](https://github.com/ethernet8023/hermes-agent/actions/runs/34043635705/job/101514755236) | +| `installer-script+desktop` → `hermes-desktop-app-update` | [101514760893](https://github.com/ethernet8023/hermes-agent/actions/runs/34043635705/job/101514760893) | +| `installer-script+desktop` → `open-app-update` | [101514756508](https://github.com/ethernet8023/hermes-agent/actions/runs/34043635705/job/101514756508) | + +These script installs have no staged updater. The released Electron code (`apps/desktop/electron/main.cjs:2212-2214`) logs `no staged updater; surfacing manual` and returns `{ ok: true, manual: true, command }`. It does not start an update. Each job's `logs/desktop.log` records that branch followed by `[updates] manual: hermes update`; no target checkout/result signal appears. + +Evidence required: an app-update leg from this released commit and those explicit manual-update log entries. A hand-off timeout without the manual message is not this limitation. Desktop-installer installs have a different staged-updater path and are not covered by this classification. + +## Not classified as unfixable + +The July desktop-installer → app-update failure was a driver lifetime bug, not a released-updater exception. The driver treated an expected page closure as failure and could exit before Playwright released its launch process. On Windows, inherited pipes delayed the `close` event even after the launch process exited with code 0. Playwright then ran its tree-kill cleanup. The driver now waits independently of the closing page, releases its pipe handles after process exit, and waits for `close` before it exits. [The real July rerun](https://github.com/ethernet8023/hermes-agent/actions/runs/34075042380/job/101599434616) reached the target commit, cleared the update marker, passed the CLI check, and relaunched the app. + +Onboarding click failures, zoom drift, native permission dialogs, AutoHotkey window waits, stale update markers, autostash conflicts, network failures, and generic timeouts remain actionable or unclassified until diagnosed. They must not inherit a historical label because they occurred on an old release. diff --git a/tests/install/README.md b/tests/install/README.md index fe84c68018..cc4a41c730 100644 --- a/tests/install/README.md +++ b/tests/install/README.md @@ -10,8 +10,8 @@ The test family has four layers. Each layer has one job. 1. `scripts/sandbox/generate-e2e-matrix.mjs` declares the support matrix. It lists every {os, install-method, update-method} pair. It expands the pairs against the sampled release tags. It knows nothing about which pairs CI can run. 2. `.github/workflows/install-e2e.yml` is the primary workflow. It picks the release tags, runs the generator, and fans out one matrix job per OS. It also writes the plan chart and the result chart on the run summary. -3. The run workflows own the capability knowledge. `install-e2e-run.yml` serves linux and macos with one OS-agnostic driver. `install-e2e-windows-run.yml` serves windows. A job-level `if:` gate in each run workflow lists the pairs its driver can run. All other pairs skip natively and show as grey. -4. The drivers do the work. `tests/install/installer-script-e2e.sh` is the POSIX driver. `tests/install/windows-e2e.ps1` is the windows driver; its install phase and update phase dispatch on separate method parameters, so any implemented update method can follow any implemented install method. +3. The run workflows own the capability knowledge. `install-e2e-run.yml` serves linux. `install-e2e-windows-run.yml` serves windows. `install-e2e-macos-run.yml` selects either the shared script driver or the macOS GUI driver. Job-level `if:` gates select the supported pairs. All other pairs skip natively and show as grey. +4. The drivers do the work. `tests/install/installer-script-e2e.sh` handles POSIX script installs, `tests/install/macos-desktop-e2e.sh` handles macOS dmg installs, and `tests/install/windows-e2e.ps1` handles Windows installs. Install and update methods are separate axes, subject to each workflow's capability gates. To declare a new method, edit the generator. To implement a method, flip the gate in the run workflow and extend a driver. @@ -53,7 +53,7 @@ A leg can install a release from months back. The driver must not assume that th The desktop app has two launch paths, so the matrix has two app-update methods. Both click "Update now" in the running app. They differ in how the app starts: -- `open-app-update`: the app starts from the OS entry point that the install created. On windows these are the Start Menu and Desktop shortcuts to the installed `Hermes.exe`; the desktop installer always creates them. The installer scripts do not create entry points: their opt-in desktop stage (`--include-desktop` / `-IncludeDesktop`) builds the app inside the checkout but does not register it with the OS. So `open-app-update` legs pair with a `desktop-installer` install. +- `open-app-update`: the app starts from the installed app entry point. On Windows, both the desktop installer and `installer-script+desktop` create shortcuts, so both support this route. On Linux and macOS, the script's opt-in desktop stage builds inside the checkout without registering an OS entry point. The macOS route therefore requires a desktop-installer install; Linux has no open-app-update leg. - `hermes-desktop-app-update`: the app starts with the `hermes desktop` command. Every install method provides this command, on each OS that ships the desktop app. On linux this is the only app surface: no desktop installer and no packaged desktop artifact exist for linux. The driver captures the product's own launch call (argv, cwd, environment) with `e2e-assets/launch-capture/sitecustomize.py` and re-executes it under Playwright, which owns the app and clicks the update flow. ## Skips @@ -63,7 +63,7 @@ A grey leg is normal. There are two causes: - The method pair is declared but cannot run: either no OS entry point exists for it (open-app-update after a plain script install registers nothing to open), or no driver arm exists yet. The gate in the run workflow lists the pairs that run. - The starting release predates the surface under test. Example: a release without `apps/desktop` has no window to launch. The tag annotation `tag_has_desktop` from the primary workflow marks these releases. -The result chart on the run summary shows each leg as passed, failed, or skipped. +The result chart on the run summary shows each leg as passed, failed, or skipped. [Confirmed historical upgrade limitations](KNOWN_FAILURES.md) records failures that cannot be fixed in the update target, with exact release commits and CI evidence. These are not blanket skips: the original paths still run. Exact signature matches are non-red, counted separately as known failures, and linked to footnotes at the bottom of the result chart. An unrelated error on the same tag still fails. ## Triggers and cost @@ -77,7 +77,7 @@ The matrix does not run on pull requests. One leg installs real toolchains and t gh workflow run install-e2e.yml --ref -f route=both -f tag-count=2 ``` -Cost per run, so nobody is surprised: 41 legs per sampled tag (windows 18, macos 15, linux 8), so the default 2 tags is up to 82 legs. A typical green leg finishes in 7-15 minutes; every leg is capped at 60. Route slices for cheaper reads: `update` (linux only, 8/tag), `windows-desktop` (18/tag), `macos-desktop` (15/tag). `tag-count` is validated to 1-10. GitHub's 256-job cap applies to each OS matrix separately, not to the combined leg count; at 10 tags the matrices hold 180 windows, 150 macos, and 80 linux entries. Windows would first exceed the cap at 15 tags (270). +Cost per run, so nobody is surprised: 41 legs per sampled tag (windows 18, macos 15, linux 8), so scheduled and release-tag runs sample 2 tags for up to 82 legs. Manual dispatch defaults to 3 tags for up to 123 legs. A typical green leg finishes in 7-15 minutes; every leg is capped at 60. Route slices for cheaper reads: `update` (linux only, 8/tag), `windows-desktop` (18/tag), `macos-desktop` (15/tag). `tag-count` is validated to 1-10. GitHub's 256-job cap applies to each OS matrix separately, not to the combined leg count; at 10 tags the matrices hold 180 windows, 150 macos, and 80 linux entries. Windows would first exceed the cap at 15 tags (270). Running the drivers locally: don't, except in a disposable VM. The windows driver kills every process named Hermes during teardown and the macos driver operates on `/Applications/Hermes.app`; on a machine with a real Hermes install they will interfere with it. diff --git a/tests/install/e2e-assets/drive-update.cjs b/tests/install/e2e-assets/drive-update.cjs index e20cd76e18..e7343a1aa4 100644 --- a/tests/install/e2e-assets/drive-update.cjs +++ b/tests/install/e2e-assets/drive-update.cjs @@ -20,6 +20,8 @@ const path = require('node:path') const fs = require('node:fs') const { _electron } = require('@playwright/test') +const { prepareWindowForInput } = require('./window-input.cjs') +const { observeProcessClose } = require('./process-close.cjs') const exePath = process.argv[2] const proofDir = process.argv[3] @@ -92,6 +94,10 @@ async function main() { env: { ...process.env }, timeout: 120_000 }) + const child = app.process() + + const waitForProcessClose = observeProcessClose(child) + log(`launched Electron pid=${child.pid}`) // firstWindow() can grab a helper webContents (wake indicator etc.), not // the main app window. Pick the window that actually renders UI (has a @@ -117,29 +123,8 @@ async function main() { log(`window picked (${app.windows().length} windows, url=${page.url()})`) log('first window acquired') - // The app ships a 90% UI-zoom default, so a fresh install renders at - // devicePixelRatio 0.9 and Playwright's input coordinates land ~10% off - // target on the CI runners. Force 100% through the same webContents API - // the app's zoom control uses. A single set gets reverted (the boot path - // re-applies the default asynchronously), so set-verify-retry until dpr - // reads 1. - try { - const before = await page.evaluate(() => window.devicePixelRatio) - let after = before - for (let i = 0; i < 20; i++) { - await app.evaluate(({ BrowserWindow }) => { - for (const w of BrowserWindow.getAllWindows()) { - w.webContents.setZoomLevel(0) - } - }) - await page.waitForTimeout(1000) - after = await page.evaluate(() => window.devicePixelRatio) - if (Math.abs(after - 1) < 0.001) break - } - log(`[zoom] forced 100% via webContents.setZoomLevel(0): dpr ${before} -> ${after}`) - } catch (e) { - log(`[zoom] direct zoom set failed (continuing): ${e.message}`) - } + await prepareWindowForInput(app, page) + log('[zoom] app window prepared at 100%') // Boot: wait for the composer to exist — the shell is mounted by then. // The real backend (`hermes serve`) is booting underneath; give it time. @@ -181,6 +166,8 @@ async function main() { let iter = 0 while (!openedSettings) { + // A boot-time restore or focus event can move the scale after preparation. + await prepareWindowForInput(app, page) iter++ for (const make of laterLocators) { try { @@ -193,28 +180,6 @@ async function main() { log(`[overlay] iter ${iter} dismiss click failed: ${brief(e)}`) } } - // Escalation: the picker's button can be actionable while the OS-level - // click point lands on a wrapping container that swallows the pointer, - // so real clicks never register. dispatchEvent fires the DOM handler - // directly, bypassing hit-testing — but only after real clicks have - // had two full iterations to land, so runs where clicking works never - // take the shortcut. - if (iter >= 3) { - for (const make of laterLocators) { - try { - const el = make(page).first() - if (await el.isVisible({ timeout: 500 })) { - await el.dispatchEvent('click') - log('dismissed onboarding overlay (dispatchEvent fallback)') - await page.waitForTimeout(2500) - await shot(page, '01b-onboarding-dismissed') - break - } - } catch (e) { - log(`[overlay] iter ${iter} dispatch fallback failed: ${brief(e)}`) - } - } - } for (const make of settingsLocators) { try { await make(page).first().click({ timeout: 2_500 }) @@ -295,7 +260,8 @@ async function main() { // The "Updating Hermes — this window will close" overlay should appear, // then the app quits (hand-off dwell). Screenshot the overlay while the // window is still alive. - await page.waitForTimeout(1200) + // The app can close during the dwell. This wait must outlive its page. + await new Promise(resolve => setTimeout(resolve, 1200)) await shot(page, '05-updating-overlay') // ── Wait for the hand-off to take over ──────────────────────────────── @@ -355,7 +321,10 @@ async function main() { throw new Error('no hand-off within 150s of Update now (no marker, no result, app still alive)') } - log('hand-off confirmed — detached updater owns the rest') + // A marker appears before Electron exits. Exiting this driver at that point + // lets Playwright taskkill the entire tree, including the detached updater. + await waitForProcessClose() + log('Electron process closed — detached updater owns the rest') } main() diff --git a/tests/install/e2e-assets/install-and-launch.ahk b/tests/install/e2e-assets/install-and-launch.ahk index 960a740bb3..31a14eb7a7 100644 --- a/tests/install/e2e-assets/install-and-launch.ahk +++ b/tests/install/e2e-assets/install-and-launch.ahk @@ -199,14 +199,16 @@ if launchFound { } Log("Launch clicked; waiting for the Hermes desktop app window") -; The installer spawns Hermes.exe detached and exits itself. +; WinWait returns 0 on timeout; it does not throw. The old unchecked return +; led to WinGetPos throwing "Target window not found." Reuse the bounded +; real-window poll so transient handles are ignored and failures name the wait. +; CI's installer remained on LAUNCHING past 120s after a successful bootstrap. try { - WinWait(appWin, , 120) + appRect := WaitForRealWindow(appWin, 300000) } catch { - throw Error("Hermes.exe window did not appear within 120s of clicking Launch") + throw Error("Hermes.exe real-sized window did not appear within 300s of clicking Launch") } -WinGetPos(&ax, &ay, &aw, &ah, appWin) -Log(Format("App window appeared at x={1} y={2} w={3} h={4}", ax, ay, aw, ah)) +Log(Format("App window appeared at x={1} y={2} w={3} h={4}", appRect.x, appRect.y, appRect.w, appRect.h)) Sleep(8000) ; let the renderer paint (recorded as proof) Log("done") diff --git a/tests/install/e2e-assets/known-failures.cjs b/tests/install/e2e-assets/known-failures.cjs new file mode 100644 index 0000000000..d350a77af2 --- /dev/null +++ b/tests/install/e2e-assets/known-failures.cjs @@ -0,0 +1,54 @@ +const fs = require('node:fs') +const path = require('node:path') +const rules = require('./known-failures.json') + +function matchKnownFailure({ platform, phase, commit, installMethod, updateMethod, error, logs }) { + if (platform !== 'windows' || phase !== 'update' || !/^[0-9a-f]{40}$/.test(commit || '')) return null + return rules.find(rule => + rule.commits.includes(commit) && + rule.cases.some(([install, update]) => install === installMethod && update === updateMethod) && + rule.errors.some(pattern => new RegExp(pattern).test(error || '')) && + rule.signatures.every(pattern => new RegExp(pattern, 'i').test(logs[rule.log] || '')), + ) || null +} + +function readOptional(file) { + try { return fs.readFileSync(file, 'utf8').replace(/^\uFEFF/, '') } catch (error) { + if (error.code === 'ENOENT') return '' + throw error + } +} + +function classifyWorkRoot(root, installMethod, updateMethod, error) { + const state = JSON.parse(fs.readFileSync(path.join(root, 'shas.json'), 'utf8').replace(/^\uFEFF/, '')) + const rule = matchKnownFailure({ + platform: 'windows', phase: 'update', commit: state.old, installMethod, updateMethod, error, + logs: { + update: readOptional(path.join(root, 'logs', 'update.log')), + desktop: readOptional(path.join(root, 'hermes-home', 'logs', 'desktop.log')), + }, + }) + if (!rule) return null + return { + id: rule.id, title: rule.title, explanation: rule.explanation, evidence: rule.evidence, + commit: state.old, target: state.current, installRef: state.old_ref, + installMethod, updateMethod, error, + } +} + +module.exports = { matchKnownFailure, classifyWorkRoot, rules } + +if (require.main === module) { + const [root, install, update, error] = process.argv.slice(2) + try { + const receipt = classifyWorkRoot(root, install, update, error) + if (!receipt) process.exitCode = 1 + else { + fs.writeFileSync(path.join(root, 'known-failure.json'), JSON.stringify(receipt, null, 2) + '\n') + console.log(JSON.stringify(receipt)) + } + } catch (error) { + console.error(`known-failure classification failed: ${error.message}`) + process.exitCode = 2 + } +} diff --git a/tests/install/e2e-assets/known-failures.json b/tests/install/e2e-assets/known-failures.json new file mode 100644 index 0000000000..eaf226778a --- /dev/null +++ b/tests/install/e2e-assets/known-failures.json @@ -0,0 +1,24 @@ +[ + { + "id": "windows-launcher-self-lock", + "title": "Released Windows updater locks its own console launcher", + "commits": ["a370ab8391ca5f8de7ebbc449f05cb0df36ade7c", "86960cdbb0148145890e2ee90b4e157fa899f6e1"], + "cases": [["installer-script", "hermes-update"]], + "errors": ["^E2E ASSERTION FAILED: hermes update exited [1-9][0-9]* \\(expected 0\\)$"], + "log": "update", + "signatures": ["failed to remove file[^\\r\\n]*Scripts[/\\\\]hermes\\.exe[^\\r\\n]*Access is denied\\. \\(os error 5\\)", "hermes\\.exe[/\\\\]__main__\\.py", "CalledProcessError[^\\r\\n]*pip[^\\r\\n]*install[^\\r\\n]*returned non-zero exit status 2"], + "explanation": "The March/April Windows updater is already running from hermes.exe when uv tries to replace it. Windows refuses the locked launcher. The update target cannot change that loaded code; re-running the installer is a separate recovery route.", + "evidence": "https://github.com/ethernet8023/hermes-agent/actions/runs/34055462305/job/101546487700" + }, + { + "id": "windows-july-manual-app-update", + "title": "July script installs offer a manual update instead of an app hand-off", + "commits": ["7c1a029553d87c43ecff8a3821336bc95872213b"], + "cases": [["installer-script", "hermes-desktop-app-update"], ["installer-script+desktop", "hermes-desktop-app-update"], ["installer-script+desktop", "open-app-update"]], + "errors": ["^E2E ASSERTION FAILED: app driven via captured hermes desktop spec; update completed$", "^E2E ASSERTION FAILED: GUI driver clicked Update now and the app quit for hand-off$"], + "log": "desktop", + "signatures": ["\\[updates\\] no staged updater; surfacing manual `hermes update` for CLI install at", "\\[updates\\] manual: hermes update(?:\\r?\\n|$)"], + "explanation": "The July Windows app has no staged updater after a script install. Its Update button explicitly returns manual: hermes update without starting a hand-off. Desktop-installer installs are not covered by this exception.", + "evidence": "https://github.com/ethernet8023/hermes-agent/actions/runs/34055462305/job/101546487667" + } +] diff --git a/tests/install/e2e-assets/launch-from-spec.mjs b/tests/install/e2e-assets/launch-from-spec.mjs index eaf87f28d1..bfe372e08c 100644 --- a/tests/install/e2e-assets/launch-from-spec.mjs +++ b/tests/install/e2e-assets/launch-from-spec.mjs @@ -32,6 +32,7 @@ import path from 'node:path'; import { execFileSync } from 'node:child_process'; import { parseArgs } from 'node:util'; import { _electron } from '@playwright/test'; +import { prepareWindowForInput } from './window-input.cjs'; /** * @typedef {{argv: string[], cwd: string, env: Record, @@ -150,29 +151,8 @@ async function main() { log(`window up: ${await window.title()} (${app.windows().length} windows, picked url=${window.url()})`); await window.screenshot({ path: `${values.spec}.window.png` }).catch(() => {}); - // The app ships a 90% UI-zoom default, so a fresh install renders at - // devicePixelRatio 0.9 and Playwright's input coordinates land ~10% off - // target on the CI runners. Force 100% through the same webContents API - // the app's zoom control uses. A single set gets reverted (the boot path - // re-applies the default asynchronously), so set-verify-retry until dpr - // reads 1. - try { - const before = await window.evaluate(() => window.devicePixelRatio); - let after = before; - for (let i = 0; i < 20; i++) { - await app.evaluate(({ BrowserWindow }) => { - for (const w of BrowserWindow.getAllWindows()) { - w.webContents.setZoomLevel(0); - } - }); - await window.waitForTimeout(1000); - after = await window.evaluate(() => window.devicePixelRatio); - if (Math.abs(after - 1) < 0.001) break; - } - log(`[zoom] forced 100% via webContents.setZoomLevel(0): dpr ${before} -> ${after}`); - } catch (e) { - log(`[zoom] direct zoom set failed (continuing): ${e.message}`); - } + await prepareWindowForInput(app, window); + log('[zoom] app window prepared at 100%'); if (values['no-update']) { log('smoke mode: window proven, closing'); @@ -236,6 +216,7 @@ async function main() { } }).then((d) => JSON.stringify(d)).catch((e) => `hit-dump failed: ${e.message}`) for (let iter = 1; ; iter++) { + await prepareWindowForInput(app, window); await later .click({ timeout: 2_000 }) .then(async () => { @@ -245,6 +226,9 @@ async function main() { .catch((e) => log(`[overlay] iter ${iter} chooseLater click failed: ${brief(e)}`)) try { await settingsButton.click({ timeout: 4_000 }) + // A landed click during shell hydration can be lost on a remount. + // Confirm the destination before looking for its About control. + await window.waitForURL(/[#/]settings(?:[/?]|$)/, { timeout: 4_000 }) settingsOpened = true break } catch (e) { diff --git a/tests/install/e2e-assets/process-close.cjs b/tests/install/e2e-assets/process-close.cjs new file mode 100644 index 0000000000..e0aca5ae5a --- /dev/null +++ b/tests/install/e2e-assets/process-close.cjs @@ -0,0 +1,32 @@ +// Observe at launch: a renderer can close before the native process and its +// stdio pipes. Playwright's driver exit cleanup tree-kills until that close. +function observeProcessClose(child) { + let closed = false + const completion = new Promise(resolve => child.once('close', () => { + closed = true + resolve() + })) + // Windows descendants can inherit pipe handles and postpone 'close' after + // the launch process exits. Release our handles, never kill descendants. + const releasePipes = () => { + for (const stream of child.stdio) stream?.destroy() + } + child.once('exit', releasePipes) + if (child.exitCode !== null || child.signalCode !== null) releasePipes() + return async function waitForClose(timeoutMs = 120_000) { + if (closed) return + let timer + try { + await Promise.race([ + completion, + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error('Electron process did not close after update hand-off')), timeoutMs) + }), + ]) + } finally { + clearTimeout(timer) + } + } +} + +module.exports = { observeProcessClose } diff --git a/tests/install/e2e-assets/window-input.cjs b/tests/install/e2e-assets/window-input.cjs new file mode 100644 index 0000000000..5fbbb76355 --- /dev/null +++ b/tests/install/e2e-assets/window-input.cjs @@ -0,0 +1,44 @@ +// Shared input setup for the install drivers. Only the selected app window +// is changed; helper windows retain their own coordinate system. +async function prepareWindowForInput(app, page) { + const window = await app.browserWindow(page) + // Use the same persistent setting as Appearance. A bare setZoomLevel is + // overwritten by the app's focus/navigation handlers restoring saved zoom. + const persistent = await page.evaluate(() => { + const zoom = globalThis.hermesDesktop?.zoom + if (!zoom?.setPercent || !zoom?.get) return false + zoom.setPercent(100) + return true + }) + if (persistent) { + // Playwright 1.58 treats an async waitForFunction predicate's Promise as + // truthy even when it resolves false. Await each IPC read on the driver. + const deadline = Date.now() + 15_000 + for (;;) { + const state = await page.evaluate(() => { + // Cold-start restoration can overwrite the first preference write. + // Reapply through its owner until a subsequent read observes it. + globalThis.hermesDesktop.zoom.setPercent(100) + return globalThis.hermesDesktop.zoom.get() + }) + // The renderer IPC and BrowserWindow can observe different moments of + // startup restoration. Both must agree before the driver sends input. + const factor = await window.evaluate(win => win.webContents.getZoomFactor()) + if (state.percent === 100 && Math.abs(factor - 1) < 0.001) return + if (Date.now() >= deadline) { + throw new Error(`timed out waiting for 100% app window zoom (IPC ${state.percent}%, factor ${factor})`) + } + await page.waitForTimeout(100) + } + } else { + // Older sampled releases have no zoom preference bridge. + await window.evaluate(win => win.webContents.setZoomLevel(0)) + } + // DPR includes OS display scaling; 100% page zoom is not always DPR 1. + const factor = await window.evaluate(win => win.webContents.getZoomFactor()) + if (Math.abs(factor - 1) > 0.001) { + throw new Error(`could not set app window zoom to 100% (factor ${factor})`) + } +} + +module.exports = { prepareWindowForInput } diff --git a/tests/install/installer-script-e2e.sh b/tests/install/installer-script-e2e.sh index 7aececdffa..2c028ce366 100755 --- a/tests/install/installer-script-e2e.sh +++ b/tests/install/installer-script-e2e.sh @@ -397,7 +397,7 @@ case "$UPDATE_METHOD" in (cd "$PW_DIR" && npm install --no-save --no-audit --no-fund \ "@playwright/test@1.58.2" 2>&1 | ts_prefix > "$LOG_DIR/playwright-install.log") \ || { log_group "playwright install transcript" "$LOG_DIR/playwright-install.log"; fail "playwright install failed"; } - cp "$ASSETS/launch-from-spec.mjs" "$PW_DIR/" + cp "$ASSETS/launch-from-spec.mjs" "$ASSETS/window-input.cjs" "$PW_DIR/" rc=0 (cd "$PW_DIR" && node launch-from-spec.mjs \ --spec "$SPEC" \ @@ -464,6 +464,24 @@ case "$UPDATE_METHOD" in ok "node_modules cleared for the head desktop smoke" ;; esac + +# Install-side state BEFORE the post-update assertions: on app-update legs +# the updater's transcript is streamed into the app UI (or runs detached) +# and is otherwise lost, so snapshot every place it also lands — product +# logs, update hand-off files, the venv's entry-point dir — while the +# install is still there to inspect. The assertions below can `fail` out +# of the driver; the evidence must already be on disk when they do. +ildest="$LOG_DIR/install-logs" +mkdir -p "$ildest" +cp -R "$HERMES_HOME/logs" "$ildest/hermes-logs" 2>/dev/null || true +if [ -n "${XDG_DATA_HOME:-}" ]; then + cp -R "$XDG_DATA_HOME/hermes/logs" "$ildest/desktop-userdata-logs" 2>/dev/null || true +fi +cp "$HERMES_HOME/.hermes-update-result.json" "$ildest" 2>/dev/null || true +ls -la "$HERMES_HOME" > "$ildest/hermes-home-ls.txt" 2>/dev/null || true +ls -la "$INSTALL_DIR/venv/bin" > "$ildest/venv-bin-ls.txt" 2>/dev/null || true +ok "collected install-side logs to $ildest" + assert_checkout "$HEAD_SHA" HEAD smoke_desktop head diff --git a/tests/install/macos-desktop-e2e.sh b/tests/install/macos-desktop-e2e.sh index 3467b0b811..da74976971 100755 --- a/tests/install/macos-desktop-e2e.sh +++ b/tests/install/macos-desktop-e2e.sh @@ -306,7 +306,7 @@ run_playwright_update() { local spec="$1" local pw_dir pw_dir="$(ensure_playwright)" - cp "$ASSETS/launch-from-spec.mjs" "$pw_dir/" + cp "$ASSETS/launch-from-spec.mjs" "$ASSETS/window-input.cjs" "$pw_dir/" local rc=0 (cd "$pw_dir" && node launch-from-spec.mjs \ --spec "$spec" \ @@ -405,6 +405,24 @@ PYEOF got="$(git -C "$INSTALL_DIR" rev-parse HEAD)" [ "$got" = "$HEAD_SHA" ] || fail "checkout is $got, expected HEAD ($HEAD_SHA)" ok "checkout landed on HEAD ($HEAD_SHA)" + + # Install-side state BEFORE the post-update smoke: on app-update legs the + # updater's own transcript is streamed into the app UI and otherwise lost, + # so snapshot every place it also lands (product logs, update hand-off + # files, the venv's entry-point dir) while the install is still there to + # inspect — the smoke assertion below can `fail` out of the driver, and the + # evidence must already be on disk when it does. + local ildest="$LOG_DIR/install-logs" + mkdir -p "$ildest" + cp -R "$HOME_SANDBOX/.hermes/logs" "$ildest/hermes-logs" 2>/dev/null || true + local ud="$HOME_SANDBOX/Library/Application Support/Hermes" + [ -d "$ud" ] && cp -R "$ud" "$ildest/desktop-userdata" 2>/dev/null || true + cp "$HERMES_HOME/.hermes-update-result.json" "$ildest" 2>/dev/null || true + ls -la "$HERMES_HOME" > "$ildest/hermes-home-ls.txt" 2>/dev/null || true + ls -la "$INSTALL_DIR/venv/bin" > "$ildest/venv-bin-ls.txt" 2>/dev/null || true + ls -la "$INSTALL_DIR/venv" > "$ildest/venv-ls.txt" 2>/dev/null || true + ok "collected install-side logs to $ildest" + "$INSTALL_DIR/venv/bin/hermes" --version 2>&1 | ts_prefix > "$LOG_DIR/version-head.log" \ || fail "hermes --version failed after update" ok "hermes --version works post-update" diff --git a/tests/install/windows-e2e.ps1 b/tests/install/windows-e2e.ps1 index 2d845a3238..faa45500f4 100644 --- a/tests/install/windows-e2e.ps1 +++ b/tests/install/windows-e2e.ps1 @@ -56,8 +56,9 @@ # * A dummy provider key is seeded after install so the update leg sees # the ready app shell instead of the onboarding overlay (a real # updating user has a configured provider). -# * we don't set .skip_upstream_prompt, but we shim `git` so it returns -# the real upstream url for 'git remote get-url origin' +# * The git shim reports the official URL. Detached updaters can resolve a +# different git.exe, so the test home also records that upstream setup was +# declined. The file:// transport must not prompt to add a second remote. # # USAGE (local Windows box or CI): # powershell -File tests\install\windows-e2e.ps1 -Phase all @@ -114,6 +115,11 @@ param( $ErrorActionPreference = "Stop" $ProgressPreference = "SilentlyContinue" +# Match an interactive Unicode console when Python output is piped into the +# UTF-8 transcript. Old releases otherwise select cp1252 and crash on banners. +$env:PYTHONIOENCODING = "utf-8" +[Console]::OutputEncoding = New-Object System.Text.UTF8Encoding $false +$OutputEncoding = [Console]::OutputEncoding if (-not $RepoRoot) { $RepoRoot = (Resolve-Path (Join-Path $PSScriptRoot "..\..")).Path @@ -280,7 +286,37 @@ function Get-DesktopExe { return $null } +# Install-side state snapshot, taken BEFORE Test-HermesRuns can throw: on +# app-update legs the updater runs detached and its transcript lands in the +# product logs and hand-off files, not in this driver. Copy those plus the +# venv entry-point dir while the install is still there to inspect, so a +# failed post-update assertion leaves its evidence in the proof tree. +function Save-InstallSideState([string]$Label) { + $dest = Join-Path $ProofRoot "install-side-$Label" + New-Item -ItemType Directory -Path $dest -Force | Out-Null + $logsDir = Join-Path $HermesHome "logs" + if (Test-Path -LiteralPath $logsDir) { + Copy-Item $logsDir (Join-Path $dest "hermes-logs") -Recurse -Force -ErrorAction SilentlyContinue + } + $resultFile = Join-Path $HermesHome ".hermes-update-result.json" + if (Test-Path -LiteralPath $resultFile) { + Copy-Item $resultFile $dest -Force -ErrorAction SilentlyContinue + } + $venvScripts = Join-Path $InstallDir "venv\Scripts" + if (Test-Path -LiteralPath $venvScripts) { + Get-ChildItem -LiteralPath $venvScripts | + Select-Object Name, Length, LastWriteTime | + Format-Table -AutoSize | Out-String | + Set-Content (Join-Path $dest "venv-scripts-ls.txt") + } + Get-ChildItem -LiteralPath $HermesHome -ErrorAction SilentlyContinue | + Select-Object Name, Length, LastWriteTime | + Format-Table -AutoSize | Out-String | + Set-Content (Join-Path $dest "hermes-home-ls.txt") +} + function Test-HermesRuns([string]$Label) { + Save-InstallSideState $Label $hermesExe = Join-Path $InstallDir "venv\Scripts\hermes.exe" Assert-True (Test-Path -LiteralPath $hermesExe) "$Label -- venv\Scripts\hermes.exe exists" & $hermesExe --version 2>&1 | ForEach-Object { Write-Host " hermes --version| $_" } @@ -350,7 +386,7 @@ function Invoke-HermesUpdate { $ErrorActionPreference = $prevEap } Write-LogGroup "hermes update transcript" $log - Assert-True ($updateExit -eq 0) "hermes update exited 0" + Assert-True ($updateExit -eq 0) "hermes update exited $updateExit (expected 0)" } function Invoke-HermesDesktopAppUpdate([string]$TargetSha) { @@ -399,6 +435,7 @@ function Invoke-HermesDesktopAppUpdate([string]$TargetSha) { Assert-True ($npmExit -eq 0) "npm install @playwright/test@$PlaywrightVersion into the driver dir" Copy-Item (Join-Path $AssetsDir "launch-from-spec.mjs") (Join-Path $driverDir "launch-from-spec.mjs") -Force + Copy-Item (Join-Path $AssetsDir "window-input.cjs") (Join-Path $driverDir "window-input.cjs") -Force $prevEap = $ErrorActionPreference; $ErrorActionPreference = "Continue" Push-Location $driverDir try { @@ -725,13 +762,18 @@ function Invoke-GuiUpdateDesktopRoute([string]$TargetSha) { # node_modules. $driver = Join-Path $driverDir "e2e-drive-update.cjs" Copy-Item (Join-Path $AssetsDir "drive-update.cjs") $driver -Force + Copy-Item (Join-Path $AssetsDir "window-input.cjs") (Join-Path $driverDir "window-input.cjs") -Force + Copy-Item (Join-Path $AssetsDir "process-close.cjs") (Join-Path $driverDir "process-close.cjs") -Force Push-Location $driverDir + $prevEap = $ErrorActionPreference + $ErrorActionPreference = "Continue" try { & $node $driver $desktopExe $proof 2>&1 | ForEach-Object { Write-Host " $_" } $driveExit = $LASTEXITCODE } finally { Pop-Location + $ErrorActionPreference = $prevEap Remove-Item -LiteralPath $driver -Force -ErrorAction SilentlyContinue } Assert-True ($driveExit -eq 0) "GUI driver clicked Update now and the app quit for hand-off" @@ -833,6 +875,7 @@ function Invoke-GuiUpdateDesktopRoute([string]$TargetSha) { Write-Host "::endgroup::" Copy-Item $handoffLog (Join-Path $proof "desktop-update-handoff.log") -Force -ErrorAction SilentlyContinue } + # Quit the relaunched app so job teardown is clean. Stop-HermesAppProcesses "post-update" } @@ -872,6 +915,9 @@ function Invoke-PhaseInstall { function Invoke-PhaseUpdate { $state = Read-State $env:HERMES_HOME = $HermesHome + # Match the POSIX driver's explicit opt-out when a detached updater bypasses + # the PATH shim and sees our local transport as a fork. + New-Item -ItemType File -Path (Join-Path $HermesHome ".skip_upstream_prompt") -Force | Out-Null # The update becomes available the way it does for a real user: the # remote's main moves forward. The GUI route re-advances harmlessly @@ -924,6 +970,32 @@ function Invoke-PhaseUpdate { Test-HermesRuns "post-update" } +function Invoke-CheckedPhaseUpdate { + Remove-Item -LiteralPath (Join-Path $WorkRoot "known-failure.json") -Force -ErrorAction SilentlyContinue + # Only evidence produced by this update attempt can match an exception. + foreach ($oldLog in @((Join-Path $WorkRoot "logs\update.log"), (Join-Path $HermesHome "logs\desktop.log"))) { + if (Test-Path -LiteralPath $oldLog) { Move-Item -LiteralPath $oldLog -Destination "$oldLog.before-update" -Force } + } + try { + Invoke-PhaseUpdate + } catch { + $failure = $_ + $node = Get-ManagedNode + $classification = & $node (Join-Path $AssetsDir "known-failures.cjs") $WorkRoot $InstallMethod $Route $failure.Exception.Message + $classificationExit = $LASTEXITCODE + if ($classificationExit -ne 0) { throw $failure } + $receipt = ($classification | Out-String) | ConvertFrom-Json + Write-Host "KNOWN FAILURE [$($receipt.id)]: $($receipt.title)" + Write-Host " $($receipt.explanation)" + if ($env:GITHUB_OUTPUT) { + Add-Content -LiteralPath $env:GITHUB_OUTPUT -Value "known_failure=$($receipt.id)" -Encoding UTF8 + } + if ($env:GITHUB_STEP_SUMMARY) { + Add-Content -LiteralPath $env:GITHUB_STEP_SUMMARY -Encoding UTF8 -Value "Known historical failure: $($receipt.title). See the result chart footnote and uploaded known-failure.json." + } + } +} + # ---------------------------------------------------------------------------- # Dispatch # ---------------------------------------------------------------------------- @@ -941,11 +1013,11 @@ Set-GitRedirect switch ($Phase) { "stage" { Invoke-PhaseStage } "install" { Invoke-PhaseInstall } - "update" { Invoke-PhaseUpdate } + "update" { Invoke-CheckedPhaseUpdate } "all" { Invoke-PhaseStage Invoke-PhaseInstall - Invoke-PhaseUpdate + Invoke-CheckedPhaseUpdate } }