diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml new file mode 100644 index 0000000000..c641c828a0 --- /dev/null +++ b/.github/actionlint.yaml @@ -0,0 +1,9 @@ +# actionlint knows only GitHub-hosted runner labels. An org admin names the +# larger runners. Each one therefore reads as "unknown runner label" and hides +# the real findings, unless this file declares it. +self-hosted-runner: + labels: + - ubuntu-latest-96-core + - ubuntu-latest-32-core + - ubuntu-latest-32-arm-core + - windows-latest-32-core diff --git a/.github/scripts/run-workspace-checks.mjs b/.github/scripts/run-workspace-checks.mjs new file mode 100644 index 0000000000..eabc479877 --- /dev/null +++ b/.github/scripts/run-workspace-checks.mjs @@ -0,0 +1,141 @@ +// Run every workspace check at the same time and report all failures. +// +// The unit of work is a CHECK, and not a workspace. A package that declares +// `check:*` sub-scripts gives one unit for each sub-script. A package with a +// plain `check` gives that. This is the same selection rule the old CI matrix +// used, so the set of commands is unchanged. Only the schedule is different. +// +// This is not `npm run --ws check`, because that command is serial and stops +// at the first workspace that fails. This runs every unit and fails at the +// end with the full list. +// +// The output of each unit goes to a buffer and prints on completion inside a +// group that collapses. Children that write to one stdout together interleave +// their lines, and a failure is then hard to read. +// +// This also runs on a laptop: `node .github/scripts/run-workspace-checks.mjs`. +// `--concurrency N` sets the limit. `--list` prints the units and exits. + +import { execFileSync, spawn } from 'node:child_process' +import { availableParallelism } from 'node:os' + +const IS_CI = Boolean(process.env.GITHUB_ACTIONS) +const NPM = process.platform === 'win32' ? 'npm.cmd' : 'npm' + +/** @returns {{pkg: string, script: string}[]} */ +function discoverUnits() { + const raw = execFileSync(NPM, ['query', '.workspace'], { + encoding: 'utf-8', + shell: process.platform === 'win32', + }) + /** @type {{location: string, scripts?: Record}[]} */ + const pkgs = JSON.parse(raw) + + /** @type {{pkg: string, script: string}[]} */ + const units = [] + for (const pkg of pkgs) { + const scripts = pkg.scripts || {} + const subs = Object.keys(scripts).filter((s) => /^check:.+$/.test(s)) + if (subs.length > 0) { + for (const script of subs) units.push({ pkg: pkg.location, script }) + } else if (scripts.check) { + units.push({ pkg: pkg.location, script: 'check' }) + } + } + return units +} + +/** @param {{pkg: string, script: string}} unit */ +function runUnit(unit) { + return new Promise((resolve) => { + const started = Date.now() + const child = spawn(NPM, ['run', '--prefix', unit.pkg, unit.script], { + // Buffer, and do not inherit. Children that share one stdout + // interleave their lines, and a failure is then hard to read. + stdio: ['ignore', 'pipe', 'pipe'], + shell: process.platform === 'win32', + }) + /** @type {Buffer[]} */ + const chunks = [] + child.stdout.on('data', (c) => chunks.push(c)) + child.stderr.on('data', (c) => chunks.push(c)) + child.on('error', (err) => { + chunks.push(Buffer.from(`failed to spawn: ${err.message}\n`)) + resolve({ unit, code: 1, output: Buffer.concat(chunks).toString('utf-8'), ms: Date.now() - started }) + }) + child.on('close', (code) => { + resolve({ + unit, + code: code ?? 1, + output: Buffer.concat(chunks).toString('utf-8'), + ms: Date.now() - started, + }) + }) + }) +} + +async function main() { + const argv = process.argv.slice(2) + const units = discoverUnits() + + if (units.length === 0) { + console.error( + '::error::No workspace package declares a check script — refusing to report green having run nothing.', + ) + process.exit(1) + } + + if (argv.includes('--list')) { + for (const u of units) console.log(`${u.pkg} :: ${u.script}`) + return + } + + const flagIdx = argv.indexOf('--concurrency') + const concurrency = Math.max( + 1, + flagIdx !== -1 ? Number(argv[flagIdx + 1]) : Math.min(units.length, availableParallelism()), + ) + + console.log(`running ${units.length} checks, up to ${concurrency} at a time:`) + for (const u of units) console.log(` ${u.pkg} :: ${u.script}`) + console.log('') + + const queue = [...units] + /** @type {{unit: {pkg: string, script: string}, code: number, output: string, ms: number}[]} */ + const results = [] + + async function worker() { + for (;;) { + const unit = queue.shift() + if (!unit) return + const res = await runUnit(unit) + results.push(res) + const label = `${res.unit.pkg} :: ${res.unit.script}` + const secs = (res.ms / 1000).toFixed(1) + const status = res.code === 0 ? 'PASS' : 'FAIL' + if (IS_CI) console.log(`::group::${status} ${label} (${secs}s)`) + else console.log(`----- ${status} ${label} (${secs}s) -----`) + process.stdout.write(res.output.endsWith('\n') ? res.output : res.output + '\n') + if (IS_CI) console.log('::endgroup::') + } + } + + await Promise.all(Array.from({ length: Math.min(concurrency, units.length) }, worker)) + + const failed = results.filter((r) => r.code !== 0) + console.log('\n=== summary ===') + for (const r of [...results].sort((a, b) => b.ms - a.ms)) { + console.log( + ` ${r.code === 0 ? 'pass' : 'FAIL'} ${(r.ms / 1000).toFixed(1).padStart(6)}s ${r.unit.pkg} :: ${r.unit.script}`, + ) + } + + if (failed.length > 0) { + for (const r of failed) console.error(`::error::${r.unit.pkg} :: ${r.unit.script} failed`) + console.error(`::error::${failed.length} of ${results.length} checks failed`) + process.exit(1) + } + console.log(`\nall ${results.length} checks passed`) +} + +await main() diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index e2660cc4a6..caf4e2d3cb 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -38,7 +38,7 @@ jobs: detect: name: Detect affected areas runs-on: ubuntu-latest - timeout-minutes: 10 + timeout-minutes: 1 outputs: python: ${{ steps.classify.outputs.python }} python_prod: ${{ steps.classify.outputs.python_prod }} @@ -61,6 +61,8 @@ jobs: id: classify uses: ./.github/actions/detect-changes with: + sparse-checkout: scripts/ci/classify_changes.py + sparse-checkout-cone-mode: false github-token: ${{ github.token }} # ───────────────────────────────────────────────────────────────────── @@ -72,8 +74,6 @@ jobs: needs: detect if: needs.detect.outputs.python == 'true' uses: ./.github/workflows/tests.yml - with: - slice_count: 12 # macOS + Windows lanes. The main `tests` lane above is Linux-only, and # the OS-marked tests it collects are skipped there by design (see the diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index 8c259fe872..f245708486 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -76,12 +76,14 @@ jobs: matrix: include: - arch: amd64 - runner: ubuntu-latest + runner: ubuntu-latest-32-core platform: linux/amd64 cache-from: type=gha,scope=docker-amd64 cache-to: type=gha,mode=max,scope=docker-amd64 + # arm64 builds on the native arm64 larger runner. A build of + # linux/arm64 on an x64 host uses emulation. - arch: arm64 - runner: ubuntu-24.04-arm + runner: ubuntu-latest-32-arm-core platform: linux/arm64 cache-from: type=gha,scope=docker-arm64 cache-to: type=gha,mode=max,scope=docker-arm64 @@ -169,7 +171,11 @@ jobs: OPENAI_API_KEY: "" NOUS_API_KEY: "" run: | - scripts/run_tests.sh tests/docker/ --file-timeout 600 + # Each of these tests drives a container, so the docker daemon sets + # the limit and not the processor. This caps the workers. The + # default from run_tests.sh is cpu_count*2, which starts 64 + # containers together on the 32-core amd64 runner. + HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600 # --------------------------------------------------------------------------- # Rebuild and push each architecture only after the unprivileged build/test @@ -184,12 +190,13 @@ jobs: matrix: include: - arch: amd64 - runner: ubuntu-latest + runner: ubuntu-latest-32-core platform: linux/amd64 cache-from: type=gha,scope=docker-amd64 cache-to: type=gha,mode=max,scope=docker-amd64 + # Native arm64 for the same reason as the build matrix above. - arch: arm64 - runner: ubuntu-24.04-arm + runner: ubuntu-latest-32-arm-core platform: linux/arm64 cache-from: type=gha,scope=docker-arm64 cache-to: type=gha,mode=max,scope=docker-arm64 diff --git a/.github/workflows/e2e-desktop.yml b/.github/workflows/e2e-desktop.yml index 2749e7c290..8692f05d7a 100644 --- a/.github/workflows/e2e-desktop.yml +++ b/.github/workflows/e2e-desktop.yml @@ -17,7 +17,10 @@ concurrency: jobs: e2e: name: Playwright E2E (Linux) - runs-on: ubuntu-latest + # This job builds the renderer and the electron bundle, then drives a real + # Electron app under xvfb. vite, tsc and the Playwright workers all scale + # with the core count. + runs-on: ubuntu-latest-32-core timeout-minutes: 20 outputs: review_status: ${{ steps.review-status.outputs.review_status }} diff --git a/.github/workflows/js-tests.yml b/.github/workflows/js-tests.yml index 9631cdd7fa..956908293c 100644 --- a/.github/workflows/js-tests.yml +++ b/.github/workflows/js-tests.yml @@ -5,87 +5,17 @@ on: workflow_call: jobs: - workspaces: - name: List npm workspaces - runs-on: ubuntu-latest - timeout-minutes: 20 - outputs: - checks: ${{ steps.set-matrix.outputs.checks }} - steps: - - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - with: - node-version: 26 - cache: npm - - - name: grab npm 12 - run: | - # No-op once the bundled npm is already 12.x — saves ~5-15s/job and - # keeps the installed major aligned with the npm12 cache-key tag. - npm --version | grep -q '^12\.' || npm i -g npm@12 - - # ``setup-node``'s ``cache: npm`` only caches the ~/.npm tarball cache; - # every job still re-extracts the full workspace node_modules and reruns - # postinstalls (including the Electron binary fetch). Cache the installed - # tree itself, keyed on the lockfile, and skip ``npm ci`` on an exact - # hit. No restore-keys: a partial hit would leave a stale tree, so - # anything but an exact lockfile match reinstalls from scratch. - # The discovery job installs with --ignore-scripts, so its tree differs - # from the check jobs' — hence the distinct ``-noscripts`` key. - - name: Restore node_modules - id: node-modules-cache - uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4 - with: - path: | - node_modules - apps/*/node_modules - ui-tui/node_modules - ui-tui/packages/*/node_modules - tests-js/node_modules - web/node_modules - key: node-modules-noscripts-${{ runner.os }}-node26-npm12-${{ hashFiles('package-lock.json') }} - - - uses: ./.github/actions/retry - if: steps.node-modules-cache.outputs.cache-hit != 'true' - with: - command: npm ci --ignore-scripts - - id: set-matrix - run: | - node -e ' - const { execSync } = require("child_process"); - const pkgs = JSON.parse(execSync("npm query .workspace", { encoding: "utf-8" })); - if (pkgs.length === 0) { - console.error("::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."); - process.exit(1); - } - const checks = []; - for (const pkg of pkgs) { - const scripts = pkg.scripts || {}; - const subs = Object.keys(scripts).filter(s => /^check:.+$/.test(s)); - if (subs.length > 0) { - for (const script of subs) { - checks.push({ package: pkg.location, script }); - } - } else if (scripts.check) { - checks.push({ package: pkg.location, script: "check" }); - } - } - if (checks.length === 0) { - console.error("::error::No check scripts found in any workspace package."); - process.exit(1); - } - process.stdout.write("checks=" + JSON.stringify(checks) + "\n"); - ' >> "$GITHUB_OUTPUT" - check: - name: ${{ matrix.package }} / ${{ matrix.script }} - needs: workspaces - runs-on: ubuntu-latest - timeout-minutes: 20 - strategy: - matrix: - include: ${{ fromJson(needs.workspaces.outputs.checks) }} - fail-fast: false # report all failures, not just the first one + name: JS & TS checks + # One 32-core job replaces a 14-leg matrix. The matrix spread about 612s + # of check payload over 4-core runners. It paid about 371s of repeated + # setup to do it: 14 checkouts, 14 node installs, 14 node_modules + # restores. + # + # One larger runner installs one time. vitest, tsc and eslint each size + # their own worker pool from the core count. + runs-on: ubuntu-latest-32-core + timeout-minutes: 30 steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 @@ -99,12 +29,21 @@ jobs: # keeps the installed major aligned with the npm12 cache-key tag. npm --version | grep -q '^12\.' || npm i -g npm@12 - # Same rationale as the discovery job's cache above, but this ``npm ci`` - # runs WITH install scripts, so the tree includes postinstall artifacts - # (electron's postinstall unpacks its binary into node_modules/electron/ - # dist, which lives inside the cached tree — the ~/.cache/electron - # download cache is deliberately NOT cached: with npm ci skipped on hit - # it would never be read, only inflate the archive). + # The ``cache: npm`` option of ``setup-node`` caches only the ~/.npm + # tarball cache. The job then extracts the full workspace node_modules + # again and runs the postinstalls again, which includes the Electron + # binary fetch. This caches the installed tree itself, keyed on the + # lockfile, and skips ``npm ci`` on an exact hit. There are no + # restore-keys: a partial hit leaves a stale tree, so anything other + # than an exact lockfile match reinstalls from the start. + # + # This install runs WITH scripts, so the tree holds the postinstall + # artifacts. The postinstall of electron unpacks its binary into + # node_modules/electron/dist, which is inside the cached tree. + # + # The ~/.cache/electron download cache stays out of the key on purpose. + # ``npm ci`` is skipped on a hit, so nothing reads that cache. It only + # makes the archive larger. - name: Restore node_modules id: node-modules-cache uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4 @@ -122,4 +61,23 @@ jobs: if: steps.node-modules-cache.outputs.cache-hit != 'true' with: command: npm ci - - run: npm run --prefix ${{ matrix.package }} ${{ matrix.script }} + + # Every check runs at the same time. The step fails only after all of + # them finish. There are two reasons this is not ``npm run --ws check``. + # + # * ``--ws`` is serial and stops at the first workspace that fails. A + # run then reports one failure, where the matrix this replaced + # reported every failure together. + # * The unit of work is a CHECK, and not a workspace. apps/desktop is + # most of the payload, and its own ``check`` is a serial && chain. + # A spread across workspaces alone leaves that chain as the long + # pole. This expands the ``check:*`` sub-scripts of a package, so + # its lint, ui, electron and plugin suites all run together. That + # is the same selection rule the old matrix job used. + # + # Discovery is ``npm query .workspace``. A new package or a new + # ``check:*`` script needs no change here. An empty list is an error and + # not an empty run, because an empty run reports green after it checks + # nothing. + - name: Run all workspace checks + run: node .github/scripts/run-workspace-checks.mjs diff --git a/.github/workflows/nix.yml b/.github/workflows/nix.yml index 23627cbc9f..ccd47d1b09 100644 --- a/.github/workflows/nix.yml +++ b/.github/workflows/nix.yml @@ -51,8 +51,10 @@ jobs: needs: [detect] if: needs.detect.outputs.nix == 'true' # The build compiles the package and its whole dependency closure, so this - # is minutes and not seconds when the cache misses. - runs-on: ubuntu-latest + # takes minutes and not seconds when the cache misses. `nix flake check` + # builds 21 checks, and --max-jobs defaults to the core count. It uses the + # wider runner with no more configuration. + runs-on: ubuntu-latest-32-core timeout-minutes: 60 steps: - name: Checkout code diff --git a/.github/workflows/rust-tests.yml b/.github/workflows/rust-tests.yml index 0d6c179f05..5c91e1a352 100644 --- a/.github/workflows/rust-tests.yml +++ b/.github/workflows/rust-tests.yml @@ -27,7 +27,10 @@ concurrency: jobs: bootstrap-installer: name: cargo test (bootstrap installer) - runs-on: ubuntu-latest + # cargo builds codegen units and test binaries in parallel across the + # cores. This lane also builds the crate from the start when Cargo.toml + # changes. + runs-on: ubuntu-latest-32-core timeout-minutes: 30 defaults: run: diff --git a/.github/workflows/tests-os.yml b/.github/workflows/tests-os.yml index 9ac89c20f4..12719a853c 100644 --- a/.github/workflows/tests-os.yml +++ b/.github/workflows/tests-os.yml @@ -16,9 +16,9 @@ name: OS-specific tests # # Deliberately NOT sliced. The marked set is small (tens of tests, not # thousands), so one plain ``pytest`` process per OS is both faster and far -# less machinery than the LPT-sliced per-file runner the Linux lane needs. +# less machinery than the per-file parallel runner the Linux lane uses. # If either lane grows past its timeout, that is the signal to reach for -# scripts/run_tests.sh --slice here too. +# scripts/run_tests.sh here too. # # Each lane FAILS when it selects zero tests (pytest exit code 5). Without # that guard, a renamed marker or a bad selector would report a green job @@ -48,7 +48,7 @@ jobs: runner: macos-latest marker: macos_only - name: Windows-only tests - runner: windows-latest + runner: windows-latest-32-core marker: windows_only steps: - name: Checkout code diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index a79bf08563..7286abf6a8 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -2,11 +2,6 @@ name: Tests on: workflow_call: - inputs: - slice_count: - description: Number of parallel test slices - type: number - default: 8 permissions: contents: read @@ -17,42 +12,17 @@ concurrency: cancel-in-progress: true jobs: - generate: - name: "Generate slices" - runs-on: ubuntu-latest - timeout-minutes: 10 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - steps: - - name: Checkout code - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - - name: Restore duration cache - uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 - with: - path: test_durations.json - key: test-durations - # Saves use test-durations-${run_id}, so the exact key above never - # matches — without this prefix fallback the cache ALWAYS missed, - # LPT slicing ran on no data, and unbalanced slices pushed heavy - # files toward the per-file timeout under load. - restore-keys: | - test-durations- - - - name: Generate test slices - id: matrix - run: | - MATRIX=$(python3 scripts/run_tests_parallel.py --generate-slices ${{ inputs.slice_count }}) - echo "matrix=$MATRIX" >> "$GITHUB_OUTPUT" - test: - name: Run tests slice ${{ matrix.slice.index }}/${{ inputs.slice_count }} - needs: generate - runs-on: ubuntu-latest + name: Run tests + # One 96-core runner for the whole suite. There is no slicing. Slicing + # existed to spread the suite over 4-core runners. It cost a matrix job, a + # duration cache, a per-slice artifact and a merge job to do it. + # + # 96 cores clear the floor that the slowest single test file sets (about + # 82s). A second slice divides work that is already at that floor, and + # adds a second setup. + runs-on: ubuntu-latest-96-core timeout-minutes: 30 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.generate.outputs.matrix) }} steps: - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 @@ -117,73 +87,30 @@ jobs: # re-download, keeping the persisted cache small and fast to restore. run: uv cache prune --ci - - name: Run tests (slice ${{ matrix.slice.index }}/${{ inputs.slice_count }}) + - name: Run tests # Per-file isolation via scripts/run_tests.sh: each test file runs # in its own freshly-spawned `python -m pytest ` subprocess # with bounded parallelism. No xdist, no shared workers, no # module-level state leakage between files. # - # File list is pre-computed by the generate job (--generate-slices) - # which runs LPT distribution once and passes the file list to each - # matrix job via --files. Previously each job re-discovered files and - # re-ran LPT independently — redundant N times. + # No --files: the runner discovers the suite itself. The discovered + # set is identical to the list the removed matrix job used to pass in. run: | source .venv/bin/activate - scripts/run_tests.sh --files '${{ matrix.slice.files }}' + scripts/run_tests.sh env: + # This is the maximum number of test FILES that run together. + # run_tests_parallel.py starts one pytest subprocess for each file + # from a single ThreadPoolExecutor, so this value IS the limit. The + # default is cpu_count*2, which is 192 here. + # + # A later commit sets this from a sweep on the real runner. + HERMES_TEST_WORKERS: 144 # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" NOUS_API_KEY: "" - - name: Upload per-slice durations - # Advisory artifact (feeds slice balancing) — a transient artifact- - # service blip must not fail an otherwise-green test slice. - continue-on-error: true - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: test-durations-slice-${{ matrix.slice.index }} - path: test_durations.json - retention-days: 1 - - # Merge per-slice duration data into a single cache, so future runs - # (including PRs) get balanced slicing. - save-durations: - needs: test - if: needs.test.result == 'success' && github.ref == 'refs/heads/main' - runs-on: ubuntu-latest - timeout-minutes: 10 - steps: - - name: Download all slice durations - # Each slice uploads the same file name (test_durations.json). - # With merge-multiple, the parallel downloads write to one path. - # This causes two problems: a race can write two JSON documents - # into one file, and the last write erases the other slices. - # Without merge-multiple, each artifact gets its own directory. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 - with: - pattern: test-durations-slice-* - path: durations - - - name: Merge into single durations file - run: | - python3 -c " - import json, glob, os - merged = {} - for f in glob.glob('durations/*/test_durations.json'): - with open(f) as fh: - merged.update(json.load(fh)) - with open('test_durations.json', 'w') as fh: - json.dump(merged, fh, indent=2, sort_keys=True) - print(f'Merged {len(merged)} file durations') - " - - - name: Save merged duration cache - uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 - with: - path: test_durations.json - key: test-durations-${{ github.run_id }} - e2e: runs-on: ubuntu-latest timeout-minutes: 15 diff --git a/apps/desktop/package.json b/apps/desktop/package.json index ef1a9f6d16..73a71ed35a 100644 --- a/apps/desktop/package.json +++ b/apps/desktop/package.json @@ -66,14 +66,12 @@ "test:find-in-page-native": "electron electron/find-in-page-native-fixture", "test": "vitest run", "preview": "node scripts/assert-root-install.mjs && vite preview --host 127.0.0.1 --port 4174", + "check:test:ui": "npm run test:ui", "check:test:desktop:platforms": "npm run test:desktop:platforms", - "check:test:plugins": "node --test src/plugins/*/tests/*.test.mjs", - "check:test:ui:shard-1of3": "node scripts/run-ui-shard.mjs", - "check:test:ui:shard-2of3": "node scripts/run-ui-shard.mjs", - "check:test:ui:shard-3of3": "node scripts/run-ui-shard.mjs", "check:test:desktop:all": "npm run test:desktop:all", + "check:test:plugins": "node --test src/plugins/*/tests/*.test.mjs", "check:lint": "npm run typecheck && npm run lint", - "check": "npm run check:lint && npm run test:ui && npm run test:desktop:platforms && npm run test:desktop:all", + "check": "npm run check:lint && npm run test:ui && npm run test:desktop:platforms && npm run test:desktop:all && npm run check:test:plugins", "test:e2e": "npm run build && playwright test e2e/", "test:e2e:visual": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list", "test:e2e:update-snapshots": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list --update-snapshots", diff --git a/apps/desktop/scripts/run-ui-shard.mjs b/apps/desktop/scripts/run-ui-shard.mjs deleted file mode 100644 index 8f29f16f62..0000000000 --- a/apps/desktop/scripts/run-ui-shard.mjs +++ /dev/null @@ -1,61 +0,0 @@ -// Runs one shard of the UI vitest suite, deriving the shard index/count from -// the npm script NAME (npm_lifecycle_event), so the name and the flag can -// never disagree. A copy-paste slip like "check:test:ui:shard-2of3" running -// --shard=1/3 would silently skip a third of the suite while CI stays green; -// deriving from the name makes that impossible. -// -// It also validates that this package.json declares exactly the shard family -// 1..M for a single M, so a partial 3→4 migration (adding shard-4of4 without -// updating the siblings) fails loudly instead of dropping coverage. -import { spawnSync } from 'node:child_process' -import { readFileSync } from 'node:fs' -import { dirname, join } from 'node:path' -import { fileURLToPath } from 'node:url' - -const SHARD_RE = /^check:test:ui:shard-(\d+)of(\d+)$/ - -const scriptName = process.env.npm_lifecycle_event ?? '' -const match = scriptName.match(SHARD_RE) -if (!match) { - console.error( - `run-ui-shard: must be invoked via an npm script named check:test:ui:shard-of (got ${JSON.stringify(scriptName)})`, - ) - process.exit(1) -} -const [, indexRaw, countRaw] = match -const index = Number(indexRaw) -const count = Number(countRaw) -if (!(index >= 1 && index <= count)) { - console.error(`run-ui-shard: shard index ${index} out of range 1..${count}`) - process.exit(1) -} - -// The whole family must be exactly 1..M of one M — otherwise a rename or a -// partial count bump leaves a silently untested slice of the suite. -const pkgDir = dirname(dirname(fileURLToPath(import.meta.url))) -const pkg = JSON.parse(readFileSync(join(pkgDir, 'package.json'), 'utf8')) -const family = Object.keys(pkg.scripts ?? {}) - .map((name) => name.match(SHARD_RE)) - .filter(Boolean) -const counts = new Set(family.map((m) => Number(m[2]))) -const indices = family.map((m) => Number(m[1])).sort((a, b) => a - b) -const expected = Array.from({ length: count }, (_, i) => i + 1) -if (counts.size !== 1 || indices.length !== count || indices.some((v, i) => v !== expected[i])) { - console.error( - `run-ui-shard: shard scripts must form exactly 1..M for a single M; found indices [${indices}] with counts {${[...counts]}}`, - ) - process.exit(1) -} - -// Delegate through test:ui so the vitest command stays single-sourced. -// npm resolves to npm.cmd on Windows, which needs a shell (same handling as -// test-desktop.mjs and stage-native-deps.mjs). -const result = spawnSync( - 'npm', - ['run', 'test:ui', '--', `--shard=${index}/${count}`, ...process.argv.slice(2)], - { stdio: 'inherit', cwd: pkgDir, shell: process.platform === 'win32' }, -) -if (result.error) { - console.error(`run-ui-shard: ${result.error.message}`) -} -process.exit(result.status ?? 1)