Files
hermes-agent/.github/workflows/ci.yaml
ethernet b0ab0162b0 feat(release): gate stable promotion through the full release pipeline
Run the entire CI workflow before Docker build and tests. Require Nix,
native payload smoke tests, install/update E2E and signed-package upgrade
acceptance before publishing. Keep Desktop Playwright E2E deferred.

Archive tested Docker images and signed bundle candidates with provenance
and hashes. Publishers consume those exact artifacts without rebuilding.
Advance stable channels only after all required publications succeed.
Keep canaries on their separate path and reject direct stable-builder
publication that bypasses the gate.

Move shared release transport, manifests and gates to Python. Keep native
Electron adapters in JS and share feed/MIME facts as JSON. Replace the
R2/feed JS implementation and move its protocol tests to Python.

Verified targeted Python and JS tests, real loopback transport and CLI
execution, temporary Git admission, workflow graph lint, and typechecks.
No live stable release was run. Native signing, package upgrades and real
registry/Store promotion still need their release-run receipts. Separate
services cannot promote atomically. A promotion failure keeps the run red.
2026-09-07 14:40:10 -04:00

451 lines
19 KiB
YAML

name: CI
# Orchestrator workflow. Runs ``detect-changes`` once, then conditionally
# calls the sub-workflows that a PR can actually affect. A final
# ``all-checks-pass`` gate job aggregates results so branch protection only
# needs to require a single check.
#
# Sub-workflows are triggered via ``workflow_call`` and keep their own job
# definitions, matrices, and concurrency settings. They no longer have
# ``push:`` / ``pull_request:`` triggers of their own — everything flows
# through this file.
#
# SECURITY: this workflow runs PR-controlled actions, workflows, and code.
# Do not add ``secrets: inherit`` or GitHub App credentials here. Trusted
# main-only automation uses protected environments in its own workflows.
on:
workflow_dispatch:
inputs:
release:
description: 'Stable-release candidate run: force every applicability lane and make the aggregate gate strict (skipped required lanes fail).'
required: false
type: boolean
default: false
workflow_call:
inputs:
release:
description: 'Stable-release candidate run: force every applicability lane and make the aggregate gate strict (skipped required lanes fail).'
required: false
type: boolean
default: false
pull_request:
push:
branches: [main]
permissions:
contents: read
pull-requests: write # needed by lint (PR comment) + supply-chain review_status
actions: read # needed by osv-scanner (SARIF upload)
security-events: write # needed by osv-scanner (SARIF upload)
# cancel-in-progress only ever applies to PR events. A push, a dispatch, and
# above all a stable-release workflow_call run are never cancelled by a later
# commit or a rerun of the same branch — the caller's own concurrency uses
# github.run_id, so even a parent rerun cannot kill this child mid-flight.
# Release (workflow_call with inputs.release) never cancels and never gets
# cancelled: its group is github.run_id. PRs collapse per-PR; pushes use the
# ref as before.
concurrency:
group: ci-${{ inputs.release == true && github.run_id || (github.event_name == 'pull_request' && github.event.pull_request.number || github.ref) }}
cancel-in-progress: ${{ inputs.release != true && github.event_name == 'pull_request' }}
jobs:
# ─────────────────────────────────────────────────────────────────────
# detect: run the classifier once. Every downstream job reads its outputs
# to decide whether to run. On push/dispatch the classifier fails open
# (all lanes true) so post-merge validation is never weakened.
#
# A release run (inputs.release) additionally forces every lane true via
# the `release-forced-*` outputs: a stable candidate must run the FULL
# pipeline, not the lanes its diff would touch — the diff of a release
# tag is not a meaningful applicability signal.
# ─────────────────────────────────────────────────────────────────────
detect:
name: Detect affected areas
runs-on: ubuntu-latest
timeout-minutes: 1
outputs:
python: ${{ steps.gate-lanes.outputs.python }}
python_prod: ${{ steps.gate-lanes.outputs.python_prod }}
frontend: ${{ steps.gate-lanes.outputs.frontend }}
site: ${{ steps.gate-lanes.outputs.site }}
scan: ${{ steps.gate-lanes.outputs.scan }}
deps: ${{ steps.gate-lanes.outputs.deps }}
uv_lock: ${{ steps.gate-lanes.outputs.uv_lock }}
npm_lock: ${{ steps.gate-lanes.outputs.npm_lock }}
installer: ${{ steps.gate-lanes.outputs.installer }}
bootstrap: ${{ steps.gate-lanes.outputs.bootstrap }}
desktop_updater: ${{ steps.gate-lanes.outputs.desktop_updater }}
rust: ${{ steps.gate-lanes.outputs.rust }}
docker_meta: ${{ steps.gate-lanes.outputs.docker_meta }}
mcp_catalog: ${{ steps.gate-lanes.outputs.mcp_catalog }}
ci_review: ${{ steps.classify.outputs.ci_review }}
ci_review_files: ${{ steps.classify.outputs.ci_review_files }}
event_name: ${{ github.event_name }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Detect affected areas
id: classify
uses: ./.github/actions/detect-changes
with:
github-token: ${{ github.token }}
- name: Force all lanes on release
# Overwrite the classifier outputs with 'true' when inputs.release.
# ci_review stays raw: it only ever feeds the PR review-label gate.
id: gate-lanes
env:
RELEASE: ${{ inputs.release }}
CLASSIFIED: ${{ toJSON(steps.classify.outputs) }}
run: |
python3 - <<'PY'
import json, os
values = json.loads(os.environ['CLASSIFIED'])
with open(os.environ['GITHUB_OUTPUT'], 'a', encoding='utf-8') as output:
for lane, value in values.items():
if os.environ['RELEASE'] == 'true' and value in ('true', 'false'):
value = 'true'
if '\n' in value or '\r' in value:
continue
output.write(f'{lane}={value}\n')
PY
# ─────────────────────────────────────────────────────────────────────
# Lane-gated sub-workflows. Each runs in parallel after detect finishes.
# Skipped workflows (if condition is false) don't spin up runners.
# ─────────────────────────────────────────────────────────────────────
tests:
name: Python tests
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests.yml
# macOS + Windows lanes. The main `tests` lane above is Linux-only, and
# the OS-marked tests it collects are skipped there by design (see the
# `_OS_MARKS` comment in tests/conftest.py) — this is where they run.
# Same `python` lane gate: if no Python changed, neither runs.
tests-os:
name: OS-specific tests
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests-os.yml
with:
# The Windows lane spawns the real desktop-update hand-off script
# (tests/test_desktop_update_windows_*.py) only when that surface
# changed; unit-level windows_only tests always run.
desktop_updater: ${{ needs.detect.outputs.desktop_updater == 'true' }}
lint:
name: Python lints
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/lint.yml
with:
event_name: ${{ needs.detect.outputs.event_name }}
js-tests:
name: JS & TS checks
needs: detect
if: needs.detect.outputs.frontend == 'true'
uses: ./.github/workflows/js-tests.yml
installer-tests:
name: Installer tests
needs: detect
# Windows-only, and only for PRs that touch install.ps1 or its tests.
if: needs.detect.outputs.installer == 'true'
uses: ./.github/workflows/installer-tests.yml
rust-tests:
name: Rust tests
needs: detect
# Only for PRs that touch a Rust crate. `.rs` is under apps/, so these
# changes used to run the TypeScript matrix and nothing that compiles them.
if: needs.detect.outputs.rust == 'true'
uses: ./.github/workflows/rust-tests.yml
bootstrap-installer:
name: Bootstrap installer
needs: detect
# The bootstrap-installer path: install.sh, the pin fragments it embeds,
# and the version stamp it ships. The PowerShell installer has its own
# Windows-only lane above (installer-tests).
if: needs.detect.outputs.bootstrap == 'true'
uses: ./.github/workflows/bootstrap-installer.yml
e2e-desktop:
name: Desktop E2E
needs: detect
# python_prod (not python): the Playwright suite exercises the built app
# + `hermes serve` backend, which never import anything under tests/.
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
# single job in the workflow — while still running the full pytest lanes.
#
# Re-disabled (Sep 2026): the Sep 1 re-enable is still incredibly flaky.
# Keep this a bare `if: false`. The earlier
# `${{ false && (... || ...) }}` form on this reusable-workflow job made
# GitHub's workflow parser fail at startup ("An unexpected error has
# occurred") — every ci.yaml run repo-wide dispatched 0 jobs from
# 24f5a60ed1 until this line changed. To re-enable, restore:
# if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }}
if: false
uses: ./.github/workflows/e2e-desktop.yml
docs-site:
name: Docs Site
needs: detect
if: needs.detect.outputs.site == 'true'
uses: ./.github/workflows/docs-site-checks.yml
history-check:
name: Deny unrelated histories
needs: detect
if: needs.detect.outputs.event_name == 'pull_request'
uses: ./.github/workflows/history-check.yml
contributor-check:
name: Check contributors
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/contributor-check.yml
uv-lockfile:
name: Check uv.lock
needs: detect
# Gated: `uv lock --check` re-resolves the whole dependency graph against
# PyPI, so on every PR it spent a network round-trip — and, on a registry
# blip, a blocking red X — for diffs that cannot desync the lockfile
# (docs, frontend, prose). Only pyproject.toml / uv.lock can. A
# `.github/` change still forces it on via the classifier's fail-open.
if: needs.detect.outputs.uv_lock == 'true'
uses: ./.github/workflows/uv-lockfile-check.yml
infographic-check:
name: Check no committed infographics
needs: detect
uses: ./.github/workflows/infographic-check.yml
profile-artifact-check:
name: Profile artifact check
needs: detect
uses: ./.github/workflows/profile-artifact-check.yml
icons-freshness-check:
name: Icon assets freshness
needs: detect
uses: ./.github/workflows/icons-freshness-check.yml
case-collision-check:
name: Check no case-colliding filenames
needs: detect
uses: ./.github/workflows/case-collision-check.yml
lazy-deps-guard:
name: No imports of deleted tools.lazy_deps
needs: detect
uses: ./.github/workflows/lazy-deps-guard.yml
lockfile-diff:
name: package-lock.json diff
needs: detect
if: needs.detect.outputs.event_name == 'pull_request' && needs.detect.outputs.npm_lock == 'true'
uses: ./.github/workflows/lockfile-diff.yml
docker-lint:
name: Lint Docker scripts
needs: detect
if: needs.detect.outputs.docker_meta == 'true'
uses: ./.github/workflows/docker-lint.yml
supply-chain:
name: Supply-chain scan
needs: detect
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.scan == 'true' || needs.detect.outputs.deps == 'true')
uses: ./.github/workflows/supply-chain-audit.yml
with:
event_name: ${{ needs.detect.outputs.event_name }}
scan: ${{ needs.detect.outputs.scan == 'true' }}
deps: ${{ needs.detect.outputs.deps == 'true' }}
review-labels:
name: Review label gate
needs: [detect, supply-chain]
if: always() && needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.ci_review == 'true' || needs.detect.outputs.mcp_catalog == 'true' || needs.supply-chain.outputs.critical_findings == 'true')
uses: ./.github/workflows/review-labels.yml
with:
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
ci_review_files: ${{ needs.detect.outputs.ci_review_files }}
mcp_catalog: ${{ needs.detect.outputs.mcp_catalog == 'true' }}
supply_chain: ${{ needs.supply-chain.outputs.critical_findings == 'true' }}
osv-scanner:
name: OSV scan
uses: ./.github/workflows/osv-scanner.yml
# ─────────────────────────────────────────────────────────────────────
# Gate: runs after everything. ``if: always()`` ensures it reports a
# status even when some deps were skipped.
#
# Non-release (PR/push): failure fails, skipped counts as success.
# Release (inputs.release): strict — every required job must be
# `success`. A skipped required lane fails the gate; only the PR-only
# jobs (history-check, lockfile-diff, supply-chain, review-labels) and
# the deferred Desktop E2E (e2e-desktop) may skip. The OSV scan's
# findings are advisory, but its execution is required.
#
# Branch protection should require ONLY this check.
#
# Outputs ``needs-json`` — a compact ``{job_name: result}`` dict — so
# the live comment poller can list failed jobs in the PR comment.
# ─────────────────────────────────────────────────────────────────────
all-checks-pass:
name: All required checks pass
needs:
- detect
- tests
- tests-os
- lint
- js-tests
- installer-tests
- rust-tests
- bootstrap-installer
- e2e-desktop
- docs-site
- history-check
- contributor-check
- uv-lockfile
- infographic-check
- case-collision-check
- lazy-deps-guard
- lockfile-diff
- docker-lint
- profile-artifact-check
- icons-freshness-check
- supply-chain
- review-labels
- osv-scanner
# The image build runs in its own workflow (docker.yml) and reports
# its own check. It was never required here, because it is too slow
# to block a merge. A separate run also stops it from holding this
# run open. That is what blocked ``gh run rerun``.
if: always()
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
needs-json: ${{ steps.evaluate.outputs.needs-json }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ github.sha }}
- name: Evaluate job results
id: evaluate
# Shared with the stable orchestrator (scripts/ci/required_results.py
# is imported there). NEEDS is the toJSON(needs) context; RELEASE
# switches the gate into strict mode for release runs.
env:
NEEDS: ${{ toJSON(needs) }}
RELEASE: ${{ inputs.release }}
run: |
args=()
if [ "$RELEASE" = true ]; then args+=(--release); fi
printf '%s' "$NEEDS" | python3 scripts/ci/required_results.py "${args[@]}"
# ─────────────────────────────────────────────────────────────────────
# CI timing report: collect per-job/step durations from the GitHub API,
# cache them on main (as a baseline), and on PRs generate an HTML diff
# report with a gantt chart + per-step breakdown. The report is uploaded
# as an artifact and a markdown summary is written to $GITHUB_STEP_SUMMARY.
#
# The live comment poller dynamically fetches all review-status-* artifacts
# across the orchestrator and sub-workflow runs every cycle, so its link
# points straight at that report.
# ─────────────────────────────────────────────────────────────────────
ci-timings:
name: CI timing report
needs: [all-checks-pass]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Restore baseline cache (PR only)
if: github.event_name == 'pull_request'
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
with:
path: ci-timings-baseline.json
# Prefix-match: exact key will never hit (run_id differs), so
# restore-keys finds the most recent baseline from main.
key: ci-timings-baseline-never-exact
restore-keys: |
ci-timings-baseline-
- name: Collect timings and generate report
env:
GITHUB_TOKEN: ${{ github.token }}
run: |
python3 scripts/ci/timings_report.py \
--baseline ci-timings-baseline.json \
--output ci-timings-report.html \
--json-out ci-timings.json \
--summary-out ci-timings-summary.md
- name: Upload HTML report
# Advisory report — artifact-service blips must not fail the job.
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
id: ci-timings-html
with:
name: ci-timings-report
path: ci-timings-report.html
retention-days: 14
- name: Build linked review status
if: hashFiles('ci-timings.json') != ''
env:
CI_TIMINGS_REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url }}
run: |
python3 scripts/ci/timings_report.py \
--from-json ci-timings.json \
--baseline ci-timings-baseline.json \
--review-status-out review-status.json \
--review-status-only
- name: Upload review status
if: hashFiles('review-status.json') != ''
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-ci-timings
path: review-status.json
retention-days: 14
- name: Output summary
env:
REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url}}
run: |
{
echo "# CI Timing report"
echo "[View the full interactive report]($REPORT_URL)"
} >> "$GITHUB_STEP_SUMMARY"
cat ci-timings-summary.md >> "$GITHUB_STEP_SUMMARY"
- name: Save baseline cache (main only)
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
run: |
# Degraded runs (API rate-limited) produce no ci-timings.json —
# skip rather than fail, and never cache an empty baseline.
if [ -f ci-timings.json ]; then
cp ci-timings.json ci-timings-baseline.json
else
echo "No timings JSON this run — skipping baseline update"
fi
- name: Upload baseline to cache (main only)
if: github.event_name == 'push' && github.ref == 'refs/heads/main' && hashFiles('ci-timings-baseline.json') != ''
uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
with:
path: ci-timings-baseline.json
key: ci-timings-baseline-${{ github.run_id }}