Files
hermes-agent/tests/ci/test_classify_changes.py
Brooklyn Nicholson 34833303f5 ci(install): actually run the PowerShell installer tests
scripts/tests/ has held three PowerShell suites that no workflow ever
invoked -- there is no Windows runner in CI, so they have been inert since
they landed. A regression test nothing executes is worse than none: it
reads as coverage.

Adds a windows-latest job, gated on a new `installer` lane so it only fires
for PRs touching install.ps1 or its tests. The 8.3 suite runs under both
pwsh 7 and Windows PowerShell 5.1, since install.ps1 arrives via `irm | iex`
into whichever shell the user already has and 5.1 is what ships with Windows.

Only the 8.3 suite is wired up. The other two fail on main today for
unrelated reasons; they can join once they are fixed.
2026-08-04 15:34:32 -06:00

176 lines
7.0 KiB
Python

"""Tests for scripts/ci/classify_changes.py.
Check some common patterns of file modifications and the CI lanes they should run.
We should always fail open. We may run a lane we didn't need, never skip one a
change could have broken.
"""
from __future__ import annotations
import importlib.util
from pathlib import Path
import pytest
_PATH = Path(__file__).resolve().parents[2] / "scripts" / "ci" / "classify_changes.py"
_spec = importlib.util.spec_from_file_location("classify_changes", _PATH)
if _spec is None or _spec.loader is None:
raise ImportError("Failed to load classify_changes.py")
_mod = importlib.util.module_from_spec(_spec)
_spec.loader.exec_module(_mod)
classify = _mod.classify
ci_review_files = _mod.ci_review_files
DEFAULT = {
"python": True,
"python_prod": True,
"frontend": True,
"docker_meta": True,
"site": True,
"scan": True,
"deps": True,
"npm_lock": True,
"installer": True,
"mcp_catalog": False,
"ci_review": True,
}
def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, npm_lock=False, installer=False, mcp_catalog=False, docker_meta=False, ci_review=False, python_prod=None) -> dict[str, bool]:
# python_prod tracks python except for tests-only diffs; default it to
# python so the majority of cases don't need to spell it out.
return {
"python": python,
"python_prod": python if python_prod is None else python_prod,
"frontend": frontend,
"docker_meta": docker_meta,
"site": site,
"scan": scan,
"deps": deps,
"npm_lock": npm_lock,
"installer": installer,
"mcp_catalog": mcp_catalog,
"ci_review": ci_review,
}
CASES = {
"docs-only → nothing heavy": (["README.md", "docs/guide.md"], _lanes()),
"python source → python": (["run_agent.py"], _lanes(python=True, scan=True)),
"dep manifest → python": (["pyproject.toml"], _lanes(python=True, scan=True, deps=True)),
"uv.lock → python": (["uv.lock"], _lanes(python=True)),
"ts package → frontend": (["apps/desktop/src/app.tsx"], _lanes(frontend=True)),
"ui-tui → frontend": (["ui-tui/src/entry.ts"], _lanes(frontend=True)),
# Lockfile bump shifts every TS package's tree, but not the Python suite.
"root lockfile → frontend, not python": (["package-lock.json"], _lanes(frontend=True, npm_lock=True)),
"nested lockfile → npm_lock": (["website/package-lock.json"], _lanes(site=True, npm_lock=True)),
"website → site": (["website/docs/intro.md"], _lanes(site=True)),
# SKILL.md reads like docs, but the skill-doc tests read skills/, so a
# skill edit must still run Python.
"skill md → python + site": (["skills/github/SKILL.md"], _lanes(python=True, site=True)),
"dockerfile → docker meta": (["Dockerfile"], _lanes(docker_meta=True)),
# install.ps1 is a shell script Python never imports, but it's also not
# provably prose, so python stays on (fail-open) alongside the Windows lane.
"install.ps1 → installer": (["scripts/install.ps1"], _lanes(python=True, installer=True)),
"installer test → installer": (
["scripts/tests/test-install-ps1-longpath.ps1"],
_lanes(python=True, installer=True),
),
"python source alone → no installer lane": (["run_agent.py"], _lanes(python=True, scan=True)),
# Unknown top-level file keeps Python on rather than risk a silent skip.
"unknown toplevel → python": (["Makefile"], _lanes(python=True)),
"mixed docs+python → python": (["README.md", "agent/x.py"], _lanes(python=True, scan=True)),
"mixed docs+frontend → frontend": (["README.md", "apps/x.tsx"], _lanes(frontend=True)),
# tests-only diffs: pytest lanes stay ON, product jobs (Desktop E2E,
# Docker) gate on python_prod and skip.
"tests-only → python without python_prod": (
["tests/agent/test_foo.py", "tests/conftest.py"],
_lanes(python=True, python_prod=False, scan=True),
),
"tests + prod source → both lanes": (
["tests/agent/test_foo.py", "agent/x.py"],
_lanes(python=True, scan=True),
),
# Runner infrastructure is NOT tests-only — a bad runner edit can mask
# real failures, so it keeps the conservative full lane set.
"test runner script → python_prod stays on": (
["scripts/run_tests_parallel.py"],
_lanes(python=True, scan=True),
),
# Supply-chain lanes
".pth file → scan": (["evil.pth"], _lanes(python=True, scan=True)),
"setup.py → scan": (["setup.py"], _lanes(python=True, scan=True)),
"mcp catalog manifest → mcp_catalog": (
["optional-mcps/foo/manifest.yaml"],
_lanes(python=True, mcp_catalog=True),
),
"mcp_catalog.py → mcp_catalog": (
["hermes_cli/mcp_catalog.py"],
_lanes(python=True, scan=True, mcp_catalog=True),
),
# CI-sensitive files require explicit review label.
"eslint config → ci_review": (
["apps/desktop/eslint.config.mjs"],
_lanes(frontend=True, ci_review=True),
),
"shared eslint config → ci_review": (
["eslint.config.shared.mjs"],
_lanes(python=True, ci_review=True),
),
"ui-tui eslint config → ci_review": (
["ui-tui/eslint.config.mjs"],
_lanes(frontend=True, ci_review=True),
),
"web eslint config → ci_review": (
["web/eslint.config.js"],
_lanes(frontend=True, ci_review=True),
),
"shared package eslint config → ci_review": (
["apps/shared/eslint.config.mjs"],
_lanes(frontend=True, ci_review=True),
),
"bootstrap-installer eslint config → ci_review": (
["apps/bootstrap-installer/eslint.config.mjs"],
_lanes(frontend=True, ci_review=True),
),
"prettier config → ci_review": (
[".prettierrc"],
_lanes(python=True, ci_review=True),
),
"workflow yml → ci_review (also fail-open all)": (
[".github/workflows/typecheck.yml"],
DEFAULT,
),
"composite action → ci_review (also fail-open all)": (
[".github/actions/retry/action.yml"],
DEFAULT,
),
# Normal desktop source doesn't trigger ci_review.
"desktop src → no ci_review": (
["apps/desktop/src/app.tsx"],
_lanes(frontend=True),
),
# Fail open: CI-config / empty / blank diffs run everything.
".github change → all": ([".github/workflows/tests.yml"], DEFAULT),
"action change → all": ([".github/actions/detect-changes/action.yml"], DEFAULT),
"empty diff → all": ([], DEFAULT),
"blank lines → all": (["", " "], DEFAULT),
}
@pytest.mark.parametrize("files,expected", CASES.values(), ids=CASES.keys())
def test_classify(files, expected):
assert classify(files) == expected
def test_ci_review_files_returns_only_sensitive_paths_sorted_and_unique():
assert ci_review_files([
"apps/desktop/src/app.tsx",
".github/workflows/ci.yml",
"apps/desktop/eslint.config.mjs",
".github/workflows/ci.yml",
]) == [
".github/workflows/ci.yml",
"apps/desktop/eslint.config.mjs",
]