ci: run the full Python suite on Windows and Linux, macOS runs macos_only

Fold the OS-specific lanes into tests.yml's test job as a three-OS matrix.
Linux and Windows run the full suite (conftest skips foreign-OS markers).
macOS keeps macos_only for now. Delete tests-os.yml.
This commit is contained in:
ethernet
2026-08-30 00:31:33 -04:00
parent 6e171c3aea
commit 9d5fca2488
5 changed files with 92 additions and 227 deletions

View File

@@ -75,16 +75,6 @@ jobs:
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests.yml
# macOS + Windows lanes. The main `tests` lane above is Linux-only, and
# the OS-marked tests it collects are skipped there by design (see the
# `_OS_MARKS` comment in tests/conftest.py) — this is where they run.
# Same `python` lane gate: if no Python changed, neither runs.
tests-os:
name: OS-specific tests
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests-os.yml
lint:
name: Python lints
needs: detect
@@ -218,7 +208,6 @@ jobs:
needs:
- detect
- tests
- tests-os
- lint
- js-tests
- installer-tests

View File

@@ -1,154 +0,0 @@
name: OS-specific tests
# Runs the tests that can only be trusted on their own host OS.
#
# The main Python suite (.github/workflows/tests.yml) runs on
# ubuntu-latest and covers everything that is either platform-agnostic or
# genuinely Linux-specific. Tests whose subject is macOS- or
# Windows-specific behaviour carry a marker (see the ``_OS_MARKS`` block
# comment in tests/conftest.py) and are SKIPPED on Linux, because faking
# ``sys.platform`` on a Linux runner selects the branch under test without
# reproducing any of the OS behaviour that branch exists for. This workflow
# is where those markers actually execute:
#
# macos → ``-m macos_only`` on macos-latest
# windows → ``-m windows_only`` on windows-latest
#
# Deliberately NOT sliced. The marked set is small (tens of tests, not
# thousands), so one plain ``pytest`` process per OS is both faster and far
# less machinery than the per-file parallel runner the Linux lane uses.
# If either lane grows past its timeout, that is the signal to reach for
# scripts/run_tests.sh here too.
#
# Each lane FAILS when it selects zero tests (pytest exit code 5). Without
# that guard, a renamed marker or a bad selector would report a green job
# that ran nothing — the exact silent-coverage-loss failure this workflow
# exists to prevent.
on:
workflow_call:
permissions:
contents: read
concurrency:
group: tests-os-${{ github.ref }}
cancel-in-progress: true
jobs:
os-tests:
name: ${{ matrix.name }}
runs-on: ${{ matrix.runner }}
timeout-minutes: 30
strategy:
fail-fast: false
matrix:
include:
- name: macOS-only tests
runner: macos-latest
marker: macos_only
- name: Windows-only tests
runner: windows-latest-32-core
marker: windows_only
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned for the same reason as the Linux lane: unpinned, setup-uv
# resolves "latest" by fetching a manifest on every job and a
# transient fetch failure fails the whole job.
version: "0.9.28"
enable-cache: true
cache-dependency-glob: |
pyproject.toml
uv.lock
- name: Set up Python 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Install dependencies
# Same extras as the Linux test lane so an OS-marked test can import
# anything its Linux siblings can. ``[all]`` is deliberately
# Windows/macOS-installable (see the policy comment on the extra in
# pyproject.toml — matrix/python-olm was removed from it precisely
# because it could not build here).
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
run: uv cache prune --ci
- name: Run ${{ matrix.marker }} tests
# Two-step selection:
#
# 1. scripts/ci/list_os_marked_tests.py narrows WHICH FILES are
# imported. ``-m`` filters after collection, and collection
# imports every module under tests/ — on this host that would
# drag ~900 unrelated test modules through import, where a
# single unrelated ImportError would fail a job whose own
# subject is fine. The helper exits non-zero if the marker
# matches no file at all.
# 2. ``-m`` decides WHICH TESTS run, and stays authoritative.
# Passing it on the command line REPLACES pyproject's
# ``-m 'not integration'`` addopts (same option, last wins) —
# hence repeating ``not integration``, or the integration
# suite would return through the side door.
#
# ``--timeout-method`` needs no override: tests/conftest.py's
# pytest_configure already downgrades the signal-based timer on
# Windows, which has no SIGALRM.
shell: bash
run: |
set -uo pipefail
LIST="${RUNNER_TEMP:-.}/selected-tests.txt"
# Process substitution would hide the helper's exit status, so write
# to a file and check it explicitly.
if ! uv run --no-sync python scripts/ci/list_os_marked_tests.py \
"${{ matrix.marker }}" > "$LIST"; then
echo "::error::could not enumerate ${{ matrix.marker }} test files"
exit 1
fi
if [ ! -s "$LIST" ]; then
echo "::error::empty ${{ matrix.marker }} file list"
exit 1
fi
# Deliberately NOT `mapfile`: that is a bash 4 builtin and the macOS
# runner's /bin/bash is 3.2. Word-splitting is safe here because the
# helper emits repo-relative test paths, which contain no spaces.
# shellcheck disable=SC2046
set -- $(cat "$LIST")
echo "selected $# file(s) for ${{ matrix.marker }}:"
cat "$LIST"
# ``shell: bash`` runs this script with ``-e`` injected, which
# ``set -uo pipefail`` above does not clear. A bare pytest call
# would therefore abort the script on any non-zero exit and the
# exit-5 branch below would be unreachable dead code — the job
# would still fail red, but the diagnostic would never print.
status=0
uv run --no-sync python -m pytest \
"$@" \
-m "${{ matrix.marker }} and not integration" \
-v --tb=short || status=$?
if [ "$status" -eq 5 ]; then
echo "::error::No tests matched -m ${{ matrix.marker }}. Either the" \
"marker was renamed/dropped or selection is broken — this job" \
"must never pass without running its OS's tests."
exit 1
fi
exit "$status"
env:
# Belt-and-suspenders with tests/conftest.py's env blanking: no
# test may reach a real provider API.
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
NOUS_API_KEY: ""

View File

@@ -13,21 +13,38 @@ concurrency:
jobs:
test:
name: Run tests
# One 96-core runner for the whole suite. There is no slicing. Slicing
# existed to spread the suite over 4-core runners. It cost a matrix job, a
# duration cache, a per-slice artifact and a merge job to do it.
#
# 96 cores clear the floor that the slowest single test file sets (about
# 82s). A second slice divides work that is already at that floor, and
# adds a second setup.
runs-on: ubuntu-latest-96-core
timeout-minutes: 30
# One matrix covers three host OSes. Linux and Windows run the full suite.
# macOS runs only its ``macos_only`` tests for now (the suite is not yet
# macOS-clean). tests/conftest.py's ``pytest_collection_modifyitems`` hook
# skips foreign-OS markers on each host, so an un-gated full run selects
# "generic + this host's own marker" — no ``-m`` filter is needed for the
# full-suite lanes. Nothing slices by platform now: every lane runs the
# whole discoverable suite and lets the markers do the gating.
name: Python tests (${{ matrix.os }})
runs-on: ${{ matrix.runner }}
timeout-minutes: 60
strategy:
fail-fast: false
matrix:
include:
- os: linux
runner: ubuntu-latest-96-core
marker: ""
workers: "96"
- os: windows
runner: windows-latest-32-core
marker: ""
workers: ""
- os: macos
runner: macos-latest
marker: macos_only
workers: ""
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install ripgrep (prebuilt binary)
if: matrix.os == 'linux'
run: |
set -euo pipefail
RG_VERSION=15.1.0
@@ -88,38 +105,71 @@ jobs:
run: uv cache prune --ci
- name: Run tests
# Per-file isolation via scripts/run_tests.sh: each test file runs
# in its own freshly-spawned `python -m pytest <file>` subprocess
# with bounded parallelism. No xdist, no shared workers, no
# module-level state leakage between files.
# Two shapes, selected by whether this lane carries a marker:
#
# No --files: the runner discovers the suite itself. The discovered
# set is identical to the list the removed matrix job used to pass in.
# * No marker (Linux + Windows): the full suite via
# scripts/run_tests.sh — per-file isolation, bounded parallelism,
# and the conftest OS-marker skip does the platform gating.
# * Marker (macOS for now): plain pytest over the files that carry
# the marker. Two-step selection: list_os_marked_tests.py narrows
# WHICH FILES are imported (collection otherwise drags ~900
# unrelated modules through import on the macOS host, where a
# single unrelated ImportError fails a lane whose own subject
# is fine), and `-m` stays the authoritative selector. Passing
# `-m` REPLACES pyproject's ``-m 'not integration'`` addopts, so
# ``not integration`` is repeated here or the integration suite
# returns through the side door.
# Each marked lane FAILS on pytest exit 5 (zero tests selected) so a
# renamed/broken marker can never report green while running nothing.
shell: bash
run: |
source .venv/bin/activate
set -uo pipefail
if [ -n "${{ matrix.marker }}" ]; then
LIST="${RUNNER_TEMP:-.}/selected-tests.txt"
if ! uv run --no-sync python scripts/ci/list_os_marked_tests.py \
"${{ matrix.marker }}" > "$LIST"; then
echo "::error::could not enumerate ${{ matrix.marker }} test files"
exit 1
fi
if [ ! -s "$LIST" ]; then
echo "::error::empty ${{ matrix.marker }} file list"
exit 1
fi
# Deliberately NOT `mapfile`: that is a bash 4 builtin and the
# macOS runner's /bin/bash is 3.2. Word-splitting is safe here
# because the helper emits repo-relative paths with no spaces.
# shellcheck disable=SC2046
set -- $(cat "$LIST")
echo "selected $# file(s) for ${{ matrix.marker }}:"
cat "$LIST"
status=0
uv run --no-sync python -m pytest \
"$@" \
-m "${{ matrix.marker }} and not integration" \
-v --tb=short || status=$?
if [ "$status" -eq 5 ]; then
echo "::error::No tests matched -m ${{ matrix.marker }}. Either the" \
"marker was renamed/dropped or selection is broken — this job" \
"must never pass without running its OS's tests."
exit 1
fi
exit "$status"
fi
# Full suite. run_tests.sh locates the venv itself (both the POSIX
# bin/ and the Windows Scripts/ layout), so no per-OS activation.
scripts/run_tests.sh
env:
# This is the maximum number of test FILES that run together.
# run_tests_parallel.py starts one pytest subprocess for each file
# from a single ThreadPoolExecutor, so this value IS the limit. The
# default is cpu_count*2, which is 192 here.
#
# Measured on this runner (96-core EPYC 7763, 377GB). Whole suite,
# two repetitions for each value. See run 32549672063:
#
# workers x cores mean
# 48 0.5x 138s
# 96 1.0x 126s <- fastest
# 144 1.5x 132s
# 192 2.0x 132s
# 240 2.5x 140s
# 288 3.0x 142s
#
# One worker for each core wins. The curve is shallow: 126s to 142s
# across a 6x range. The suite has sufficient concurrency at this
# size. The remaining time is the slowest files plus the setup.
# Workers above the core count only add contention.
HERMES_TEST_WORKERS: 96
# The maximum number of test FILES that run together (Linux, from
# the measured whole-suite sweep — one worker per core is fastest on
# the 96-core runner, and the curve is shallow). Windows and macOS
# leave this unset so run_tests_parallel.py falls back to
# cpu_count*2.
HERMES_TEST_WORKERS: ${{ matrix.workers }}
# Ensure tests don't accidentally call real APIs
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
@@ -149,16 +199,7 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
# Keyed on the dependency manifests, so the cache is reused until
# pyproject.toml or uv.lock changes. `uv sync` still runs every
# time, but resolves from the warm cache instead of re-downloading
# and re-building wheels.
enable-cache: true
cache-dependency-glob: |
pyproject.toml
@@ -168,22 +209,11 @@ jobs:
run: uv python install 3.11
- name: Install dependencies
# `uv sync --locked` installs the exact pinned set from uv.lock (and
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# Same extras as the test job's sync above: the hermetic test env
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
# in the venv up front.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
# re-download, keeping the persisted cache small and fast to restore.
run: uv cache prune --ci
- name: Run e2e tests

View File

@@ -1,8 +1,8 @@
#!/usr/bin/env python3
"""List the test files that carry a given OS marker.
Used by ``.github/workflows/tests-os.yml`` to scope what the macOS and
Windows lanes import.
Used by the marked-OS lane of ``.github/workflows/tests.yml`` to scope what
the macOS lane imports.
Why scope at all, when ``pytest -m macos_only`` already selects correctly?
Because ``-m`` filters AFTER collection, and collection IMPORTS every test

View File

@@ -146,8 +146,8 @@ def _split_pathspec(value: str) -> List[str]:
# behaviour, and names the CI lane where those tests actually execute.
_OS_MARKERS = {
"linux_only": ("linux", "the main Linux CI lane"),
"macos_only": ("darwin", "the tests-os CI lane (macos-latest)"),
"windows_only": ("win32", "the tests-os CI lane (windows-latest)"),
"macos_only": ("darwin", "the macOS Python-tests lane"),
"windows_only": ("win32", "the Windows Python-tests lane"),
}