diff --git a/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/cversions.2.db b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/cversions.2.db new file mode 100644 index 0000000000..0419c8df5f Binary files /dev/null and b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/cversions.2.db differ diff --git a/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{1B47D616-5F51-4756-BD03-80DA8D5C2FA2}.2.ver0x0000000000000001.db b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{1B47D616-5F51-4756-BD03-80DA8D5C2FA2}.2.ver0x0000000000000001.db new file mode 100644 index 0000000000..f50a4921e7 Binary files /dev/null and b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{1B47D616-5F51-4756-BD03-80DA8D5C2FA2}.2.ver0x0000000000000001.db differ diff --git a/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{6AF0698E-D558-4F6E-9B3C-3716689AF493}.2.ver0x0000000000000001.db b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{6AF0698E-D558-4F6E-9B3C-3716689AF493}.2.ver0x0000000000000001.db new file mode 100644 index 0000000000..5b03891de5 Binary files /dev/null and b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{6AF0698E-D558-4F6E-9B3C-3716689AF493}.2.ver0x0000000000000001.db differ diff --git a/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{DDF571F2-BE98-426D-8288-1A9A39C3FDA2}.2.ver0x0000000000000001.db b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{DDF571F2-BE98-426D-8288-1A9A39C3FDA2}.2.ver0x0000000000000001.db new file mode 100644 index 0000000000..55f0776448 Binary files /dev/null and b/%SystemDrive%/ProgramData/Microsoft/Windows/Caches/{DDF571F2-BE98-426D-8288-1A9A39C3FDA2}.2.ver0x0000000000000001.db differ diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index 465ad80907..284b9103a2 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -172,10 +172,9 @@ jobs: NOUS_API_KEY: "" run: | # Each of these tests drives a container, so the docker daemon sets - # the limit and not the processor. This pins the workers explicitly - # (nproc) so the comment and behavior stay coupled if the runner's - # core count or the default ever changes. - HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600 + # the limit and not the processor. This pins the xdist worker count + # to the core count. + HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ # --------------------------------------------------------------------------- # Rebuild and push each architecture only after the unprivileged build/test diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 0c9fa6702c..41ef22d428 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -26,8 +26,7 @@ jobs: # there has a long tail of files that spawn servers/sockets/subprocesses # and run minutes-per-file on the CI runner (they pass in seconds on an # idle dev box) — measured 88.6% completion at 60 minutes on run - # 33338310764. The per-file cap (HERMES_TEST_FILE_TIMEOUT) bounds any - # single hung file; this bounds the whole lane. + # 33338310764. timeout-minutes: ${{ matrix.timeout_min }} strategy: fail-fast: false @@ -37,19 +36,16 @@ jobs: runner: ubuntu-latest-96-core marker: "" workers: "96" - file_timeout: "" timeout_min: 30 - os: windows runner: windows-latest-32-core marker: "" workers: "32" - file_timeout: "2400" timeout_min: 150 - os: macos runner: macos-latest marker: macos_only workers: "" - file_timeout: "" timeout_min: 30 steps: - name: Disable Windows Defender real-time scanning @@ -151,20 +147,17 @@ jobs: run: uv cache prune --ci - name: Run tests - # Three shapes: + # Two shapes: # - # * Windows (EXPERIMENT, PR #96458): xdist with --dist loadfile. - # The per-file subprocess model pays a spawn+import wall of - # ~0.5-1.5s per file x ~3400 files there — a ~6-minute floor - # before any test runs. xdist pays that wall once per worker. - # loadfile keeps each file's tests on ONE worker, so the - # remaining hazard is state shared by files co-scheduled on a - # worker. Failures from that are stateful-test bugs to fix, not - # runner bugs. - # * Linux: the full suite via scripts/run_tests.sh — per-file - # isolation, bounded parallelism. The spawn floor is ~15ms per - # file there, so isolation is nearly free; keep the stronger - # guarantee. + # * Full-suite lanes (Linux + Windows): scripts/run_tests.sh — + # pytest-xdist with --dist loadfile behind a hermetic env. + # Persistent workers pay the interpreter+import wall once per + # worker instead of once per file (a ~6-minute floor on + # Windows under the old per-file subprocess model). loadfile + # keeps each file's tests on ONE worker, so the remaining + # hazard is state shared by files co-scheduled on a worker. + # Failures from that are stateful-test bugs to fix, not runner + # bugs. # * macOS (marker set): plain pytest over the files that carry the # marker. list_os_marked_tests.py narrows WHICH FILES are # imported (collection otherwise drags ~900 unrelated modules @@ -179,10 +172,6 @@ jobs: run: | set -uo pipefail - if [ "${{ matrix.os }}" = "windows" ]; then - bash scripts/run_xdist.sh 32 tests/ - exit $? - fi if [ -n "${{ matrix.marker }}" ]; then LIST="${RUNNER_TEMP:-.}/selected-tests.txt" @@ -224,19 +213,11 @@ jobs: # activation. scripts/run_tests.sh env: - # The maximum number of test FILES that run together. Linux pins - # 96 (the measured whole-suite sweep — one worker per core is - # fastest on the 96-core runner, and the curve is shallow). - # Windows and macOS leave this unset so run_tests_parallel.py's - # default (cpu_count) applies. + # xdist worker count (-n). Linux pins 96 (the measured whole-suite + # sweep — one worker per core is fastest on the 96-core runner, + # and the curve is shallow); Windows pins 32. macOS leaves this + # unset (it runs the marked-file lane, not the full suite). HERMES_TEST_WORKERS: ${{ matrix.workers }} - # Windows CI-only: the full-suite lane showed ~160 files at or - # past the 300s default under runner contention (they pass in - # seconds on an idle box); each kill also burns a retry, so the - # churn — not test time — dominated the lane's wall clock. Give - # the per-file cap real headroom there. Empty on other lanes - # (run_tests.sh drops empty vars before the runner sees them). - HERMES_TEST_FILE_TIMEOUT: ${{ matrix.file_timeout }} # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" diff --git a/AGENTS.md b/AGENTS.md index e140522e96..84b9ad7428 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1579,30 +1579,29 @@ def profile_env(tmp_path, monkeypatch): ### Python **ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8, -per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist, -worker count auto-scaled from CPU count). Direct `pytest` +pytest-xdist with `--dist loadfile` so every test of a file runs on ONE worker — +worker count defaults to the CPU count). Direct `pytest` on a 16+ core developer machine with API keys set diverges from CI in ways that have caused multiple "works locally, fails in CI" incidents (and the reverse). ```bash scripts/run_tests.sh # full suite, CI-parity scripts/run_tests.sh tests/gateway/ # one directory -scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular) +scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k) scripts/run_tests.sh -v --tb=long # pass-through pytest flags ``` -**Flake policy:** the runner auto-retries a failing test FILE once in a fresh -subprocess (`--file-retries`, default 1; `HERMES_TEST_FILE_RETRIES=0` to -disable). Pass-on-retry counts as green but is printed in a `⚠ FLAKY` summary -section with both attempts' output. A FLAKY report is a bug to fix, not noise -to ignore — timing-sensitive tests must not assume a quiet runner (loose +**Flake policy:** the runner is xdist, so a file may pass or fail depending +on which siblings share its worker — any order-dependent failure is a bug to +fix, not noise. Timing-sensitive tests must not assume a quiet runner (loose wall-clock bounds ≥ 2s, event-based sync, no `assert not _wait_until(...)` negative-timing races). -#### Subprocess-per-test-file isolation +#### File-pinned xdist workers -Every test file runs in a freshly-spawned Python subprocess via `run_tests_parallel.py`. This means module-level dicts/sets and -ContextVars from one test file cannot leak into the next. +Every test file runs pinned to ONE xdist worker (`--dist loadfile` via `scripts/run_tests.sh`), so module-level +state pollution is bounded to files co-scheduled on the same worker; files that need true isolation should reset +state in fixtures. #### Why the wrapper diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 61bacafbd8..fc5c22c2cb 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -201,8 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes ### Run tests ```bash -# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation -# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md +# Preferred — matches CI (hermetic `env -i`, pytest-xdist with --dist +# loadfile, worker count defaults to the CPU count); see AGENTS.md scripts/run_tests.sh # Alternative (activate the venv first). The wrapper is still recommended diff --git a/optional-skills/creative/comfyui/tests/README.md b/optional-skills/creative/comfyui/tests/README.md index d27fa97e32..783735ac4c 100644 --- a/optional-skills/creative/comfyui/tests/README.md +++ b/optional-skills/creative/comfyui/tests/README.md @@ -45,7 +45,7 @@ When you change a script: The parent hermes-agent repo used to enable `pytest-xdist` by default (`-n auto`); the canonical runner has since moved to per-file subprocess -isolation via `scripts/run_tests_parallel.py` and no longer uses xdist. +pytest-xdist with `--dist loadfile` via `scripts/run_tests.sh`. This suite is small enough that parallelism isn't worth the complexity, and pytest-xdist isn't always installed in the user's environment. The `-c tests/pytest.ini -o addopts="-p no:xdist"` flags make the suite run diff --git a/scripts/ci/classify_changes.py b/scripts/ci/classify_changes.py index 935703c870..cd333f385d 100644 --- a/scripts/ci/classify_changes.py +++ b/scripts/ci/classify_changes.py @@ -144,7 +144,7 @@ def _py_test_only(p: str) -> bool: Product jobs (Desktop E2E's ``hermes serve`` backend, the Docker image) run installed code — nothing under ``tests/`` is packaged or importable - there. scripts/run_tests.sh and run_tests_parallel.py are deliberately + there. scripts/run_tests.sh is deliberately NOT test-only: they are runner infrastructure, and a bad edit there can mask real failures, so they stay conservative (python_prod=true). """ diff --git a/scripts/run_tests.sh b/scripts/run_tests.sh index 4445d6d542..4dd7047e3c 100755 --- a/scripts/run_tests.sh +++ b/scripts/run_tests.sh @@ -3,10 +3,11 @@ # `pytest` directly to guarantee your local run matches CI behavior. # # What this script enforces: -# * Per-file isolation via scripts/run_tests_parallel.py — each test -# file runs in its own freshly-spawned `python -m pytest ` -# subprocess. No xdist, no shared workers, no module-level leakage -# between files. +# * pytest-xdist with --dist loadfile — each test FILE's tests all run on +# ONE worker, so file-internal ordering is preserved and cross-file +# pollution is bounded to files co-scheduled on a worker. Persistent +# workers also pay the interpreter+import wall (~0.5-1.5s on Windows) +# once per worker instead of once per file. # * TZ=UTC, LANG=C.UTF-8, PYTHONHASHSEED=0 (deterministic) # * Env vars blanked (conftest.py also does this, but this # is belt-and-suspenders for anyone running pytest outside our @@ -15,23 +16,19 @@ # # Usage: # scripts/run_tests.sh # full suite -# scripts/run_tests.sh -j 4 # cap parallelism +# scripts/run_tests.sh -j 4 # cap worker count # scripts/run_tests.sh tests/agent/ # discover only here # scripts/run_tests.sh tests/agent/ tests/acp/ # multiple roots # scripts/run_tests.sh tests/foo.py # single file # scripts/run_tests.sh tests/foo.py -q # path + bare pytest flag # scripts/run_tests.sh tests/foo.py -v --tb=long # bare flags "just work" -# scripts/run_tests.sh -k 'pattern' # value flags pass through too -# scripts/run_tests.sh tests/foo.py -- --tb=long # explicit '--' still works +# scripts/run_tests.sh -k 'pattern' # value flags pass through too # -# Bare pytest flags (anything starting with '-' that isn't one of this -# runner's own options: -j/--jobs, --paths, --slice, --file-timeout, etc.) -# are forwarded to each per-file pytest invocation automatically — no '--' -# separator required. The explicit '--' form still works and stacks with -# bare flags. Positional path arguments override the default discovery -# root (tests/). +# Bare pytest flags (anything starting with '-' that isn't -j/--jobs) are +# forwarded to pytest. Positional path arguments override the default +# discovery root (tests/). -set -euo pipefail +set -uo pipefail # ── Locate repo root ──────────────────────────────────────────────────────── SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -40,7 +37,7 @@ REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" # ── Locate python ─────────────────────────────────────────────────────────── # Probe local venvs first; fall back to the Nix devShell's editable venv # (HERMES_PYTHON is exported by the devShell hook and ships [dev] extras: -# pytest, pytest-asyncio, pytest-timeout, ruff, ty). +# pytest, pytest-asyncio, pytest-xdist). # # A candidate must have pytest INSTALLED, not merely exist. The release venv # at ~/.hermes/hermes-agent/venv has bin/activate but no pytest, so an @@ -59,12 +56,7 @@ for candidate in "$REPO_ROOT/.venv" "$REPO_ROOT/venv" "$HOME/.hermes/hermes-agen break fi SKIPPED_VENVS="$SKIPPED_VENVS $candidate" - fi - # Native Windows venv layout: python.exe and activate live under - # Scripts/, and there is no bin/. Anyone running this script from - # Git Bash / MSYS with a `python -m venv`- or uv-created venv hits - # this branch — without it the canonical runner refuses to start. - if [ -f "$candidate/Scripts/activate" ]; then + elif [ -f "$candidate/Scripts/activate" ]; then if "$candidate/Scripts/python.exe" -c 'import pytest' 2>/dev/null; then VENV="$candidate" VENV_PYTHON="$candidate/Scripts/python.exe" @@ -73,40 +65,35 @@ for candidate in "$REPO_ROOT/.venv" "$REPO_ROOT/venv" "$HOME/.hermes/hermes-agen SKIPPED_VENVS="$SKIPPED_VENVS $candidate" fi done - -if [ -n "$SKIPPED_VENVS" ]; then - for skipped in $SKIPPED_VENVS; do - echo "▶ skipping venv without pytest: $skipped" >&2 - done -fi - -if [ -n "$VENV" ]; then - PYTHON="$VENV_PYTHON" -elif [ -n "${HERMES_PYTHON:-}" ] && [ -x "$HERMES_PYTHON" ] \ - && "$HERMES_PYTHON" -c 'import pytest' 2>/dev/null; then - # Guard with an import check: HERMES_PYTHON may point at the RELEASE - # venv (no pytest) when inherited from a wrapped `hermes` binary rather - # than the devShell hook. - PYTHON="$HERMES_PYTHON" - echo "▶ no local venv — using Nix dev venv via HERMES_PYTHON: $PYTHON" -else - echo "error: no virtualenv with pytest found in $REPO_ROOT/.venv or $REPO_ROOT/venv," >&2 - echo " and HERMES_PYTHON is not a python with pytest (enter the Nix devShell or create a venv)" >&2 - if [ -n "$SKIPPED_VENVS" ]; then - echo " (skipped for missing pytest:$SKIPPED_VENVS — install dev extras there, or create $REPO_ROOT/.venv)" >&2 +if [ -z "$VENV_PYTHON" ]; then + if [ -n "${HERMES_PYTHON:-}" ] && "${HERMES_PYTHON}" -c 'import pytest' 2>/dev/null; then + VENV_PYTHON="$HERMES_PYTHON" + else + echo "✗ No venv with pytest found. Install dev extras:" >&2 + echo " uv sync --extra dev" >&2 + if [ -n "$SKIPPED_VENVS" ]; then + echo " (skipped for missing pytest:$SKIPPED_VENVS — install dev extras there, or create $REPO_ROOT/.venv)" >&2 + fi + exit 1 fi - exit 1 -fi - - -# ── Live-gateway plugin (computed before we drop env) ─────────────────────── -EXTRA_PYTHONPATH="" -EXTRA_PYTEST_PLUGINS="" -if [ -f "$HOME/.hermes/pytest_live_guard.py" ]; then - EXTRA_PYTHONPATH="$HOME/.hermes" - EXTRA_PYTEST_PLUGINS="pytest_live_guard" fi +PYTHON="$VENV_PYTHON" +# ── Split args: our -j/--jobs vs pytest passthrough ───────────────────────── +N="${HERMES_TEST_WORKERS:-auto}" +PYTEST_ARGS=() +while [ $# -gt 0 ]; do + case "$1" in + -j|--jobs) + N="$2"; shift 2 ;; + -j*) + N="${1#-j}"; shift ;; + --jobs=*) + N="${1#--jobs=}"; shift ;; + *) + PYTEST_ARGS+=("$1"); shift ;; + esac +done # ── Windows location variables (computed before we drop env) ─────────────── # `env -i` forwards HOME, which is enough on POSIX. Native Windows CPython @@ -123,27 +110,28 @@ for _win_var in USERPROFILE HOMEDRIVE HOMEPATH LOCALAPPDATA APPDATA SYSTEMROOT T fi done -# ── Test-runner knobs (computed before we drop env) ──────────────────────── -# The runner's own documented environment knobs must survive the hermetic -# `env -i` below, or they are silent no-ops for anyone invoking this script: -# -# * HERMES_TEST_WORKERS / PATHS / FILE_TIMEOUT / FILE_RETRIES / SLICE are -# read by run_tests_parallel.py at argparse-default time — inside the -# stripped environment. +# ── Live-gateway plugin (computed before we drop env) ─────────────────────── +EXTRA_PYTHONPATH="" +EXTRA_PYTEST_PLUGINS="" +if [ -f "$HOME/.hermes/pytest_live_guard.py" ]; then + EXTRA_PYTHONPATH="$HOME/.hermes" + EXTRA_PYTEST_PLUGINS="pytest_live_guard" +fi + +# ── Test-runner knobs (computed before we drop env) ────────────────────────── # * HERMES_TEST_IMAGE is read by tests/docker/conftest.py to skip its # session-scoped `docker build`. CI's docker.yml sets it to the image -# the build step just loaded; stripping it made every per-file pytest -# subprocess rebuild the 5GB image from a cold builder cache instead -# (~4 min per worker per run, and the rebuilt image lacked the -# HERMES_GIT_SHA build-arg the workflow bakes in). +# the build step just loaded; stripping it made every pytest subprocess +# rebuild the 5GB image from a cold builder cache instead (~4 min per +# worker per run, and the rebuilt image lacked the HERMES_GIT_SHA +# build-arg the workflow bakes in). # # These are test-infrastructure knobs, not credentials — same class as the # HERMES_RUN_SLOW_PET_TESTS / HERMES_E2E_BROWSER opt-ins already forwarded. # Keep this an explicit allowlist (no HERMES_TEST_* glob) so the "no # credential can leak" property stays auditable at a glance. TEST_ENV=() -for _test_var in HERMES_TEST_IMAGE HERMES_TEST_WORKERS HERMES_TEST_PATHS \ - HERMES_TEST_FILE_TIMEOUT HERMES_TEST_FILE_RETRIES HERMES_TEST_SLICE; do +for _test_var in HERMES_TEST_IMAGE; do if [ -n "${!_test_var:-}" ]; then TEST_ENV+=("$_test_var=${!_test_var}") fi @@ -152,20 +140,18 @@ done # ── Run in hermetic env ────────────────────────────────────────────────────── # env -i: start with empty environment, opt-in only what we need. # No credential var can leak — you'd have to explicitly add it here. -echo "▶ running per-file parallel test suite via run_tests_parallel.py" +echo "▶ running pytest-xdist (-n $N --dist loadfile)" echo " (TZ=UTC LANG=C.UTF-8 PYTHONHASHSEED=0; clean env)" cd "$REPO_ROOT" # ── Pre-compile .pyc bytecode cache ───────────────────────────────────────── -# Each test file runs in its own subprocess via run_tests_parallel.py. -# Pre-building the bytecode cache once here (instead of each subprocess -# compiling on first import) avoids redundant work across ~2000 processes. -# Uses git to list tracked .py files (skips venv, node_modules, etc). +# xdist workers import the same modules; pre-building the bytecode cache once +# here avoids every worker compiling on first import. echo "▶ pre-compiling bytecode cache" "$PYTHON" -m compileall -q -j 0 -- $(git ls-files '*.py') >/dev/null 2>&1 || true -echo "▶ launching test runner" +echo "▶ pytest -n $N --dist loadfile" exec env -i \ PATH="$PATH" \ HOME="$HOME" \ @@ -180,4 +166,6 @@ exec env -i \ ${HERMES_E2E_BROWSER:+HERMES_E2E_BROWSER="$HERMES_E2E_BROWSER"} \ ${EXTRA_PYTHONPATH:+PYTHONPATH="$EXTRA_PYTHONPATH"} \ ${EXTRA_PYTEST_PLUGINS:+PYTEST_PLUGINS="$EXTRA_PYTEST_PLUGINS"} \ - "$PYTHON" "$SCRIPT_DIR/run_tests_parallel.py" "$@" + "$PYTHON" -m pytest -n "$N" --dist loadfile -p no:cacheprovider \ + -m "not integration" -q --tb=line \ + ${PYTEST_ARGS[@]+"${PYTEST_ARGS[@]}"} diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py deleted file mode 100755 index ea989635d9..0000000000 --- a/scripts/run_tests_parallel.py +++ /dev/null @@ -1,1262 +0,0 @@ -#!/usr/bin/env python3 -"""Per-file parallel test runner. - -The minimum-viable replacement for pytest-xdist + a subprocess-isolation -plugin. Discovers test files under ``tests/`` (excluding integration/e2e -unless explicitly requested), then runs one ``python -m pytest `` -subprocess per file, with bounded parallelism (default: ``os.cpu_count()``). - -Why per-file rather than per-test? - Per-test spawn overhead (~250ms × 17k tests = 70min CPU minimum) - swamped the actual work. Per-file spawn (~250ms × ~850 files = ~3.5min) - fits in the budget while still giving every file a fresh Python - interpreter — the only isolation boundary that actually matters - (cross-file module-level state leakage was the original flake source; - intra-file state is the test author's responsibility). - -Why drop xdist entirely? - xdist's persistent workers accumulate state across files, which is - exactly the leakage we wanted to fix. xdist also adds complexity - (loadfile vs loadscope, --max-worker-restart, internal control plane) - that we don't need when the unit of work is "run pytest on one file". - A subprocess.Popen pool gated by a semaphore is ~60 lines and does - the job. - -Usage: - python scripts/run_tests_parallel.py [pytest_args...] - - Common pytest args pass through to each per-file pytest invocation - (e.g. ``-q``, ``-v``, ``-x``, ``--tb=long``, ``-k 'pattern'``, ``--lf``) - with no special separator — a bare ``-q`` "just works". Anything after - a literal ``--`` is also passed through, and stacks with bare flags. - -Environment: - HERMES_TEST_WORKERS Override worker count (default: os.cpu_count()) - HERMES_TEST_PATHS Override discovery roots (colon-sep; on Windows - ';' also works and drive letters are handled; - default: 'tests') - -Exit code: 0 if every file's pytest exited 0; 1 otherwise. -""" - -from __future__ import annotations - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -import threading -import time -from concurrent.futures import ThreadPoolExecutor, Future -from pathlib import Path -from typing import Dict, List, Tuple - - -# Default test discovery roots. -_DEFAULT_ROOTS = ["tests"] - -# Directories to skip during discovery — these suites require real -# external services (a model gateway, a docker daemon with a prebuilt -# image, etc.) and are run in their own dedicated CI jobs: -# -# tests/e2e/ — .github/workflows/tests.yml :: e2e job -# tests/integration/ — historical; legacy --ignore flags -# tests/docker/ — .github/workflows/docker.yml :: -# build-amd64 job (runs against the freshly-loaded -# nousresearch/hermes-agent:test image, via -# ``HERMES_TEST_IMAGE`` so the fixture skips -# rebuild). The full pytest-shard runner can't -# host these because the session-scoped -# ``built_image`` fixture would do a 3-7min -# ``docker build``, -# so the build is guaranteed to die in fixture -# setup. The dedicated job sidesteps both costs. -_SKIP_PARTS = {"integration", "e2e", "docker"} - -# Per-file wall-clock cap. Override -# via --file-timeout or HERMES_TEST_FILE_TIMEOUT. -# -# Set to 300s (5 min) deliberately generous: the per-test subprocess -# isolation plugin spawns a fresh Python process per test, so a -# large-collection file pays N × (interpreter startup + import) of -# overhead before any test logic runs — and that overhead dilates under -# load on shared CI runners, producing false "no tests ran" timeouts on -# files that finish in ~100s on a quiet box. The Docker build matrix jobs -# take 7-10 min anyway, so this headroom costs nothing on total CI wall -# time while keeping a genuinely hung file bounded. -_DEFAULT_FILE_TIMEOUT_SECONDS = 300.0 - -# One-shot retry of failing test FILES. A file that exits non-zero is re-run -# once in a fresh subprocess; if the re-run passes, the file counts as passed -# but is loudly reported as FLAKY so it gets fixed rather than hidden. -# Deterministic failures fail both attempts — a real regression can never be -# laundered into green by this (it would have to flake in our favor twice in -# a row on the same runner, which is exactly the definition of a flake). -# Set to 0 to disable (env: HERMES_TEST_FILE_RETRIES). -_DEFAULT_FILE_RETRIES = 1 - -# Duration cache: maps relative file paths to last-observed subprocess -# wall-clock seconds. Used by ``--slice`` to distribute files across -# CI jobs by estimated total time, so no one job gets all the slow files. -_DURATIONS_FILE = "test_durations.json" - - -def _split_pathspec(value: str) -> List[str]: - """Split a separator-joined path list (``--paths``/``--files``/ - ``HERMES_TEST_PATHS``) into individual paths. - - POSIX: ``:``-separated, as documented. - - Windows: ``;`` (``os.pathsep``) and ``:`` are both accepted as - separators, but a ``:`` that forms a drive letter (``C:\\...`` or - ``C:/...``) stays glued to its path — a naive ``split(":")`` turns - ``C:\\repo\\tests`` into ``['C', '\\repo\\tests']``, where the bogus - ``C`` becomes a phantom discovery root and the rooted remainder only - resolves by accident of ``Path.__truediv__`` re-anchoring it onto - ``repo_root``'s drive. - """ - if sys.platform != "win32": - return [p for p in value.split(":") if p.strip()] - parts: List[str] = [] - for chunk in value.split(";"): - raw = chunk.split(":") - i = 0 - while i < len(raw): - part = raw[i] - if ( - len(part) == 1 - and part.isalpha() - and i + 1 < len(raw) - and raw[i + 1][:1] in ("\\", "/") - ): - part = f"{part}:{raw[i + 1]}" - i += 1 - parts.append(part) - i += 1 - return [p for p in parts if p.strip()] - -# Host-OS gating (see the ``_OS_MARKS`` block in tests/conftest.py): tests -# marked for another host are collected and SKIPPED by the conftest hook — -# this runner never executes them, by construction. The summary calls that -# out explicitly so a local run isn't misread as covering macOS/Windows -# behaviour, and names the CI lane where those tests actually execute. -_OS_MARKERS = { - "linux_only": ("linux", "the main Linux CI lane"), - "macos_only": ("darwin", "the macOS Python-tests lane"), - "windows_only": ("win32", "the Windows Python-tests lane"), -} - - -def _off_host_marker_files(files: List[Path]) -> dict[str, int]: - """Count discovered files referencing each marker for an OS we are not on. - - Whole-word text match, same approach as scripts/ci/list_os_marked_tests.py: - over-counting a prose mention is harmless here (the note is informational); - what matters is never reporting 0 while gated tests exist. - """ - off_host = { - marker: re.compile(rf"\b{marker}\b") - for marker, (host_prefix, _) in _OS_MARKERS.items() - if not sys.platform.startswith(host_prefix) - } - counts = {marker: 0 for marker in off_host} - for path in files: - try: - text = path.read_text(encoding="utf-8", errors="replace") - except OSError: - continue - for marker, pattern in off_host.items(): - if pattern.search(text): - counts[marker] += 1 - return {marker: n for marker, n in counts.items() if n} - - -def _approximately_count_tests( - files: List[Path], repo_root: Path -) -> dict[Path, int]: - """ - Make a decent estimate at individual tests per file. - Running ``pytest --co -q`` is WAY too slow because it actually imports everything. - - Returns a mapping ``{file_path: test_count}``. Files with zero - collected tests are omitted from the dict (not an error — e.g. the - file only defines fixtures / conftest helpers). - - """ - - results = {} - - for path in files: - with open(path, "r", encoding="utf-8") as f: - contents = f.read() - results[path] = contents.count("def test_") - - return results - - -def _discover_files(roots: List[Path]) -> List[Path]: - """Return every ``test_*.py`` under the given roots (sorted). - - Roots may be directories (recursed for ``test_*.py``) or explicit - ``.py`` files (included as-is, even if they don't match the - ``test_*`` prefix — caller knows what they want). - - Exclude any file whose path contains a component in ``_SKIP_PARTS``, - UNLESS the user explicitly named it as a root (in which case the - user's intent overrides the skip filter). This makes - ``scripts/run_tests.sh tests/docker/`` work locally the same way - ``pytest tests/docker/`` does — the CI-level skip exists to keep - the sharded matrix from blowing up, not to block targeted runs. - """ - seen: set[Path] = set() - out: List[Path] = [] - for root in roots: - if not root.exists(): - continue - if root.is_file(): - # Explicit file: include it as-is, skip the _SKIP_PARTS filter - # since the user named it directly. - real = root.resolve() - if real not in seen: - seen.add(real) - out.append(root) - continue - # If the explicit root itself sits inside a skipped dir (e.g. - # the user said ``tests/docker``), the user has overridden the - # skip for that subtree. Compute the set of skip-parts the user - # opted into, and only filter files whose path crosses a - # skip-part *outside* that opt-in. - root_skip_overrides = { - part for part in root.parts if part in _SKIP_PARTS - } - effective_skips = _SKIP_PARTS - root_skip_overrides - for path in root.rglob("test_*.py"): - if any(part in effective_skips for part in path.parts): - continue - real = path.resolve() - if real in seen: - continue - seen.add(real) - out.append(path) - return sorted(out) - - -def _kill_tree(proc: "subprocess.Popen", pgid: int | None = None) -> None: - """Kill the pytest subprocess and every descendant it spawned. - - A test run can spin up uvicorn servers, async runtimes, or other - long-running grandchildren that survive the pytest subprocess exit - if we don't kill the whole tree. ``subprocess.Popen.kill()`` only - targets the immediate child; grandchildren reparent to PID 1 - (Linux) / get adopted by services.exe (Windows) and leak. - - POSIX: the caller must pass ``pgid`` — the process group id captured - immediately after Popen (via ``os.getpgid(proc.pid)``). We can't - look it up here in the happy path because by the time we get - called the leader process has already been reaped and its pid is - gone from the kernel's process table, even though descendants in - the group are still alive. SIGKILL'ing the captured pgid takes out - everything in that group atomically. - - Windows: ``taskkill /F /T /PID`` walks the recorded ppid chain and - terminates the whole tree, even when the root has already exited. - - Why not psutil: psutil walks the parent-child tree, but in the - happy path the root has already been reaped so ``psutil.Process(pid)`` - can't find it; grandchildren reparented to PID 1 are also - unreachable by tree walk at that point. The platform-native - primitives (process groups / taskkill) handle both cases correctly - without an extra abstraction layer. - """ - if proc.pid is None: - return - - if sys.platform == "win32": - try: - - subprocess.run( - ["taskkill", "/F", "/T", "/PID", str(proc.pid)], - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - timeout=10, - ) # windows-footgun: ok - except (subprocess.TimeoutExpired, FileNotFoundError, OSError): - pass - else: - # POSIX: kill the captured pgid. Local-import signal so the - # SIGKILL attribute is never referenced on Windows. - if pgid is not None: - try: - import signal as _signal - os.killpg(pgid, _signal.SIGKILL) # windows-footgun: ok - except (ProcessLookupError, PermissionError, OSError): - pass - - # Belt-and-suspenders: ensure subprocess.communicate() sees the exit. - try: - proc.kill() - except (ProcessLookupError, OSError): - pass - - -def _run_one_file( - file: Path, - pytest_args: List[str], - repo_root: Path, - file_timeout: float, - retries: int = 0, -) -> Tuple[Path, int, str, dict[str, int], float]: - """Run ``python -m pytest `` in a fresh subprocess. - - Returns (file, returncode, captured_combined_output, summary_counts, subprocess_wall_seconds). - - ``retries`` > 0 enables the one-shot flake retry: a non-zero exit is - re-run in a fresh subprocess; if the re-run passes, the file counts as - passed but the output is prefixed with a FLAKY banner and the file/output - are recorded in ``_FLAKY_RESULTS`` so the summary can call it out. A - deterministic failure fails every attempt, so real regressions cannot - be laundered green. - - ``summary_counts`` is the result of ``_parse_pytest_summary(output)`` — - - pytest exit codes (https://docs.pytest.org/en/stable/reference/exit-codes.html): - 0 = all tests passed - 1 = some tests failed - 2 = test execution interrupted - 3 = internal error - 4 = pytest CLI usage error - 5 = no tests collected - - We treat exit 5 as a pass: it just means every test in the file was - skipped or filtered by a marker (e.g. ``-m 'not integration'`` skips - files where every test is marked integration). That's intentional and - not a failure mode. - - On per-file timeout (``file_timeout`` seconds) or any other exception - during ``communicate()``, we kill the whole process group / process - tree so grandchildren (uvicorn servers, async runtimes, etc.) do not - orphan onto PID 1. This outer timeout exists only to - bound a pathologically slow or hung file as a whole. - """ - file, rc, output, summary, subproc_wall = _run_one_file_once( - file, pytest_args, repo_root, file_timeout - ) - attempt = 0 - while rc != 0 and attempt < retries: - attempt += 1 - first_output = output - file, rc, output, summary, subproc_wall2 = _run_one_file_once( - file, pytest_args, repo_root, file_timeout - ) - subproc_wall += subproc_wall2 - if rc == 0: - output = ( - f"⚠ FLAKY: failed on attempt 1, passed on retry " - f"(attempt {attempt + 1}). Fix the flake — do not ignore this.\n" - f"--- first-attempt output ---\n{first_output}\n" - f"--- retry output ---\n{output}" - ) - with _flaky_lock: - _FLAKY_RESULTS.append((file, output)) - return file, rc, output, summary, subproc_wall - - -# Files that failed once and passed on retry, with both attempts' output. -# Keeping the traceback is load-bearing: a self-healed flake without its -# failing assertion is only a filename, which forces another expensive full -# run to rediscover the race. -_FLAKY_RESULTS: List[Tuple[Path, str]] = [] -_flaky_lock = threading.Lock() - - -def _run_one_file_once( - file: Path, - pytest_args: List[str], - repo_root: Path, - file_timeout: float, -) -> Tuple[Path, int, str, dict[str, int], float]: - """Single attempt of a per-file pytest subprocess (see _run_one_file).""" - cmd = [sys.executable, "-m", "pytest", str(file), *pytest_args] - - # Give this subprocess its own pytest temp root. - # - # pytest builds its tmp_path root as /pytest-of-/. At the - # end of a session it walks that directory with cleanup_dead_symlinks(). - # The walk lists the directory. Then it asks whether the `pytest-current` - # symlink resolves. Then it unlinks the symlink. - # - # Every file shared one root. A second process replaced that symlink - # between the question and the unlink. The first process then died with - # FileNotFoundError after all of its tests passed. - # - # The risk grows with the number of processes that finish together. At 8 - # workers it never occurred. At 144 workers it occurs. - # - # One root for each subprocess removes the shared directory that the race - # needs. The parent deletes the root after the attempt. - env = os.environ.copy() - temproot = tempfile.mkdtemp(prefix="hermes-pytest-tmproot-") - env["PYTEST_DEBUG_TEMPROOT"] = temproot - - subproc_start = time.monotonic() - # launch the pytest process - proc = subprocess.Popen( - cmd, - cwd=repo_root, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, encoding="utf-8", errors="replace", - env=env, - # POSIX: place the child at the head of its own process group so - # _kill_tree can SIGKILL the group atomically. - # Windows: this maps to CREATE_NEW_PROCESS_GROUP in CPython 3.12+; - # _kill_tree handles the Windows path via taskkill /F /T. - start_new_session=True, - ) - - # Capture the pgid NOW, before the leader can exit and be reaped. Once - # the leader is reaped, os.getpgid(proc.pid) raises ProcessLookupError - # even though grandchildren in that group are still alive — defeating - # the whole cleanup. None on Windows where the pgid concept doesn't apply. - pgid: int | None = None - if sys.platform != "win32": - try: - pgid = os.getpgid(proc.pid) - except (ProcessLookupError, PermissionError): - pgid = None - - try: - output, _ = proc.communicate(timeout=file_timeout) - rc = proc.returncode - except subprocess.TimeoutExpired: - _kill_tree(proc, pgid=pgid) - try: - output, _ = proc.communicate(timeout=10) - except subprocess.TimeoutExpired: - output = "(file timeout exceeded; output unavailable)" - rc = 124 # de facto convention for "killed by timeout". - output = ( - f"({file_timeout:.0f}s exceeded; " - f"process tree SIGKILL'd)\n{output}" - ) - except BaseException: - # KeyboardInterrupt / runner crash — make sure no zombie - # grandchildren outlive us. - _kill_tree(proc, pgid=pgid) - raise - else: - # Happy path: pytest exited on its own. Kill the group anyway in - # case it left grandchildren behind; already-dead is a no-op. - _kill_tree(proc, pgid=pgid) - - output += "\n" - finally: - # Delete the temp root for this attempt. Nothing reads it after the - # subprocess exits. More than 3000 of them fill the disk of the - # runner over one suite. - shutil.rmtree(temproot, ignore_errors=True) - - if rc == 5: - # No tests collected in THIS file — legitimate per-file: a - # platform-gated or fully-marker-filtered file (e.g. a win32-only - # suite on Linux) collects nothing and must not fail the suite. - # Tolerated here; the RUN-level guard in main() still fails when - # NOTHING was collected across every file, so a broken invocation - # (venv without pytest, -k that matches nothing) can't report green. - rc = 0 - summary = _parse_pytest_summary(output) - subproc_wall = time.monotonic() - subproc_start - return file, rc, output, summary, subproc_wall - - -def _parse_pytest_summary(output: str) -> dict[str, int]: - """Extract per-file test pass/fail/skip counts from pytest output. - - pytest prints a summary line like ``12 passed, 3 skipped, 1 failed in 2.1s`` - as the last non-empty line before the short test summary. We scrape that - line for the individual counts so the progress display can show test-level - granularity instead of just file-level pass/fail. - - Returns a dict with keys ``passed``, ``failed``, ``skipped``, ``errors``, - ``xfailed``, ``xpassed`` (only keys found in the output are present). - """ - result: dict[str, int] = {} - # Walk backwards from the end — the summary line is always near the tail. - for line in reversed(output.splitlines()): - line = line.strip() - if not line: - continue - # Match "N passed", "N failed", "N skipped", "N errors", "N xfailed", "N xpassed" - for m in re.finditer(r"(\d+)\s+(passed|failed|skipped|errors|xfailed|xpassed)", line): - result[m.group(2)] = int(m.group(1)) - # Also match "N error" (singular — pytest uses this sometimes). - for m in re.finditer(r"(\d+)\s+error\b", line): - result.setdefault("errors", result.get("errors", 0) + int(m.group(1))) - if result: - # Found the counts line — done. - break - # Stop at the short test summary header (if any) — everything above - # that is individual failure details, not the counts line. - if line.startswith("FAILED") or line.startswith("SHORT TEST SUMMARY"): - break - return result - - -def _format_file(file: Path, repo_root: Path) -> str: - """Render a test-file path for display: strip the repo-root prefix - when possible so output reads ``tests/acp/test_auth.py`` instead of - ``/home/runner/work/hermes-agent/hermes-agent/tests/acp/test_auth.py``. - - Falls back to the absolute path for anything outside the repo root. - """ - try: - return str(file.resolve().relative_to(repo_root.resolve())) - except ValueError: - return str(file) - - -def _print_progress( - tests_done: int, - approx_total_tests: int, - file: Path, - rc: int, - dur: float, - repo_root: Path, - tests_passed: int, - tests_failed: int, - test_counts: dict[Path, int], - file_summary: dict[str, int] | None = None, - subproc_wall: float | None = None, -) -> None: - """Single-line live progress. - - When ``file_summary`` is provided (parsed from pytest output), the - per-file parenthetical shows individual test pass/fail counts instead - of just the total test count. - - ``subproc_wall`` is the actual subprocess wall-clock time (excluding - queue-wait). When available, the display shows both the subprocess - time and the queue-inclusive elapsed time. - """ - status = "✓" if rc == 0 else "✗" - pct = min((tests_done / approx_total_tests * 100), 100) if approx_total_tests else 0 - # Digit width for left-side counter padding (derived from total file count). - fw = len(str(tests_passed + tests_failed)) - # Build per-file test count string. - if file_summary: - parts = [] - p = file_summary.get("passed", 0) - f = file_summary.get("failed", 0) - s = file_summary.get("skipped", 0) - e = file_summary.get("errors", 0) - if p: - parts.append(f"{p}✓") - if f: - parts.append(f"{f}✗") - if s: - parts.append(f"{s}s") - if e: - parts.append(f"{e}e") - # xfailed/xpassed are rare; include if present. - xf = file_summary.get("xfailed", 0) - xp = file_summary.get("xpassed", 0) - if xf: - parts.append(f"{xf}xf") - if xp: - parts.append(f"{xp}xp") - test_str = " ".join(parts) + ", " if parts else "" - else: - n_tests = test_counts.get(file, 0) - test_str = f"{n_tests} tests, " if n_tests else "" - # Show subprocess time when available; fall back to queue-inclusive dur. - if subproc_wall is not None: - time_str = f"{subproc_wall:.1f}s" - else: - time_str = f"{dur:.1f}s" - msg = ( - f"[{pct:5.1f}% | {tests_done:>5}/~{approx_total_tests}" - f" | ✓{tests_passed:>{fw}} | ✗{tests_failed:>{fw}}] " - f"{status} {_format_file(file, repo_root)} ({test_str}{time_str})" - ) - # Truncate to terminal width if available (no clobbering ANSI lines). - try: - cols = os.get_terminal_size().columns - if len(msg) > cols: - msg = msg[: cols - 1] + "…" - except OSError: - pass - print(msg, flush=True) - - -def _print_inline_failure( - file: Path, output: str, repo_root: Path, pytest_passthrough: List[str] -) -> None: - """Print a compact failure summary immediately when a file fails. - - Shows the tail of the pytest output (the failure section with stack - traces) and a ready-to-run repro command, so the developer doesn't - have to wait for the full run to finish before seeing what broke. - """ - rel = _format_file(file, repo_root) - # Build a repro command the developer can copy-paste. - passthrough_str = " ".join(pytest_passthrough) if pytest_passthrough else "" - repro = f"python -m pytest {rel}" - if passthrough_str: - repro += f" {passthrough_str}" - - # Grab just the failure lines (last ~30 lines of pytest output — - # typically the FAILED summary + short test info). - lines = output.rstrip().splitlines() - tail = "\n".join(lines[-30:]) - - print(flush=True) - print(f" ╔╍ Failed: {rel} ╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍", flush=True) - for line in tail.splitlines(): - print(f" ║ {line}", flush=True) - print(" ║", flush=True) - print(f" ║ Repro: {repro}", flush=True) - print(" ╚╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍╍", flush=True) - print(flush=True) - - -def _load_durations(repo_root: Path) -> dict[str, float]: - """Read the duration cache from the repo root. - - Returns a dict mapping relative file paths (e.g. - ``tests/tools/test_code_execution.py``) to wall-clock seconds from - the last run. Missing or corrupt file → empty dict (safe fallback). - """ - path = repo_root / _DURATIONS_FILE - if not path.is_file(): - return {} - try: - return json.loads(path.read_text(encoding="utf-8")) - except (json.JSONDecodeError, OSError) as e: - print("[ERROR] Failed to load json durations file! {e}") - return {} - - -def _save_durations( - file_times: List[Tuple[Path, float]], - repo_root: Path, -) -> None: - """Write the duration cache so future ``--slice`` runs can use it. - - Merges with any existing cache so entries from files not in the - current run (e.g. from a different slice) are preserved. Keys are - repo-relative paths so the cache is portable across checkouts - and CI runners. - """ - data: dict[str, float] = _load_durations(repo_root) - for f, t in file_times: - key = _format_file(f, repo_root) - data[key] = round(t, 3) - path = repo_root / _DURATIONS_FILE - path.write_text(json.dumps(data, indent=2, sort_keys=True) + "\n", encoding="utf-8") - - -def _compute_lpt_slices( - files: List[Path], - slice_count: int, - durations: dict[str, float], - repo_root: Path, -) -> List[List[Path]]: - """Distribute files across N slices using LPT (Longest Processing Time first). - - Sorts files by estimated duration descending, then greedily assigns each - file to the slice with the smallest accumulated time so far. This - minimizes the makespan (max slice duration) and keeps CI jobs balanced. - - Files with no cached duration get a default estimate of 2.0s (roughly - the P50 from profiling). This means first-time runs (no cache) still - get reasonable distribution, and new files don't all land in one slice. - - Returns a list of N file-lists, one per slice (0-indexed). - """ - if slice_count < 2: - return [files] - - default_dur = 2.0 - file_durs: List[Tuple[Path, float]] = [] - for f in files: - rel = _format_file(f, repo_root) - dur = durations.get(rel, default_dur) - file_durs.append((f, dur)) - - # Sort longest first (LPT). - file_durs.sort(key=lambda x: x[1], reverse=True) - - # Greedy assignment: for each file, add it to the slice with the - # smallest current total. - bucket_files: List[List[Path]] = [[] for _ in range(slice_count)] - bucket_totals: List[float] = [0.0] * slice_count - - for f, dur in file_durs: - min_idx = min(range(slice_count), key=lambda i: bucket_totals[i]) - bucket_files[min_idx].append(f) - bucket_totals[min_idx] += dur - - return bucket_files - - -def _slice_files( - files: List[Path], - slice_index: int, - slice_count: int, - durations: dict[str, float], - repo_root: Path, -) -> List[Path]: - """Return the subset of *files* belonging to slice *slice_index*. - - Uses :func:`_compute_lpt_slices` for LPT distribution. - - ``slice_index`` is 1-indexed (1..slice_count) for ergonomics — - ``--slice 1/4`` reads more naturally than ``--slice 0/4``. - """ - if slice_count < 2: - return files - if not (1 <= slice_index <= slice_count): - print( - f"error: --slice index must be 1..{slice_count}, got {slice_index}", - file=sys.stderr, - ) - sys.exit(2) - - bucket_files = _compute_lpt_slices(files, slice_count, durations, repo_root) - - target = bucket_files[slice_index - 1] - target_dur = sum( - durations.get(_format_file(f, repo_root), 2.0) for f in target - ) - total_dur = sum( - durations.get(_format_file(f, repo_root), 2.0) - for bucket in bucket_files - for f in bucket - ) - print( - f"Slice {slice_index}/{slice_count}: {len(target)} files " - f"(~{target_dur:.0f}s estimated of {total_dur:.0f}s total)", - flush=True, - ) - - return target - - -def _make_stdio_glyph_safe() -> None: - """Keep status glyphs from killing the runner on narrow console encodings. - - On native Windows, piped or legacy-console stdio defaults to a locale - codec (usually cp1252) that cannot encode the ✓/✗ progress glyphs — the - first per-file status line then dies with UnicodeEncodeError before a - single test result is reported. Declare the runner's own output UTF-8 - (what CI and every modern terminal already are), with errors="replace" - as the can't-crash backstop; where the encoding can't be changed, fall - back to errors="replace" alone so glyphs degrade to "?" instead of - killing the run. On already-UTF-8 stdio this is a no-op. - """ - for stream in (sys.stdout, sys.stderr): - reconfigure = getattr(stream, "reconfigure", None) - if reconfigure is None: - continue - try: - reconfigure(encoding="utf-8", errors="replace") - except Exception: - try: - reconfigure(errors="replace") - except Exception: - pass - - -def main() -> int: - _make_stdio_glyph_safe() - parser = argparse.ArgumentParser( - description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter, - ) - parser.add_argument( - "-j", - "--jobs", - type=int, - default=int(os.environ.get("HERMES_TEST_WORKERS") or (os.cpu_count() or 4)), - help="Parallel worker count (default: $HERMES_TEST_WORKERS or cpu_count)", - ) - parser.add_argument( - "--paths", - default=os.environ.get("HERMES_TEST_PATHS", ":".join(_DEFAULT_ROOTS)), - help=( - "Colon-separated discovery roots (default: 'tests'). On " - "Windows, ';' also separates and drive letters (C:\\...) are " - "kept intact." - ), - ) - parser.add_argument( - "--include-integration", - action="store_true", - help="Don't skip integration/ e2e/ during discovery", - ) - parser.add_argument( - "--file-timeout", - type=float, - default=float( - os.environ.get("HERMES_TEST_FILE_TIMEOUT", _DEFAULT_FILE_TIMEOUT_SECONDS) - ), - help=( - "Per-file wall-clock cap in seconds. On timeout, the pytest " - "subprocess and its full process tree are SIGKILL'd. " - f"Default: {_DEFAULT_FILE_TIMEOUT_SECONDS}s ({round(_DEFAULT_FILE_TIMEOUT_SECONDS/60)} min), env: HERMES_TEST_FILE_TIMEOUT." - ), - ) - parser.add_argument( - "--file-retries", - type=int, - default=int( - os.environ.get("HERMES_TEST_FILE_RETRIES", _DEFAULT_FILE_RETRIES) - ), - help=( - "Re-run a failing test FILE this many times in a fresh subprocess " - "before declaring it failed. A pass-on-retry counts as passed but " - "is reported as FLAKY in the summary. 0 disables. " - f"Default: {_DEFAULT_FILE_RETRIES}, env: HERMES_TEST_FILE_RETRIES." - ), - ) - parser.add_argument( - "--slice", - metavar="I/N", - help=( - "Run only slice I of N (e.g. --slice 1/4). " - "Files are distributed across slices using cached durations " - "so each slice takes roughly equal wall time. " - "Without a duration cache, files are distributed by count. " - "Env: HERMES_TEST_SLICE (format: I/N)." - ), - ) - parser.add_argument( - "--generate-slices", - metavar="N", - type=int, - help=( - "Discover test files, distribute them across N slices using " - "LPT on cached durations, and print a JSON matrix to stdout " - "then exit (no tests run). The JSON has the shape " - "'{\"slices\": [{\"index\": 1, \"files\": [\"tests/foo.py\", ...]}, ...]}' " - "so the CI generate job can feed it directly into a matrix." - ), - ) - parser.add_argument( - "--files", - metavar="LIST", - help=( - "Explicit colon-separated list of test files to run (on " - "Windows, ';' also separates and drive letters are kept " - "intact). Bypasses discovery entirely — used by CI matrix " - "jobs that receive their file list from the generate job." - ), - ) - parser.add_argument( - "paths_positional", - nargs="*", - metavar="PATH", - help=( - "Restrict discovery to these paths (directories or .py files). " - "Mutually exclusive with --paths. Anything after a literal '--' " - "separator is passed through to each per-file pytest invocation." - ), - ) - # Split argv into "our flags + positional paths" vs "pytest passthrough". - # - # Two ways to pass args through to the per-file pytest invocation: - # 1. Explicit ``--`` separator: everything after it goes to pytest. - # 2. Bare pytest flags anywhere before ``--``: any token starting with - # ``-`` that isn't one of OUR options is routed to pytest, so a bare - # ``-q`` / ``-v`` / ``-x`` / ``--tb=long`` / ``-k expr`` "just works" - # without the developer remembering the ``--``. This matches the - # docstring's promise and pytest muscle-memory. - # - # The subtlety bare-flag routing must handle: value-taking pytest flags - # given in space-separated form (``-k expr``, ``-m mark``, ``-p plugin``, - # ``-o name=val``). Naively, ``expr`` would look like a positional path and - # clobber discovery. We peel the following token along with such flags so - # it never reaches our positional ``paths``. ``=``-joined forms - # (``-k=expr``, ``--tb=long``) are self-contained and need no lookahead. - OUR_FLAGS = { - "-j", "--jobs", "--paths", "--include-integration", - "--file-timeout", "--file-retries", "--slice", "--generate-slices", "--files", - } - # pytest short flags that consume the NEXT token as their value. - PYTEST_VALUE_FLAGS = {"-k", "-m", "-p", "-o", "-c", "-r", "-W"} - - def _is_our_flag(tok: str) -> bool: - # Match exact (``-j``, ``--paths``), ``=``-joined (``--paths=x``), - # and attached short-value (``-j4``) forms of our own options. - if tok in OUR_FLAGS: - return True - head = tok.split("=", 1)[0] - if head in OUR_FLAGS: - return True - # Attached short value, e.g. ``-j4`` → ``-j``. - if len(tok) > 2 and tok[:2] in OUR_FLAGS and not tok[1] == "-": - return True - return False - - argv = sys.argv[1:] - if "--" in argv: - sep = argv.index("--") - before, explicit_passthrough = argv[:sep], argv[sep + 1 :] - else: - before, explicit_passthrough = argv, [] - - our_args: List[str] = [] - bare_passthrough: List[str] = [] - i = 0 - while i < len(before): - tok = before[i] - if tok.startswith("-") and not _is_our_flag(tok): - bare_passthrough.append(tok) - # Pull the value token for space-separated value flags. - if tok in PYTEST_VALUE_FLAGS and i + 1 < len(before): - bare_passthrough.append(before[i + 1]) - i += 2 - continue - else: - our_args.append(tok) - i += 1 - - args = parser.parse_args(our_args) - - # ── Node-id selectors → file + ``-k`` filter ──────────────────────────── - # This runner is FILE-granular: it spawns one ``pytest `` per test - # file. A pytest node id (``tests/foo.py::TestBar::test_baz``) is not an - # existing path, so discovery silently dropped it and the run exited with - # "No test files to run" — the selector looked accepted but nothing ran. - # Translate instead: run the FILE and narrow with ``-k`` on the last - # segment, which is what the caller meant. - node_id_selectors: List[Tuple[str, str]] = [] - if args.paths_positional: - translated: List[str] = [] - for raw in args.paths_positional: - if "::" not in raw: - translated.append(raw) - continue - file_part, _, selector = raw.partition("::") - leaf = selector.rsplit("::", 1)[-1] - # Strip a parametrized id (``test_x[case]``) down to the function - # name; ``-k`` matches substrings, and brackets are -k syntax. - leaf = leaf.split("[", 1)[0] - node_id_selectors.append((raw, leaf)) - translated.append(file_part) - if node_id_selectors: - args.paths_positional = translated - keys = [leaf for _, leaf in node_id_selectors] - expr = " or ".join(dict.fromkeys(keys)) - for raw, leaf in node_id_selectors: - print( - f"note: '{raw}' is a pytest node id; this runner is " - f"file-granular. Running the file with -k {leaf!r}.", - file=sys.stderr, - ) - # Only inject -k when the caller didn't pass one themselves; their - # explicit filter wins over our inferred one. - if not any( - t == "-k" or t.startswith("-k=") or (t.startswith("-k") and len(t) > 2) - for t in bare_passthrough + explicit_passthrough - ): - bare_passthrough = bare_passthrough + ["-k", expr] - - # Bare flags run before any explicit ``--`` passthrough so ordering is - # intuitive (``run_tests.sh tests/foo.py -q -- --tb=long`` → ``-q --tb=long``). - pytest_passthrough = bare_passthrough + explicit_passthrough - - # Parse --slice (or HERMES_TEST_SLICE) early so we can exit on bad input - # before doing any expensive discovery. - slice_raw = args.slice or os.environ.get("HERMES_TEST_SLICE") - slice_index: int | None = None - slice_count: int = 1 - if slice_raw: - try: - idx_s, count_s = slice_raw.split("/", 1) - slice_index = int(idx_s) - slice_count = int(count_s) - except (ValueError, AttributeError): - print(f"error: --slice must be I/N (e.g. 1/4), got: {slice_raw!r}", file=sys.stderr) - sys.exit(2) - - repo_root = Path(__file__).resolve().parent.parent - - # --files: explicit file list from the CI generate job — skip discovery. - if args.files: - files = [repo_root / f for f in _split_pathspec(args.files)] - roots = [] - else: - # Resolve discovery roots: positional path args override --paths if any - # were supplied, otherwise --paths (which itself defaults to 'tests'). - if args.paths_positional: - roots = [repo_root / p for p in args.paths_positional] - else: - roots = [repo_root / p for p in _split_pathspec(args.paths)] - - if args.include_integration: - # Caller takes responsibility — typically used via explicit -k filter. - global _SKIP_PARTS # noqa: PLW0603 — config knob - _SKIP_PARTS = set() - - files = _discover_files(roots) - - if not files: - print("No test files to run", file=sys.stderr) - return 1 - - # --generate-slices: compute LPT distribution and emit JSON, then exit. - if args.generate_slices is not None: - durations = _load_durations(repo_root) - slices = _compute_lpt_slices( - files, args.generate_slices, durations, repo_root - ) - matrix = { - "slice": [ - { - "index": i + 1, - "files": ":".join(_format_file(f, repo_root) for f in bucket), - } - for i, bucket in enumerate(slices) - ] - } - # Print to stdout so the CI step can capture it with $(). - print(json.dumps(matrix)) - return 0 - - # Count individual tests per file - test_counts = _approximately_count_tests(files, repo_root) - approx_total_tests = sum(test_counts.values()) - - # Apply slicing if requested — distribute files across CI jobs by - # estimated duration so no one job gets all the slow files. - if slice_index is not None: - durations = _load_durations(repo_root) - files = _slice_files(files, slice_index, slice_count, durations, repo_root) - # Recount after slicing. - test_counts = {f: test_counts[f] for f in files if f in test_counts} - approx_total_tests = sum(test_counts.values()) - - if roots: - roots_str = [str(r.relative_to(repo_root)) if r.is_relative_to(repo_root) else str(r) for r in roots] - print( - f"Discovered {len(files)} test files (~{approx_total_tests} tests) under " - f"{roots_str}; running with -j {args.jobs}", - flush=True, - ) - else: - print( - f"Running {len(files)} test files (~{approx_total_tests} tests) " - f"with -j {args.jobs}", - flush=True, - ) - - # Capture and print on completion (out-of-order is fine — keeps the - # terminal clean rather than interleaving N parallel pytest outputs). - failures: List[Tuple[Path, str, Dict[str, int]]] = [] - file_times: List[Tuple[Path, float]] = [] # (file, subprocess_wall) for distribution - started = time.monotonic() - files_done = 0 - tests_done = 0 - pass_count = 0 - fail_count = 0 - tests_passed = 0 - tests_failed = 0 - tests_skipped = 0 - # Every collected outcome, not just pass/fail: a legitimately all-skipped - # (platform-gated) file reports "2 skipped" and must NOT trip the - # nothing-ran guard, whereas a file that died before collection reports - # nothing at all and must. - tests_collected = 0 - lock = threading.Lock() - - def _on_done(file: Path, started_at: float, fut: "Future[Tuple[Path, int, str, Dict[str, int], float]]") -> None: - nonlocal files_done, tests_done, pass_count, fail_count, tests_passed, tests_failed, tests_skipped - nonlocal tests_collected - n_tests = test_counts.get(file, 0) - try: - fpath, rc, output, summary, subproc_wall = fut.result() - except Exception as exc: # noqa: BLE001 — must always advance counter - with lock: - files_done += 1 - tests_done += n_tests - fail_count += 1 - failures.append((file, f"runner crashed: {exc!r}", {})) - _print_progress( - tests_done, approx_total_tests, file, 1, - time.monotonic() - started_at, - repo_root, tests_passed, tests_failed, - test_counts, - subproc_wall=0.0, - ) - return - with lock: - files_done += 1 - tests_done += n_tests - # Accumulate test-level counts from parsed summary. - tests_passed += summary.get("passed", 0) - tests_failed += summary.get("failed", 0) - tests_skipped += summary.get("skipped", 0) - tests_collected += sum( - summary.get(k, 0) - for k in ("passed", "failed", "skipped", "errors", "xfailed", "xpassed") - ) - file_times.append((fpath, subproc_wall)) - if rc == 0: - pass_count += 1 - else: - fail_count += 1 - failures.append((fpath, output, summary)) - _print_progress( - tests_done, approx_total_tests, fpath, rc, - time.monotonic() - started_at, - repo_root, tests_passed, tests_failed, - test_counts, - file_summary=summary, - subproc_wall=subproc_wall, - ) - if rc != 0: - _print_inline_failure(fpath, output, repo_root, pytest_passthrough) - - with ThreadPoolExecutor(max_workers=args.jobs) as pool: - futures: List[Future] = [] - for file in files: - t0 = time.monotonic() - fut = pool.submit( - _run_one_file, file, pytest_passthrough, repo_root, - args.file_timeout, args.file_retries, - ) - fut.add_done_callback(lambda f, file=file, t0=t0: _on_done(file, t0, f)) - futures.append(fut) - # Block until everything's done. ThreadPoolExecutor.__exit__ waits - # for all submitted work, but doing it explicitly here makes the - # control flow obvious. - for fut in futures: - fut.result() if fut.exception() is None else None - - elapsed = time.monotonic() - started - print() - pct = min(100, (tests_done / approx_total_tests * 100)) if approx_total_tests else 0 - skipped_note = f", {tests_skipped} skipped" if tests_skipped else "" - print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===") - - # Host-OS gating note: tests marked for another OS were skipped by the - # conftest hook, not run. Say so explicitly — a green local run on Linux - # proves nothing about the macos_only/windows_only tests, and the reader - # should know where they DO run rather than misreading skips as coverage. - off_host = _off_host_marker_files(files) - if off_host: - print() - for marker, n in sorted(off_host.items()): - _, lane = _OS_MARKERS[marker] - print( - f" note: {marker} tests (in {n} file{'s' if n != 1 else ''}) were " - f"SKIPPED on this host ({sys.platform}); they run on {lane}." - ) - - # Zero tests collected across the WHOLE run is NOT a pass. Per-file rc=5 - # is deliberately tolerated above (platform-gated files), but if NOTHING - # ran anywhere the invocation itself was broken — a venv without pytest, a - # -k/-m filter that matched nothing, or collection erroring everywhere. - # The summary line above reads green at a glance ("0 failed ... 100% - # complete"), which has been misread as a successful verification, so say - # it plainly AND fail the exit code. - no_tests_ran_at_all = bool(files) and tests_collected == 0 - if no_tests_ran_at_all: - print() - print( - "=== ✗ NO TESTS RAN — 0 collected across " - f"{len(files)} file{'s' if len(files) != 1 else ''}. " - "This is NOT a pass. ===" - ) - print( - " Common causes: the selected venv has no pytest; a -k/-m filter " - "matched nothing; or collection errored in every file." - ) - print(" Check the per-file output above for the real error.") - - # Flaky files: failed once, passed on the automatic retry. Green, but - # loudly reported so they get fixed instead of silently re-flaking. - if _FLAKY_RESULTS: - print() - print(f"=== ⚠ {len(_FLAKY_RESULTS)} FLAKY file{'s' if len(_FLAKY_RESULTS) != 1 else ''} (failed once, passed on retry — fix these) ===") - for f, output in _FLAKY_RESULTS: - print(f" {_format_file(f, repo_root)}") - print(output.rstrip()) - - # Save durations for future --slice runs. Each slice writes its own - # partial test_durations.json; a CI merge step joins them later. - # Locally, _save_durations merges with any existing cache so entries - # from previous runs aren't lost. - if file_times: - _save_durations(file_times, repo_root) - print(f" Durations cached to {_DURATIONS_FILE} ({len(file_times)} files)") - - # Per-file time distribution (throwaway diagnostic — shows how - # subprocess time is distributed so we can see if startup dominates). - if file_times: - times = sorted([t for _, t in file_times]) - total_subproc = sum(times) - median_t = times[len(times) // 2] - p50 = median_t - p90 = times[int(len(times) * 0.90)] - p95 = times[int(len(times) * 0.95)] - p99 = times[min(int(len(times) * 0.99), len(times) - 1)] - max_t = times[-1] - # How many files finish in <1s? That's roughly "just startup". - fast = sum(1 for t in times if t < 1.0) - fast_2s = sum(1 for t in times if t < 2.0) - print() - print("=== Per-file subprocess time distribution ===") - print(f" Files: {len(times)}") - print(f" Total subprocess CPU-wall: {total_subproc:.1f}s (runner wall: {elapsed:.1f}s, parallelism: {args.jobs}x)") - print(f" P50: {p50:.2f}s P90: {p90:.2f}s P95: {p95:.2f}s P99: {p99:.2f}s Max: {max_t:.2f}s") - print(f" <1s: {fast} files ({fast/len(times)*100:.0f}%) <2s: {fast_2s} files ({fast_2s/len(times)*100:.0f}%)") - # Top 10 slowest files — likely the ones dragging the run. - slowest = sorted(file_times, key=lambda x: x[1], reverse=True)[:10] - print(" Top 10 slowest:") - for f, t in slowest: - print(f" {t:>6.2f}s {_format_file(f, repo_root)}") - - if failures: - print() - print("=== Failure output ===") - for file, output, _summary in failures: - print() - print(f"--- {_format_file(file, repo_root)} ---") - print(output.rstrip()) - print() - # Split: files with actual test failures vs non-zero exit for other reasons - test_fail_files = [(f, s) for f, _o, s in failures if s.get("failed", 0) > 0] - all_passed_but_nonzero = [(f, s) for f, _o, s in failures - if s.get("failed", 0) == 0 and s.get("passed", 0) > 0] - no_tests_ran = [(f, s) for f, _o, s in failures - if s.get("failed", 0) == 0 and s.get("passed", 0) == 0] - if test_fail_files: - total_tf = sum(s.get("failed", 0) for _, s in test_fail_files) - print(f"=== {len(test_fail_files)} file{'s' if len(test_fail_files) != 1 else ''} with test failures ({total_tf} test{'s' if total_tf != 1 else ''} failed) ===") - for file, s in test_fail_files: - nf = s.get("failed", 0) - print(f" {_format_file(file, repo_root)} ({nf} test{'s' if nf != 1 else ''} failed)") - if all_passed_but_nonzero: - print(f"=== {len(all_passed_but_nonzero)} file{'s' if len(all_passed_but_nonzero) != 1 else ''} where all tests passed but pytest exited non-zero (warnings-as-errors, hook failures, etc.) ===") - for file, s in all_passed_but_nonzero: - print(f" {_format_file(file, repo_root)} ({s.get('passed', 0)} passed)") - if no_tests_ran: - print(f"=== {len(no_tests_ran)} file{'s' if len(no_tests_ran) != 1 else ''} where no tests ran (collection/import error, timeout before collection, etc.) ===") - for file, s in no_tests_ran: - print(f" {_format_file(file, repo_root)}") - return 1 - - if no_tests_ran_at_all: - return 1 - - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/scripts/run_xdist.sh b/scripts/run_xdist.sh deleted file mode 100644 index 99bbf0acfd..0000000000 --- a/scripts/run_xdist.sh +++ /dev/null @@ -1,75 +0,0 @@ -#!/usr/bin/env bash -# xdist runner: persistent workers sharing one collection pass, instead of -# the per-file subprocess model. EXPERIMENT (PR #96458): the per-file model -# pays a spawn+import wall of ~0.5-1.5s per file x 3400 files (~6 min floor -# on Windows, the dominant share of the lane's runtime). xdist pays that -# import wall once per worker. -# -# TWO PHASES: -# -# 1. xdist -n N --dist loadfile over the whole suite EXCEPT the -# process-killer files below. loadfile keeps a file's tests on ONE -# worker, bounding cross-file pollution to co-scheduled files. -# -# 2. The killer files run SERIALLY, one plain pytest per file. These are -# the tests that spawn and kill REAL process trees (live_system_guard -# bypasses, venv-holder sweeps, taskkill /F /T, stale-process reaps). -# Under the per-file runner each ran alone in its own session -# (start_new_session), so a sweep could only hit its own children. -# Under shared xdist workers, those sweeps enumerate every python -# process — 32 sibling workers and the controller match — and the -# first CI xdist run died with "runner lost communication" (run -# 33389773498, 1h15m, no logs flushed). Serially, a sweep's only -# reachable victims are its own children. -# -# Failures in phase 1 are stateful-test bugs to fix, not runner bugs. -set -uo pipefail -cd "$(dirname "$0")/.." - -export PATH="/c/Program Files/Git/bin:$PATH" -unset PYTHONPATH PYTHONPYCACHEPREFIX HERMES_RUNTIME_DIR -export TZ=UTC LANG=C.UTF-8 LC_ALL=C.UTF-8 PYTHONHASHSEED=0 PYTHONUTF8=1 - -N="${1:-auto}" -shift || true - -# Tests that spawn/kill real process trees. Do NOT run these inside xdist -# workers — see the phase-2 comment above. -KILLERS=( - tests/cron/test_cron_script.py - tests/gateway/test_control_socket_windows_live.py - tests/gateway/test_replace_child_reap.py - tests/gateway/test_whatsapp_bridge_pidfile.py - tests/gateway/test_whatsapp_connect.py - tests/gateway/test_whatsapp_stale_bridge.py - tests/hermes_cli/test_dashboard_lifecycle_flags.py - tests/hermes_cli/test_gateway_windows.py - tests/hermes_cli/test_update_orphan_backend_reap.py - tests/hermes_cli/test_update_stale_dashboard.py - tests/test_install_autostash_conflict_recovery.py - tests/test_install_lockfile_churn.py - tests/test_install_ps1_venv_process_tree.py - tests/test_install_unmerged_index.py - tests/tools/test_process_registry.py -) - -IGNORES=() -for f in "${KILLERS[@]}"; do - [ -f "$f" ] && IGNORES+=(--ignore "$f") -done - -echo "▶ phase 1: xdist -n $N --dist loadfile (bulk)" -PYTEST_STATUS=0 -.venv/Scripts/python.exe -m pytest -n "$N" --dist loadfile \ - -p no:cacheprovider -m "not integration" -q --tb=line \ - "${IGNORES[@]}" "$@" || PYTEST_STATUS=$? - -echo "▶ phase 2: serial per-file (process-killer tests)" -for f in "${KILLERS[@]}"; do - [ -f "$f" ] || continue - echo " - $f" - .venv/Scripts/python.exe -m pytest "$f" -p no:cacheprovider \ - -m "not integration" -q --tb=line || PYTEST_STATUS=$? -done - -exit "$PYTEST_STATUS" diff --git a/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md b/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md index a578f06d25..3db9ed055a 100644 --- a/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md +++ b/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md @@ -86,7 +86,7 @@ run_conversation(): Use the canonical runner — it enforces CI-parity (hermetic `env -i`, unset credentials, TZ=UTC, per-file subprocess isolation via -`scripts/run_tests_parallel.py` — no xdist, worker count auto-scaled): +`scripts/run_tests.sh` — pytest-xdist, `--dist loadfile`, worker count = CPU count): ```bash scripts/run_tests.sh # full suite diff --git a/skills/software-development/python-debugpy/SKILL.md b/skills/software-development/python-debugpy/SKILL.md index c94dfdf526..b5279161fb 100644 --- a/skills/software-development/python-debugpy/SKILL.md +++ b/skills/software-development/python-debugpy/SKILL.md @@ -107,7 +107,7 @@ scripts/run_tests.sh tests/path/to/test_file.py::test_name --trace scripts/run_tests.sh tests/path/to/test_file.py --showlocals --tb=long ``` -Note: `scripts/run_tests.sh` runs each test file in a captured subprocess via `run_tests_parallel.py` (no xdist), so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: +Note: `scripts/run_tests.sh` runs pytest under xdist workers, so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: ```bash source .venv/bin/activate diff --git a/tests/ci/test_classify_changes.py b/tests/ci/test_classify_changes.py index 81e93d6809..8de4166465 100644 --- a/tests/ci/test_classify_changes.py +++ b/tests/ci/test_classify_changes.py @@ -176,10 +176,12 @@ CASES = { _lanes(python=True, scan=True), ), # Runner infrastructure is NOT tests-only — a bad runner edit can mask - # real failures, so it keeps the conservative full lane set. + # real failures, so it keeps the conservative full lane set. (scan is + # off for the .sh form: the supply-chain lanes scan executable .py/.pth + # payloads, and a shell script isn't one.) "test runner script → python_prod stays on": ( - ["scripts/run_tests_parallel.py"], - _lanes(python=True, scan=True), + ["scripts/run_tests.sh"], + _lanes(python=True), ), # Supply-chain lanes ".pth file → scan": (["evil.pth"], _lanes(python=True, scan=True)), diff --git a/tests/conftest.py b/tests/conftest.py index ffa5e7b928..24a7029e12 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -115,20 +115,18 @@ os.environ["HERMES_TEST_ISOLATION"] = os.environ.get("HERMES_HOME", "") or "1" HERMES_HOME_AT_CONFTEST_IMPORT = os.environ.get("HERMES_HOME", "") -# ── Per-file process isolation ────────────────────────────────────────────── -# Tests run via ``scripts/run_tests_parallel.py``, which spawns a fresh -# ``python -m pytest `` subprocess per test file. Cross-file state -# leakage (module-level dicts, ContextVars, caches) is impossible: each -# file gets a clean Python interpreter. Intra-file ordering is the test -# author's responsibility — if test A in foo.py mutates state that test B -# in foo.py reads, that's a real bug to fix in the file (it would also -# bite anyone running ``pytest tests/foo.py`` directly). +# ── File-level scheduling isolation ────────────────────────────────────────── +# Tests run via ``scripts/run_tests.sh`` — pytest-xdist with ``--dist +# loadfile``, which pins every test of a FILE to ONE worker. Cross-file +# state leakage is bounded to files co-scheduled on the same worker; the +# historic per-file-subprocess model gave full process isolation but paid +# a spawn+import wall per file that dominated Windows runtimes. Intra-file +# ordering is the test author's responsibility — if test A in foo.py +# mutates state that test B in foo.py reads, that's a real bug to fix in +# the file (it would also bite anyone running ``pytest tests/foo.py`` +# directly). # -# This replaces the historic _reset_module_state autouse fixture (manual -# state clearing) and the brief experiment with subprocess-per-test -# isolation (too slow at ~17k tests). -# -# See ``scripts/run_tests_parallel.py`` for the runner. +# See ``scripts/run_tests.sh`` for the runner. # ── Credential env-var filter ────────────────────────────────────────────── @@ -783,16 +781,14 @@ def _state_db_write_guard(request, monkeypatch): yield -# ── Module-level state reset — replaced by per-file process isolation ────── +# ── Module-level state reset — replaced by file-pinned xdist workers ──────── # -# Each test FILE runs in a freshly-spawned ``python -m pytest `` -# subprocess via ``scripts/run_tests_parallel.py``, so module-level dicts / -# sets / ContextVars from tests in one file cannot leak into tests in -# another file. No manual per-module clearing needed. -# -# Within a single file, ordering is the author's responsibility. If your -# tests in the same file share mutable state, either reset it explicitly -# in a fixture or split them across files. +# ``--dist loadfile`` pins each test FILE to ONE worker, so heavy +# co-scheduling pollution (module-level dicts / sets / ContextVars shared +# by many files) is bounded to files that land on the same worker. Within +# a single file, ordering is the author's responsibility. If your tests +# in the same file share mutable state, either reset it explicitly in a +# fixture or split them across files. # # The skill ``test-suite-cascade-diagnosis`` documents the cascade patterns # this replaces; the running example was ``test_command_guards`` failing diff --git a/tests/test_run_tests_parallel.py b/tests/test_run_tests_parallel.py deleted file mode 100644 index f83e9a5563..0000000000 --- a/tests/test_run_tests_parallel.py +++ /dev/null @@ -1,462 +0,0 @@ -"""Verify scripts/run_tests_parallel.py kills test-spawned grandchildren. - -Setup ------ -A test in this file spawns a long-lived Python grandchild that writes -its PID + a nonce to a tempfile, then exits without cleaning up. -With the old ``subprocess.run`` runner, that grandchild would orphan -and outlive the test (and the whole runner). With the current Popen + -``start_new_session`` + ``_kill_tree`` runner, the grandchild gets -SIGKILL'd via process-group kill when its file's pytest exits. - -The leaker test always passes — its only job is to spawn a grandchild -and walk away. The verifier runs the runner over the leaker file in a -subprocess, then waits for the grandchild PID to disappear from the -kernel's process table. - -POSIX-only: Windows has its own grandchild lifecycle (no shared session, -``taskkill /F /T`` semantics). Marked accordingly. -""" - -from __future__ import annotations - -import json -import os -import subprocess -import sys -import textwrap -import time -from pathlib import Path - -import pytest - - -# Both tests share the same handoff file: the leaker writes here, the -# verifier reads here. We park it in $TMPDIR with a unique-per-run name -# so concurrent invocations of the suite don't clobber each other. -_HANDOFF_DIR = Path(os.environ.get("TMPDIR", "/tmp")) / "hermes-isolation-probe" -_HANDOFF_DIR.mkdir(exist_ok=True) - - -def _handoff_path_for(nonce: str) -> Path: - return _HANDOFF_DIR / f"grandchild-{nonce}.json" - - -def _pid_alive(pid: int) -> bool: - """POSIX: send signal 0 to probe whether ``pid`` is still alive. - - ``os.kill(pid, 0)`` raises ``ProcessLookupError`` if the process is - gone, ``PermissionError`` if it exists but we can't signal it - (someone else's pid). We treat PermissionError as "alive" because - the process exists and that's all we need to know. - """ - if sys.platform == "win32": # pragma: no cover — POSIX-only test - # On Windows we'd use OpenProcess + GetExitCodeProcess; this - # test is skipped on Windows so the path is unreachable. - raise RuntimeError("_pid_alive POSIX-only") - try: - os.kill(pid, 0) - except ProcessLookupError: - return False - except PermissionError: - return True - return True - - -def test_progress_output_tolerates_legacy_stdout_encoding(tmp_path: Path) -> None: - """Progress glyphs must not crash the runner on non-UTF-8 consoles.""" - repo_root = Path(__file__).resolve().parent.parent - runner = repo_root / "scripts" / "run_tests_parallel.py" - - probe_dir = tmp_path / "probe" - probe_dir.mkdir() - probe = probe_dir / "test_probe_smoke.py" - probe.write_text("def test_smoke():\n assert True\n", encoding="utf-8") - - env = os.environ.copy() - env["PYTHONIOENCODING"] = "cp1252:strict" - - proc = subprocess.run( - [ - sys.executable, - str(runner), - "--paths", - str(probe_dir), - "-j", - "1", - "--file-timeout", - "30", - ], - cwd=repo_root, - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - timeout=60, - ) - - assert proc.returncode == 0, proc.stdout - assert "UnicodeEncodeError" not in proc.stdout - assert "1 tests passed" in proc.stdout - - -@pytest.mark.skipif(sys.platform == "win32", reason="POSIX-only probe") -@pytest.mark.live_system_guard_bypass -def test_grandchild_leak_is_killed_by_runner(tmp_path: Path) -> None: - """Run the parallel runner over a probe file and verify cleanup. - - 1. Materialize a probe file that spawns a long-lived grandchild and - writes its PID to disk before exiting. - 2. Invoke ``scripts/run_tests_parallel.py`` against the probe file. - 3. Wait for the grandchild PID to vanish (poll for ~5s). - 4. Assert the runner exited cleanly AND the grandchild is dead. - """ - repo_root = Path(__file__).resolve().parent.parent - runner = repo_root / "scripts" / "run_tests_parallel.py" - assert runner.exists(), f"runner missing at {runner}" - - # Probe lives in a temp dir, NOT under tests/, so the regular suite - # never picks it up — only our explicit invocation does. - probe_dir = tmp_path / "probe" - probe_dir.mkdir() - probe = probe_dir / "test_probe_leaker.py" - nonce = f"{os.getpid()}-{int(time.time() * 1000)}" - handoff = _handoff_path_for(nonce) - if handoff.exists(): - handoff.unlink() - - probe_src = textwrap.dedent(f""" - import json, os, subprocess, sys, time - from pathlib import Path - - HANDOFF = Path({str(handoff)!r}) - - def test_spawns_grandchild_and_walks_away(): - # Long-lived grandchild: detached, ignores SIGTERM (we want - # SIGKILL or process-group kill to be the only thing that - # works, simulating a misbehaving server). - child = subprocess.Popen( - [ - sys.executable, "-c", - "import os, signal, sys, time; " - "signal.signal(signal.SIGTERM, signal.SIG_IGN); " - "sys.stdout.write(f'gc-pgid={{os.getpgid(0)}} gc-pid={{os.getpid()}}\\\\n'); " - "sys.stdout.flush(); " - "time.sleep(600)", - ], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - # IMPORTANT: do NOT pass start_new_session here. We want - # the grandchild to inherit the pytest subprocess's - # process group, so when the runner kills the group the - # grandchild dies too. - ) - # Read the first line so we can record gc's pgid in the - # handoff, then walk away — don't close the pipe (would - # signal EOF and let the child see SIGPIPE on next write). - first_line = child.stdout.readline().decode().strip() - HANDOFF.write_text(json.dumps({{ - "pid": child.pid, - "diag": first_line, - "test_pid": os.getpid(), - "test_pgid": os.getpgid(0), - }})) - assert child.pid > 0 - """).strip() - probe.write_text(probe_src + "\n") - - # Run the parallel runner against just the probe file. The runner - # discovers under ``tests/`` by default, so we override via --paths. - proc = subprocess.run( - [ - sys.executable, - str(runner), - "--paths", - str(probe_dir), - "-j", - "1", - # Tight per-file timeout: the probe finishes in <1s, no - # need for 10min. - "--file-timeout", - "30", - ], - cwd=repo_root, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - # The runner declares its stdio UTF-8 (see _make_stdio_glyph_safe); - # decode the same way so ✓-glyph assertions hold on Windows, where - # text=True alone would decode with the locale codec (cp1252). - encoding="utf-8", - errors="replace", - timeout=60, - ) - - assert handoff.exists(), ( - f"probe never wrote handoff file; runner output:\n{proc.stdout}" - ) - handoff_data = json.loads(handoff.read_text()) - grandchild_pid = handoff_data["pid"] - diag = handoff_data.get("diag", "(no diag)") - test_pid = handoff_data.get("test_pid") - test_pgid = handoff_data.get("test_pgid") - handoff.unlink() - - # The runner must have exited cleanly (probe test passes). - assert proc.returncode == 0, ( - f"runner exited {proc.returncode}; output:\n{proc.stdout}" - ) - - # The grandchild must be gone. Poll for a bit because process-group - # SIGKILL + reaping isn't synchronous; on a loaded box it can take - # a beat. - deadline = time.monotonic() + 5.0 - while time.monotonic() < deadline: - if not _pid_alive(grandchild_pid): - break - time.sleep(0.05) - else: - # Test cleanup: kill the leaked grandchild ourselves so a - # FAILED assertion doesn't leave a sleep(600) running. - try: - os.kill(grandchild_pid, 9) - except ProcessLookupError: - pass - pytest.fail( - f"grandchild PID {grandchild_pid} survived runner exit; " - f"diag={diag!r} test_pid={test_pid} test_pgid={test_pgid}; " - f"runner output:\n{proc.stdout}" - ) - - -# ── Bare pytest-flag passthrough ───────────────────────────────────────────── -# -# The runner routes any token starting with ``-`` that isn't one of its own -# options (``-j``/``--jobs``, ``--paths``, ``--slice``, ``--file-timeout``, -# ``--generate-slices``, ``--files``, ``--include-integration``) straight -# through to each per-file pytest invocation — no ``--`` separator required. -# Before this, a bare ``-q`` errored out with "unrecognized arguments", -# forcing a retry on every run. These tests are behavior contracts, not -# snapshots: they assert that bare flags reach pytest and that value-taking -# flags (``-k expr``) keep their value instead of having it stolen by the -# positional-path discovery. - - -def _make_probe_dir(tmp_path: Path) -> Path: - """Two trivial passing tests, one named test_alpha, one test_beta.""" - probe_dir = tmp_path / "probe" - probe_dir.mkdir() - (probe_dir / "test_flagprobe.py").write_text( - "def test_alpha():\n assert True\n\n" - "def test_beta():\n assert True\n" - ) - return probe_dir - - -def _run_runner(probe_dir: Path, *extra: str) -> subprocess.CompletedProcess: - repo_root = Path(__file__).resolve().parent.parent - runner = repo_root / "scripts" / "run_tests_parallel.py" - return subprocess.run( - [sys.executable, str(runner), "--paths", str(probe_dir), - "-j", "1", "--file-timeout", "30", *extra], - cwd=repo_root, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - # The runner declares its stdio UTF-8 (see _make_stdio_glyph_safe); - # decode the same way so ✓-glyph assertions hold on Windows, where - # text=True alone would decode with the locale codec (cp1252). - encoding="utf-8", - errors="replace", - timeout=60, - ) - - - - -def test_bare_value_flag_keeps_its_value(tmp_path: Path) -> None: - """``-k test_alpha`` reaches pytest as a selector, not as a path. - - The value token (``test_alpha``) must NOT be swallowed by the runner's - positional-path discovery — if it were, discovery would look for a path - named ``test_alpha``, find nothing, and the run would degrade. We assert - the run succeeds AND only one of the two tests was selected (proving the - ``-k`` filter actually applied inside pytest). - """ - probe_dir = _make_probe_dir(tmp_path) - proc = _run_runner(probe_dir, "-k", "test_alpha") - assert proc.returncode == 0, proc.stdout - # Exactly one test selected: the per-file summary shows "1✓" (1 passed). - # test_beta is deselected by the -k filter. - assert "1✓" in proc.stdout or "1 passed" in proc.stdout, proc.stdout - assert "2✓" not in proc.stdout, ( - f"both tests ran — -k filter did not apply:\n{proc.stdout}" - ) - - - - -def test_positional_path_not_treated_as_flag(tmp_path: Path) -> None: - """A positional path arg still overrides discovery (not routed to pytest).""" - probe_dir = _make_probe_dir(tmp_path) - repo_root = Path(__file__).resolve().parent.parent - runner = repo_root / "scripts" / "run_tests_parallel.py" - # Pass the probe dir positionally (no --paths), plus a bare -q. - proc = subprocess.run( - [sys.executable, str(runner), str(probe_dir), "-j", "1", - "--file-timeout", "30", "-q"], - cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, - encoding="utf-8", errors="replace", timeout=60, - ) - assert proc.returncode == 0, proc.stdout - # Discovery found the probe file (2 tests), proving the positional path - # was consumed as a root, not forwarded to pytest as a bad flag. - assert "test_flagprobe.py" in proc.stdout, proc.stdout - - -def test_file_retry_self_heals_and_prints_both_attempts(tmp_path: Path) -> None: - """A pass-on-retry is green, loud, and retains the failing traceback.""" - repo_root = Path(__file__).resolve().parent.parent - runner = repo_root / "scripts" / "run_tests_parallel.py" - marker = tmp_path / "ran-once" - probe = tmp_path / "test_flaky_probe.py" - probe.write_text( - textwrap.dedent( - f""" - from pathlib import Path - - def test_flaky_once(): - marker = Path({str(marker)!r}) - if not marker.exists(): - marker.write_text("failed once") - assert False, "simulated first-attempt flake" - assert True - """ - ), - encoding="utf-8", - ) - - proc = subprocess.run( - [ - sys.executable, - str(runner), - "--files", - str(probe), - "--file-retries", - "1", - "-j", - "1", - "-q", - ], - cwd=repo_root, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - timeout=60, - ) - - assert proc.returncode == 0, proc.stdout - assert "FLAKY file" in proc.stdout - assert "simulated first-attempt flake" in proc.stdout - assert "first-attempt output" in proc.stdout - assert "retry output" in proc.stdout - - - - -# --------------------------------------------------------------------------- -# Zero-collection is not a pass; node ids are translated, not dropped. -# -# Both behaviors were real foot-guns: a run where NOTHING was collected printed -# "0 tests passed, 0 failed (100% complete)" (reads green), and a pytest node id -# (`file.py::Class::test`) was silently discarded by path discovery so the run -# ended with "No test files to run" while looking like an accepted selector. - - -def test_zero_collected_across_run_fails_and_says_so(tmp_path: Path) -> None: - """A -k that matches nothing must FAIL, not report a green summary.""" - probe_dir = _make_probe_dir(tmp_path) - proc = _run_runner(probe_dir, "-k", "zzz_matches_nothing") - assert proc.returncode == 1, proc.stdout - assert "NO TESTS RAN" in proc.stdout - assert "NOT a pass" in proc.stdout - - - - -def test_node_id_selector_runs_the_named_test(tmp_path: Path) -> None: - """``file.py::test_alpha`` runs that test instead of discovering nothing.""" - probe_dir = _make_probe_dir(tmp_path) - target = probe_dir / "test_flagprobe.py" - repo_root = Path(__file__).resolve().parent.parent - proc = subprocess.run( - [sys.executable, str(repo_root / "scripts" / "run_tests_parallel.py"), - f"{target}::test_alpha", "-j", "1", "--file-timeout", "30"], - cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, - text=True, timeout=60, - ) - assert proc.returncode == 0, proc.stdout - assert "No test files to run" not in proc.stdout - assert "node id" in proc.stdout # explains the translation - # Ran exactly the one selected test, not both in the file. - assert "1 tests passed" in proc.stdout - - -def test_explicit_k_wins_over_node_id_inference(tmp_path: Path) -> None: - """A caller's own ``-k`` is not overridden by the node-id translation.""" - probe_dir = _make_probe_dir(tmp_path) - target = probe_dir / "test_flagprobe.py" - repo_root = Path(__file__).resolve().parent.parent - proc = subprocess.run( - [sys.executable, str(repo_root / "scripts" / "run_tests_parallel.py"), - f"{target}::test_alpha", "-k", "test_beta", - "-j", "1", "--file-timeout", "30"], - cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, - text=True, timeout=60, - ) - # -k test_beta wins: one test ran, and it wasn't filtered to nothing. - assert proc.returncode == 0, proc.stdout - assert "1 tests passed" in proc.stdout - - -def test_multiple_absolute_paths_split_on_pathsep(tmp_path: Path) -> None: - """``--paths`` accepts ``os.pathsep``-joined absolute paths. - - On Windows the absolute paths contain drive-letter colons, so a naive - ``split(":")`` shreds them into phantom roots and only one (or neither) - of the two probe dirs would be discovered. - """ - dir_a = _make_probe_dir(tmp_path) - dir_b = tmp_path / "probe_b" - dir_b.mkdir() - (dir_b / "test_flagprobe_b.py").write_text( - "def test_gamma():\n assert True\n" - ) - repo_root = Path(__file__).resolve().parent.parent - runner = repo_root / "scripts" / "run_tests_parallel.py" - proc = subprocess.run( - [sys.executable, str(runner), - "--paths", os.pathsep.join([str(dir_a), str(dir_b)]), - "-j", "1", "--file-timeout", "30", "-q"], - cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, - encoding="utf-8", errors="replace", timeout=60, - ) - assert proc.returncode == 0, proc.stdout - assert "Discovered 2 test files" in proc.stdout, proc.stdout - - -@pytest.mark.skipif(sys.platform != "win32", reason="drive-letter paths") -def test_drive_letter_colon_is_not_a_path_separator(tmp_path: Path) -> None: - """An absolute ``--paths`` value stays one root on Windows. - - The naive split used to produce a phantom relative root ``'C'`` (the - drive letter) alongside the real path; discovery only worked by the - accident of ``repo_root / '\\rooted\\rest'`` re-anchoring onto the - repo's drive. - """ - probe_dir = _make_probe_dir(tmp_path) - proc = _run_runner(probe_dir, "-q") - assert proc.returncode == 0, proc.stdout - drive = str(probe_dir)[0] - assert f"['{drive}', " not in proc.stdout, ( - f"drive letter split off as a phantom root:\n{proc.stdout}" - ) - assert "Discovered 1 test files" in proc.stdout, proc.stdout diff --git a/tests/test_run_tests_parallel_stdio.py b/tests/test_run_tests_parallel_stdio.py deleted file mode 100644 index 2add57c3e2..0000000000 --- a/tests/test_run_tests_parallel_stdio.py +++ /dev/null @@ -1,69 +0,0 @@ -"""The runner's status glyphs must not crash narrow console encodings. - -On native Windows, piped or legacy-console stdio defaults to cp1252, which -cannot encode the runner's ✓/✗ progress glyphs — before the fix, the first -per-file status line killed the whole run with UnicodeEncodeError. The -failure depends only on the stream's encoding, so these tests pin it on -every OS by building a cp1252 stream explicitly. -""" - -from __future__ import annotations - -import importlib.util -import io -import sys -from pathlib import Path - -REPO_ROOT = Path(__file__).resolve().parents[1] -_RUNNER_PATH = REPO_ROOT / "scripts" / "run_tests_parallel.py" - - -def _load_runner(): - spec = importlib.util.spec_from_file_location("run_tests_parallel", _RUNNER_PATH) - mod = importlib.util.module_from_spec(spec) - spec.loader.exec_module(mod) - return mod - - -def _cp1252_stream() -> tuple[io.TextIOWrapper, io.BytesIO]: - raw = io.BytesIO() - return io.TextIOWrapper(raw, encoding="cp1252", errors="strict"), raw - - -def test_cp1252_stream_reproduces_the_crash_without_the_fix() -> None: - # Baseline for the bug: a strict cp1252 stream cannot take the glyph. - stream, _raw = _cp1252_stream() - try: - stream.write("✓") - except UnicodeEncodeError: - return - raise AssertionError("expected UnicodeEncodeError on strict cp1252") - - -def test_glyph_safe_stdio_survives_cp1252(monkeypatch) -> None: - mod = _load_runner() - stream, raw = _cp1252_stream() - monkeypatch.setattr(sys, "stdout", stream) - monkeypatch.setattr(sys, "stderr", stream) - - mod._make_stdio_glyph_safe() - print("✓ tests/foo.py (3 tests, 1.2s) ✗") - sys.stdout.flush() - - out = raw.getvalue() - assert "✓".encode("utf-8") in out, "stream should now carry UTF-8 glyphs" - assert b"tests/foo.py (3 tests, 1.2s)" in out, "line content must survive" - - -def test_glyph_safe_stdio_noop_without_reconfigure(monkeypatch) -> None: - # Streams without .reconfigure (e.g. pytest's capture buffers, plain - # StringIO) must pass through untouched instead of raising. - mod = _load_runner() - plain = io.StringIO() - monkeypatch.setattr(sys, "stdout", plain) - monkeypatch.setattr(sys, "stderr", plain) - - mod._make_stdio_glyph_safe() - print("✓ still fine") - - assert "✓ still fine" in plain.getvalue() diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md b/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md index e6f1120e08..491afaf9d7 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md @@ -125,7 +125,7 @@ scripts/run_tests.sh tests/path/to/test_file.py::test_name --trace scripts/run_tests.sh tests/path/to/test_file.py --showlocals --tb=long ``` -Note: `scripts/run_tests.sh` runs each test file in a captured subprocess via `run_tests_parallel.py` (no xdist), so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: +Note: `scripts/run_tests.sh` runs pytest under xdist workers, so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: ```bash source .venv/bin/activate