fix(ci): report an interpreter crash as CRASHED, not "no tests ran"

When a per-file pytest subprocess dies by signal (the sqlite cross-thread
close in #113186 was a SIGSEGV after every test had passed), faulthandler
prints "Fatal Python error: Segmentation fault" and no summary line, so
every count parses to 0. The runner filed that under "1 file where no
tests ran (collection/import error, ...)" beneath a summary that read
"0 failed" and exited 1 — two wrong diagnoses for one real bug, and it
was misread as a runner problem twice on main.

The runner now detects a signal death or a "Fatal Python error:" banner,
prefixes the captured output with the diagnosis (same convention as the
timeout path), marks the progress line CRASHED, counts "N files CRASHED"
on the summary line, lists the file in its own failure bucket, and no
longer trips the "NO TESTS RAN" guard for a crash that ran tests. The
flake retry already covers crashes (any non-zero rc), so nothing changes
there.
This commit is contained in:
teknium1
2026-09-18 19:33:52 -07:00
committed by Teknium
parent 1af1db281e
commit 0ddba07ad7
2 changed files with 92 additions and 6 deletions

View File

@@ -522,6 +522,12 @@ def _run_one_file_once(
# (venv without pytest, -k that matches nothing) can't report green.
rc = 0
summary = _parse_pytest_summary(output)
crash = _describe_interpreter_crash(rc, output) if rc != 0 else None
if crash:
# Same convention as the timeout path: the diagnosis leads the
# captured output, so the failure dump reads correctly on its own.
summary["crashed"] = 1
output = f"(interpreter crashed: {crash})\n{output}"
subproc_wall = time.monotonic() - subproc_start
return file, rc, output, summary, subproc_wall
@@ -559,6 +565,35 @@ def _parse_pytest_summary(output: str) -> dict[str, int]:
return result
def _describe_interpreter_crash(rc: int, output: str) -> Optional[str]:
"""Return a one-line description when the pytest subprocess died instead of exiting.
A native fault (sqlite stepping a connection another thread closed,
#113186) kills the interpreter mid-file: faulthandler prints ``Fatal
Python error: Segmentation fault`` and the process dies by signal, so
there is no summary line and every count parses to 0. Without this the
file is reported as "no tests ran (collection/import error)" under a
summary that says ``0 failed`` — the wrong diagnosis in both places.
"""
fatal = next(
(line.strip() for line in output.splitlines() if "Fatal Python error:" in line),
None,
)
if fatal:
# The faulthandler banner is appended to the progress dots of the
# last test; keep only the banner.
fatal = fatal[fatal.index("Fatal Python error:"):]
if rc < 0:
import signal as _signal
try:
name = _signal.Signals(-rc).name
except ValueError:
name = f"signal {-rc}"
return f"{fatal} ({name})" if fatal else f"killed by {name}"
return fatal
def _format_file(file: Path, repo_root: Path) -> str:
"""Render a test-file path for display: strip the repo-root prefix
when possible so output reads ``tests/acp_adapter/test_auth.py`` instead of
@@ -621,6 +656,8 @@ def _print_progress(
parts.append(f"{xf}xf")
if xp:
parts.append(f"{xp}xp")
if file_summary.get("crashed"):
parts.append("CRASHED")
test_str = " ".join(parts) + ", " if parts else ""
else:
n_tests = test_counts.get(file, 0)
@@ -1165,11 +1202,12 @@ def main() -> int:
# nothing-ran guard, whereas a file that died before collection reports
# nothing at all and must.
tests_collected = 0
files_crashed = 0
lock = threading.Lock()
def _on_done(file: Path, started_at: float, fut: "Future[Tuple[Path, int, str, Dict[str, int], float]]") -> None:
nonlocal files_done, tests_done, pass_count, fail_count, tests_passed, tests_failed, tests_skipped
nonlocal tests_collected
nonlocal tests_collected, files_crashed
n_tests = test_counts.get(file, 0)
try:
fpath, rc, output, summary, subproc_wall = fut.result()
@@ -1194,6 +1232,7 @@ def main() -> int:
tests_passed += summary.get("passed", 0)
tests_failed += summary.get("failed", 0)
tests_skipped += summary.get("skipped", 0)
files_crashed += summary.get("crashed", 0)
tests_collected += sum(
summary.get(k, 0)
for k in ("passed", "failed", "skipped", "errors", "xfailed", "xpassed")
@@ -1242,7 +1281,13 @@ def main() -> int:
print()
pct = min(100, (tests_done / approx_total_tests * 100)) if approx_total_tests else 0
skipped_note = f", {tests_skipped} skipped" if tests_skipped else ""
print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===")
# A crashed interpreter has no failed-test count; say so on the one line
# everyone reads, or "0 failed" + exit 1 looks like a runner bug.
crashed_note = (
f", {files_crashed} file{'s' if files_crashed != 1 else ''} CRASHED"
if files_crashed else ""
)
print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{crashed_note}{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===")
# Host-OS gating note: tests marked for another OS were skipped by the
# conftest hook, not run. Say so explicitly — a green local run on Linux
@@ -1265,7 +1310,7 @@ def main() -> int:
# The summary line above reads green at a glance ("0 failed ... 100%
# complete"), which has been misread as a successful verification, so say
# it plainly AND fail the exit code.
no_tests_ran_at_all = bool(files) and tests_collected == 0
no_tests_ran_at_all = bool(files) and tests_collected == 0 and not files_crashed
if no_tests_ran_at_all:
print()
print(
@@ -1333,11 +1378,17 @@ def main() -> int:
print(output.rstrip())
print()
# Split: files with actual test failures vs non-zero exit for other reasons
test_fail_files = [(f, s) for f, _o, s in failures if s.get("failed", 0) > 0]
all_passed_but_nonzero = [(f, s) for f, _o, s in failures
crashed_files = [(f, o, s) for f, o, s in failures if s.get("crashed")]
rest = [(f, s) for f, _o, s in failures if not s.get("crashed")]
test_fail_files = [(f, s) for f, s in rest if s.get("failed", 0) > 0]
all_passed_but_nonzero = [(f, s) for f, s in rest
if s.get("failed", 0) == 0 and s.get("passed", 0) > 0]
no_tests_ran = [(f, s) for f, _o, s in failures
no_tests_ran = [(f, s) for f, s in rest
if s.get("failed", 0) == 0 and s.get("passed", 0) == 0]
if crashed_files:
print(f"=== {len(crashed_files)} file{'s' if len(crashed_files) != 1 else ''} where the interpreter CRASHED mid-run (native fault — a real bug, not a collection error; the tests that did run are not counted) ===")
for file, output, _s in crashed_files:
print(f" {_format_file(file, repo_root)} {output.splitlines()[0]}")
if test_fail_files:
total_tf = sum(s.get("failed", 0) for _, s in test_fail_files)
print(f"=== {len(test_fail_files)} file{'s' if len(test_fail_files) != 1 else ''} with test failures ({total_tf} test{'s' if total_tf != 1 else ''} failed) ===")

View File

@@ -516,3 +516,38 @@ def test_drive_letter_colon_is_not_a_path_separator(tmp_path: Path) -> None:
f"drive letter split off as a phantom root:\n{proc.stdout}"
)
assert "Discovered 1 test files" in proc.stdout, proc.stdout
@pytest.mark.skipif(sys.platform == "win32", reason="POSIX signal death; Windows has no SIGSEGV exit")
def test_interpreter_crash_is_reported_as_a_crash_not_as_no_tests_ran(tmp_path: Path) -> None:
"""A file whose interpreter dies by signal is classified as CRASHED (#113186).
A native fault after some tests passed leaves no pytest summary line, so
every count parses to 0. The runner used to file that under "no tests ran
(collection/import error)" beneath a summary reading ``0 failed`` — two
wrong diagnoses for one real bug. The crash must be named on the summary
line and in the failure buckets, and the run must still exit non-zero.
"""
probe_dir = tmp_path / "probe"
probe_dir.mkdir()
(probe_dir / "test_probe_crash.py").write_text(
textwrap.dedent(
"""
import os, signal
def test_before():
assert True
def test_crash():
os.kill(os.getpid(), signal.SIGSEGV)
"""
)
)
proc = _run_runner(probe_dir, "--file-retries", "0")
assert proc.returncode != 0
assert "1 file CRASHED" in proc.stdout
assert "SIGSEGV" in proc.stdout
assert "where no tests ran" not in proc.stdout
assert "NO TESTS RAN" not in proc.stdout