diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py index 602e033f3d6..09b585ac970 100755 --- a/scripts/run_tests_parallel.py +++ b/scripts/run_tests_parallel.py @@ -85,6 +85,15 @@ _SKIP_PARTS = {"integration", "e2e", "docker"} # time while keeping a genuinely hung file bounded. _DEFAULT_FILE_TIMEOUT_SECONDS = 300.0 +# One-shot retry of failing test FILES. A file that exits non-zero is re-run +# once in a fresh subprocess; if the re-run passes, the file counts as passed +# but is loudly reported as FLAKY so it gets fixed rather than hidden. +# Deterministic failures fail both attempts — a real regression can never be +# laundered into green by this (it would have to flake in our favor twice in +# a row on the same runner, which is exactly the definition of a flake). +# Set to 0 to disable (env: HERMES_TEST_FILE_RETRIES). +_DEFAULT_FILE_RETRIES = 1 + # Duration cache: maps relative file paths to last-observed subprocess # wall-clock seconds. Used by ``--slice`` to distribute files across # CI jobs by estimated total time, so no one job gets all the slow files. @@ -224,11 +233,19 @@ def _run_one_file( pytest_args: List[str], repo_root: Path, file_timeout: float, + retries: int = 0, ) -> Tuple[Path, int, str, dict[str, int], float]: """Run ``python -m pytest `` in a fresh subprocess. Returns (file, returncode, captured_combined_output, summary_counts, subprocess_wall_seconds). + ``retries`` > 0 enables the one-shot flake retry: a non-zero exit is + re-run in a fresh subprocess; if the re-run passes, the file counts as + passed but the output is prefixed with a FLAKY banner and the file is + recorded in ``_FLAKY_FILES`` so the summary can call it out. A + deterministic failure fails every attempt, so real regressions cannot + be laundered green. + ``summary_counts`` is the result of ``_parse_pytest_summary(output)`` — pytest exit codes (https://docs.pytest.org/en/stable/reference/exit-codes.html): @@ -250,6 +267,41 @@ def _run_one_file( orphan onto PID 1. This outer timeout exists only to bound a pathologically slow or hung file as a whole. """ + file, rc, output, summary, subproc_wall = _run_one_file_once( + file, pytest_args, repo_root, file_timeout + ) + attempt = 0 + while rc != 0 and attempt < retries: + attempt += 1 + first_output = output + file, rc, output, summary, subproc_wall2 = _run_one_file_once( + file, pytest_args, repo_root, file_timeout + ) + subproc_wall += subproc_wall2 + if rc == 0: + with _flaky_lock: + _FLAKY_FILES.append(file) + output = ( + f"⚠ FLAKY: failed on attempt 1, passed on retry " + f"(attempt {attempt + 1}). Fix the flake — do not ignore this.\n" + f"--- first-attempt output ---\n{first_output}\n" + f"--- retry output ---\n{output}" + ) + return file, rc, output, summary, subproc_wall + + +# Files that failed once and passed on retry — reported in the summary. +_FLAKY_FILES: List[Path] = [] +_flaky_lock = threading.Lock() + + +def _run_one_file_once( + file: Path, + pytest_args: List[str], + repo_root: Path, + file_timeout: float, +) -> Tuple[Path, int, str, dict[str, int], float]: + """Single attempt of a per-file pytest subprocess (see _run_one_file).""" cmd = [sys.executable, "-m", "pytest", str(file), *pytest_args] subproc_start = time.monotonic() @@ -624,6 +676,19 @@ def main() -> int: f"Default: {_DEFAULT_FILE_TIMEOUT_SECONDS}s ({round(_DEFAULT_FILE_TIMEOUT_SECONDS/60)} min), env: HERMES_TEST_FILE_TIMEOUT." ), ) + parser.add_argument( + "--file-retries", + type=int, + default=int( + os.environ.get("HERMES_TEST_FILE_RETRIES", _DEFAULT_FILE_RETRIES) + ), + help=( + "Re-run a failing test FILE this many times in a fresh subprocess " + "before declaring it failed. A pass-on-retry counts as passed but " + "is reported as FLAKY in the summary. 0 disables. " + f"Default: {_DEFAULT_FILE_RETRIES}, env: HERMES_TEST_FILE_RETRIES." + ), + ) parser.add_argument( "--slice", metavar="I/N", @@ -684,7 +749,7 @@ def main() -> int: # (``-k=expr``, ``--tb=long``) are self-contained and need no lookahead. OUR_FLAGS = { "-j", "--jobs", "--paths", "--include-integration", - "--file-timeout", "--slice", "--generate-slices", "--files", + "--file-timeout", "--file-retries", "--slice", "--generate-slices", "--files", } # pytest short flags that consume the NEXT token as their value. PYTEST_VALUE_FLAGS = {"-k", "-m", "-p", "-o", "-c", "-r", "-W"} @@ -875,7 +940,8 @@ def main() -> int: for file in files: t0 = time.monotonic() fut = pool.submit( - _run_one_file, file, pytest_passthrough, repo_root, args.file_timeout + _run_one_file, file, pytest_passthrough, repo_root, + args.file_timeout, args.file_retries, ) fut.add_done_callback(lambda f, file=file, t0=t0: _on_done(file, t0, f)) futures.append(fut) @@ -890,6 +956,14 @@ def main() -> int: pct = min(100, (tests_done / approx_total_tests * 100)) if approx_total_tests else 0 print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===") + # Flaky files: failed once, passed on the automatic retry. Green, but + # loudly reported so they get fixed instead of silently re-flaking. + if _FLAKY_FILES: + print() + print(f"=== ⚠ {len(_FLAKY_FILES)} FLAKY file{'s' if len(_FLAKY_FILES) != 1 else ''} (failed once, passed on retry — fix these) ===") + for f in _FLAKY_FILES: + print(f" {_format_file(f, repo_root)}") + # Save durations for future --slice runs. Each slice writes its own # partial test_durations.json; a CI merge step joins them later. # Locally, _save_durations merges with any existing cache so entries