hermes-agent/tests/test_run_tests_parallel.py
teknium1 35b1e57862 fix(tests): a run that collects nothing can no longer look green
Three foot-guns in the canonical test runner, each of which cost real
debugging time by making an unverified run look verified.

1. Zero collection across the whole run reported success-shaped output.
   Per-file rc=5 is rewritten to rc=0 so a platform-gated file (every test
   skipped on this OS) doesn't fail the suite — correct, but it also meant a
   run where NOTHING was collected anywhere printed
   "0 tests passed, 0 failed (100% complete)" and, with no failures
   recorded, could exit 0. Now the run-level guard counts every collected
   outcome (passed/failed/skipped/errors/xfailed/xpassed): an all-skipped
   file still passes, but zero-collected-anywhere prints an explicit
   "✗ NO TESTS RAN — this is NOT a pass" block naming the likely causes and
   returns 1.

2. A venv without pytest was selected merely for existing. The probe
   accepted any directory with bin/activate, so in a checkout/worktree
   without a local .venv it picked the RELEASE venv
   (~/.hermes/hermes-agent/venv, no pytest). Every file then died with
   "No module named pytest" and the run reported 0 tests. Candidates are now
   import-checked for pytest — the same guard the HERMES_PYTHON fallback
   already applied — and a skipped candidate is named on stderr.

3. Pytest node ids were silently discarded. This runner is file-granular,
   so `tests/foo.py::TestBar::test_baz` isn't an existing path: discovery
   dropped it and the run ended "No test files to run" while the selector
   looked accepted. Node ids are now translated to the FILE plus an inferred
   `-k` on the leaf name (parametrized ids reduced to the function name),
   with a note explaining the translation. An explicit caller `-k` wins over
   the inferred one.

Tests: 4 behavior contracts in tests/test_run_tests_parallel.py. Verified
by sabotage — reverting the runner fails 3 of the 4 (the fourth pins the
pre-existing all-skipped tolerance so fix 1 can't regress it).
2026-07-25 07:09:22 -07:00

433 lines
16 KiB
Python

"""Verify scripts/run_tests_parallel.py kills test-spawned grandchildren.
Setup
-----
A test in this file spawns a long-lived Python grandchild that writes
its PID + a nonce to a tempfile, then exits without cleaning up.
With the old ``subprocess.run`` runner, that grandchild would orphan
and outlive the test (and the whole runner). With the current Popen +
``start_new_session`` + ``_kill_tree`` runner, the grandchild gets
SIGKILL'd via process-group kill when its file's pytest exits.
The leaker test always passes — its only job is to spawn a grandchild
and walk away. The verifier runs the runner over the leaker file in a
subprocess, then waits for the grandchild PID to disappear from the
kernel's process table.
POSIX-only: Windows has its own grandchild lifecycle (no shared session,
``taskkill /F /T`` semantics). Marked accordingly.
"""
from __future__ import annotations
import json
import os
import subprocess
import sys
import textwrap
import time
from pathlib import Path
import pytest
# Both tests share the same handoff file: the leaker writes here, the
# verifier reads here. We park it in $TMPDIR with a unique-per-run name
# so concurrent invocations of the suite don't clobber each other.
_HANDOFF_DIR = Path(os.environ.get("TMPDIR", "/tmp")) / "hermes-isolation-probe"
_HANDOFF_DIR.mkdir(exist_ok=True)
def _handoff_path_for(nonce: str) -> Path:
return _HANDOFF_DIR / f"grandchild-{nonce}.json"
def _pid_alive(pid: int) -> bool:
"""POSIX: send signal 0 to probe whether ``pid`` is still alive.
``os.kill(pid, 0)`` raises ``ProcessLookupError`` if the process is
gone, ``PermissionError`` if it exists but we can't signal it
(someone else's pid). We treat PermissionError as "alive" because
the process exists and that's all we need to know.
"""
if sys.platform == "win32": # pragma: no cover — POSIX-only test
# On Windows we'd use OpenProcess + GetExitCodeProcess; this
# test is skipped on Windows so the path is unreachable.
raise RuntimeError("_pid_alive POSIX-only")
try:
os.kill(pid, 0)
except ProcessLookupError:
return False
except PermissionError:
return True
return True
@pytest.mark.skipif(sys.platform == "win32", reason="POSIX-only probe")
@pytest.mark.live_system_guard_bypass
def test_grandchild_leak_is_killed_by_runner(tmp_path: Path) -> None:
"""Run the parallel runner over a probe file and verify cleanup.
1. Materialize a probe file that spawns a long-lived grandchild and
writes its PID to disk before exiting.
2. Invoke ``scripts/run_tests_parallel.py`` against the probe file.
3. Wait for the grandchild PID to vanish (poll for ~5s).
4. Assert the runner exited cleanly AND the grandchild is dead.
"""
repo_root = Path(__file__).resolve().parent.parent
runner = repo_root / "scripts" / "run_tests_parallel.py"
assert runner.exists(), f"runner missing at {runner}"
# Probe lives in a temp dir, NOT under tests/, so the regular suite
# never picks it up — only our explicit invocation does.
probe_dir = tmp_path / "probe"
probe_dir.mkdir()
probe = probe_dir / "test_probe_leaker.py"
nonce = f"{os.getpid()}-{int(time.time() * 1000)}"
handoff = _handoff_path_for(nonce)
if handoff.exists():
handoff.unlink()
probe_src = textwrap.dedent(f"""
import json, os, subprocess, sys, time
from pathlib import Path
HANDOFF = Path({str(handoff)!r})
def test_spawns_grandchild_and_walks_away():
# Long-lived grandchild: detached, ignores SIGTERM (we want
# SIGKILL or process-group kill to be the only thing that
# works, simulating a misbehaving server).
child = subprocess.Popen(
[
sys.executable, "-c",
"import os, signal, sys, time; "
"signal.signal(signal.SIGTERM, signal.SIG_IGN); "
"sys.stdout.write(f'gc-pgid={{os.getpgid(0)}} gc-pid={{os.getpid()}}\\\\n'); "
"sys.stdout.flush(); "
"time.sleep(600)",
],
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
# IMPORTANT: do NOT pass start_new_session here. We want
# the grandchild to inherit the pytest subprocess's
# process group, so when the runner kills the group the
# grandchild dies too.
)
# Read the first line so we can record gc's pgid in the
# handoff, then walk away — don't close the pipe (would
# signal EOF and let the child see SIGPIPE on next write).
first_line = child.stdout.readline().decode().strip()
HANDOFF.write_text(json.dumps({{
"pid": child.pid,
"diag": first_line,
"test_pid": os.getpid(),
"test_pgid": os.getpgid(0),
}}))
assert child.pid > 0
""").strip()
probe.write_text(probe_src + "\n")
# Run the parallel runner against just the probe file. The runner
# discovers under ``tests/`` by default, so we override via --paths.
proc = subprocess.run(
[
sys.executable,
str(runner),
"--paths",
str(probe_dir),
"-j",
"1",
# Tight per-file timeout: the probe finishes in <1s, no
# need for 10min.
"--file-timeout",
"30",
],
cwd=repo_root,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=60,
)
assert handoff.exists(), (
f"probe never wrote handoff file; runner output:\n{proc.stdout}"
)
handoff_data = json.loads(handoff.read_text())
grandchild_pid = handoff_data["pid"]
diag = handoff_data.get("diag", "(no diag)")
test_pid = handoff_data.get("test_pid")
test_pgid = handoff_data.get("test_pgid")
handoff.unlink()
# The runner must have exited cleanly (probe test passes).
assert proc.returncode == 0, (
f"runner exited {proc.returncode}; output:\n{proc.stdout}"
)
# The grandchild must be gone. Poll for a bit because process-group
# SIGKILL + reaping isn't synchronous; on a loaded box it can take
# a beat.
deadline = time.monotonic() + 5.0
while time.monotonic() < deadline:
if not _pid_alive(grandchild_pid):
break
time.sleep(0.05)
else:
# Test cleanup: kill the leaked grandchild ourselves so a
# FAILED assertion doesn't leave a sleep(600) running.
try:
os.kill(grandchild_pid, 9)
except ProcessLookupError:
pass
pytest.fail(
f"grandchild PID {grandchild_pid} survived runner exit; "
f"diag={diag!r} test_pid={test_pid} test_pgid={test_pgid}; "
f"runner output:\n{proc.stdout}"
)
# ── Bare pytest-flag passthrough ─────────────────────────────────────────────
#
# The runner routes any token starting with ``-`` that isn't one of its own
# options (``-j``/``--jobs``, ``--paths``, ``--slice``, ``--file-timeout``,
# ``--generate-slices``, ``--files``, ``--include-integration``) straight
# through to each per-file pytest invocation — no ``--`` separator required.
# Before this, a bare ``-q`` errored out with "unrecognized arguments",
# forcing a retry on every run. These tests are behavior contracts, not
# snapshots: they assert that bare flags reach pytest and that value-taking
# flags (``-k expr``) keep their value instead of having it stolen by the
# positional-path discovery.
def _make_probe_dir(tmp_path: Path) -> Path:
"""Two trivial passing tests, one named test_alpha, one test_beta."""
probe_dir = tmp_path / "probe"
probe_dir.mkdir()
(probe_dir / "test_flagprobe.py").write_text(
"def test_alpha():\n assert True\n\n"
"def test_beta():\n assert True\n"
)
return probe_dir
def _run_runner(probe_dir: Path, *extra: str) -> subprocess.CompletedProcess:
repo_root = Path(__file__).resolve().parent.parent
runner = repo_root / "scripts" / "run_tests_parallel.py"
return subprocess.run(
[sys.executable, str(runner), "--paths", str(probe_dir),
"-j", "1", "--file-timeout", "30", *extra],
cwd=repo_root,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=60,
)
def test_bare_q_flag_passes_through(tmp_path: Path) -> None:
"""A bare ``-q`` (no ``--``) runs clean instead of erroring out."""
probe_dir = _make_probe_dir(tmp_path)
proc = _run_runner(probe_dir, "-q")
assert proc.returncode == 0, proc.stdout
assert "unrecognized arguments" not in proc.stdout
def test_bare_value_flag_keeps_its_value(tmp_path: Path) -> None:
"""``-k test_alpha`` reaches pytest as a selector, not as a path.
The value token (``test_alpha``) must NOT be swallowed by the runner's
positional-path discovery — if it were, discovery would look for a path
named ``test_alpha``, find nothing, and the run would degrade. We assert
the run succeeds AND only one of the two tests was selected (proving the
``-k`` filter actually applied inside pytest).
"""
probe_dir = _make_probe_dir(tmp_path)
proc = _run_runner(probe_dir, "-k", "test_alpha")
assert proc.returncode == 0, proc.stdout
# Exactly one test selected: the per-file summary shows "1✓" (1 passed).
# test_beta is deselected by the -k filter.
assert "1✓" in proc.stdout or "1 passed" in proc.stdout, proc.stdout
assert "2✓" not in proc.stdout, (
f"both tests ran — -k filter did not apply:\n{proc.stdout}"
)
def test_explicit_double_dash_still_works(tmp_path: Path) -> None:
"""The legacy ``--`` separator keeps working alongside bare flags."""
probe_dir = _make_probe_dir(tmp_path)
proc = _run_runner(probe_dir, "-q", "--", "--tb=short")
assert proc.returncode == 0, proc.stdout
assert "unrecognized arguments" not in proc.stdout
def test_positional_path_not_treated_as_flag(tmp_path: Path) -> None:
"""A positional path arg still overrides discovery (not routed to pytest)."""
probe_dir = _make_probe_dir(tmp_path)
repo_root = Path(__file__).resolve().parent.parent
runner = repo_root / "scripts" / "run_tests_parallel.py"
# Pass the probe dir positionally (no --paths), plus a bare -q.
proc = subprocess.run(
[sys.executable, str(runner), str(probe_dir), "-j", "1",
"--file-timeout", "30", "-q"],
cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, timeout=60,
)
assert proc.returncode == 0, proc.stdout
# Discovery found the probe file (2 tests), proving the positional path
# was consumed as a root, not forwarded to pytest as a bad flag.
assert "test_flagprobe.py" in proc.stdout, proc.stdout
def test_file_retry_self_heals_and_prints_both_attempts(tmp_path: Path) -> None:
"""A pass-on-retry is green, loud, and retains the failing traceback."""
repo_root = Path(__file__).resolve().parent.parent
runner = repo_root / "scripts" / "run_tests_parallel.py"
marker = tmp_path / "ran-once"
probe = tmp_path / "test_flaky_probe.py"
probe.write_text(
textwrap.dedent(
f"""
from pathlib import Path
def test_flaky_once():
marker = Path({str(marker)!r})
if not marker.exists():
marker.write_text("failed once")
assert False, "simulated first-attempt flake"
assert True
"""
),
encoding="utf-8",
)
proc = subprocess.run(
[
sys.executable,
str(runner),
"--files",
str(probe),
"--file-retries",
"1",
"-j",
"1",
"-q",
],
cwd=repo_root,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=60,
)
assert proc.returncode == 0, proc.stdout
assert "FLAKY file" in proc.stdout
assert "simulated first-attempt flake" in proc.stdout
assert "first-attempt output" in proc.stdout
assert "retry output" in proc.stdout
def test_file_retry_does_not_launder_deterministic_failure(tmp_path: Path) -> None:
"""A real regression fails both attempts and the runner remains red."""
repo_root = Path(__file__).resolve().parent.parent
runner = repo_root / "scripts" / "run_tests_parallel.py"
probe = tmp_path / "test_red_probe.py"
probe.write_text(
"def test_always_red():\n assert False, 'deterministic regression'\n",
encoding="utf-8",
)
proc = subprocess.run(
[
sys.executable,
str(runner),
"--files",
str(probe),
"--file-retries",
"1",
"-j",
"1",
"-q",
],
cwd=repo_root,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=60,
)
assert proc.returncode == 1, proc.stdout
assert "deterministic regression" in proc.stdout
assert "FLAKY file" not in proc.stdout
# ---------------------------------------------------------------------------
# Zero-collection is not a pass; node ids are translated, not dropped.
#
# Both behaviors were real foot-guns: a run where NOTHING was collected printed
# "0 tests passed, 0 failed (100% complete)" (reads green), and a pytest node id
# (`file.py::Class::test`) was silently discarded by path discovery so the run
# ended with "No test files to run" while looking like an accepted selector.
def test_zero_collected_across_run_fails_and_says_so(tmp_path: Path) -> None:
"""A -k that matches nothing must FAIL, not report a green summary."""
probe_dir = _make_probe_dir(tmp_path)
proc = _run_runner(probe_dir, "-k", "zzz_matches_nothing")
assert proc.returncode == 1, proc.stdout
assert "NO TESTS RAN" in proc.stdout
assert "NOT a pass" in proc.stdout
def test_all_skipped_file_is_still_a_pass(tmp_path: Path) -> None:
"""Per-file zero-collection stays tolerated.
A platform-gated file (every test skipped) reports "N skipped" — collected,
just not executed — and must NOT trip the nothing-ran guard.
"""
probe_dir = tmp_path / "skipprobe"
probe_dir.mkdir()
(probe_dir / "test_allskipped.py").write_text(
"import pytest\n\n"
"pytestmark = pytest.mark.skip(reason='platform-gated')\n\n"
"def test_one():\n assert True\n\n"
"def test_two():\n assert True\n"
)
proc = _run_runner(probe_dir)
assert proc.returncode == 0, proc.stdout
assert "NO TESTS RAN" not in proc.stdout
def test_node_id_selector_runs_the_named_test(tmp_path: Path) -> None:
"""``file.py::test_alpha`` runs that test instead of discovering nothing."""
probe_dir = _make_probe_dir(tmp_path)
target = probe_dir / "test_flagprobe.py"
repo_root = Path(__file__).resolve().parent.parent
proc = subprocess.run(
[sys.executable, str(repo_root / "scripts" / "run_tests_parallel.py"),
f"{target}::test_alpha", "-j", "1", "--file-timeout", "30"],
cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, timeout=60,
)
assert proc.returncode == 0, proc.stdout
assert "No test files to run" not in proc.stdout
assert "node id" in proc.stdout # explains the translation
# Ran exactly the one selected test, not both in the file.
assert "1 tests passed" in proc.stdout
def test_explicit_k_wins_over_node_id_inference(tmp_path: Path) -> None:
"""A caller's own ``-k`` is not overridden by the node-id translation."""
probe_dir = _make_probe_dir(tmp_path)
target = probe_dir / "test_flagprobe.py"
repo_root = Path(__file__).resolve().parent.parent
proc = subprocess.run(
[sys.executable, str(repo_root / "scripts" / "run_tests_parallel.py"),
f"{target}::test_alpha", "-k", "test_beta",
"-j", "1", "--file-timeout", "30"],
cwd=repo_root, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, timeout=60,
)
# -k test_beta wins: one test ran, and it wasn't filtered to nothing.
assert proc.returncode == 0, proc.stdout
assert "1 tests passed" in proc.stdout