diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py index 5e20caf3cb..9f87f6b75c 100755 --- a/scripts/run_tests_parallel.py +++ b/scripts/run_tests_parallel.py @@ -303,6 +303,36 @@ def _kill_tree(proc: "subprocess.Popen", pgid: int | None = None) -> None: pass +def _effective_file_timeout( + file: Path, + repo_root: Path, + file_timeout: float, + durations: dict[str, float] | None, +) -> float: + """Scale the per-file timeout for files whose last observed runtime + approaches the flat cap. + + The flat ``file_timeout`` (default 300s) is sized for the typical file, + but a handful of large-collection files (e.g. ``tests/test_hermes_state.py``, + 239 tests × subprocess-per-test overhead) legitimately run 200s+ on a + quiet runner. Under CI load that dilates past the cap, the file is + SIGKILL'd mid-run, and the automatic retry then passes — a manufactured + FLAKY report for a file that was never broken (seen 2026-08-18 on main: + first attempt killed at 300s, retry passed in 205s). + + Rule: a file gets ``max(flat_cap, 3 × last_observed_duration)``. Files + without a cache entry keep the flat cap. This only ever *raises* the + bound — a genuinely hung file is still killed, just with headroom + proportional to its known-good runtime. + """ + if not durations: + return file_timeout + cached = durations.get(_format_file(file, repo_root)) + if not cached: + return file_timeout + return max(file_timeout, float(cached) * 3.0) + + def _run_one_file( file: Path, pytest_args: List[str], @@ -1123,12 +1153,19 @@ def main() -> int: _print_inline_failure(fpath, output, repo_root, pytest_passthrough) with ThreadPoolExecutor(max_workers=args.jobs) as pool: + # Duration cache for the timeout scaler: known-slow files get + # proportional headroom instead of a false timeout-kill under + # CI load (see _effective_file_timeout). + timeout_durations = _load_durations(repo_root) futures: List[Future] = [] for file in files: t0 = time.monotonic() fut = pool.submit( _run_one_file, file, pytest_passthrough, repo_root, - args.file_timeout, args.file_retries, + _effective_file_timeout( + file, repo_root, args.file_timeout, timeout_durations + ), + args.file_retries, ) fut.add_done_callback(lambda f, file=file, t0=t0: _on_done(file, t0, f)) futures.append(fut) diff --git a/tests/test_run_tests_parallel_timeout_scaling.py b/tests/test_run_tests_parallel_timeout_scaling.py new file mode 100644 index 0000000000..8d05295743 --- /dev/null +++ b/tests/test_run_tests_parallel_timeout_scaling.py @@ -0,0 +1,54 @@ +"""Duration-aware per-file timeout scaling in scripts/run_tests_parallel.py. + +The flat --file-timeout cap (default 300s) falsely SIGKILL'd +known-slow large-collection files under CI load, then the automatic +retry passed — manufacturing FLAKY reports for healthy files +(tests/test_hermes_state.py, 2026-08-18 on main). The scaler gives a +file max(flat_cap, 3 × last observed duration) and never lowers the cap. +""" + +from __future__ import annotations + +import importlib.util +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +_RUNNER_PATH = REPO_ROOT / "scripts" / "run_tests_parallel.py" + + +def _load_runner(): + spec = importlib.util.spec_from_file_location("run_tests_parallel", _RUNNER_PATH) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +def test_uncached_file_keeps_flat_cap() -> None: + mod = _load_runner() + f = REPO_ROOT / "tests" / "test_example.py" + assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, {}) == 300.0 + assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, None) == 300.0 + + +def test_fast_file_keeps_flat_cap() -> None: + mod = _load_runner() + f = REPO_ROOT / "tests" / "test_fast.py" + durations = {mod._format_file(f, REPO_ROOT): 4.2} + # 3 × 4.2 « 300 — the flat cap stays; the scaler never lowers a bound. + assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 300.0 + + +def test_slow_file_gets_proportional_headroom() -> None: + mod = _load_runner() + f = REPO_ROOT / "tests" / "test_hermes_state.py" + durations = {mod._format_file(f, REPO_ROOT): 205.0} + # 205s last run → 615s bound: a load-dilated healthy run survives, + # a genuine hang is still killed. + assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 615.0 + + +def test_zero_or_missing_duration_is_ignored() -> None: + mod = _load_runner() + f = REPO_ROOT / "tests" / "test_zero.py" + durations = {mod._format_file(f, REPO_ROOT): 0.0} + assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 300.0