fix(ci): scale per-file test timeout by cached duration to stop false FLAKY kills
The flat 300s --file-timeout SIGKILL'd known-slow large-collection files when CI load dilated their runtime past the cap; the automatic one-shot retry then passed, manufacturing a FLAKY report for a healthy file. Seen 2026-08-18 on main run 32155223248's sibling PR runs: tests/test_hermes_state.py (239 tests) killed at 300s on attempt 1, passed in 205s on retry. _effective_file_timeout() now gives each file max(flat_cap, 3 x last cached duration) from test_durations.json. The bound is only ever raised — genuinely hung files are still killed, uncached files keep the flat cap, and --file-timeout/HERMES_TEST_FILE_TIMEOUT semantics are unchanged. Includes a sabotage-verified unit test (fails without the scaler).
This commit is contained in:
@@ -303,6 +303,36 @@ def _kill_tree(proc: "subprocess.Popen", pgid: int | None = None) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def _effective_file_timeout(
|
||||
file: Path,
|
||||
repo_root: Path,
|
||||
file_timeout: float,
|
||||
durations: dict[str, float] | None,
|
||||
) -> float:
|
||||
"""Scale the per-file timeout for files whose last observed runtime
|
||||
approaches the flat cap.
|
||||
|
||||
The flat ``file_timeout`` (default 300s) is sized for the typical file,
|
||||
but a handful of large-collection files (e.g. ``tests/test_hermes_state.py``,
|
||||
239 tests × subprocess-per-test overhead) legitimately run 200s+ on a
|
||||
quiet runner. Under CI load that dilates past the cap, the file is
|
||||
SIGKILL'd mid-run, and the automatic retry then passes — a manufactured
|
||||
FLAKY report for a file that was never broken (seen 2026-08-18 on main:
|
||||
first attempt killed at 300s, retry passed in 205s).
|
||||
|
||||
Rule: a file gets ``max(flat_cap, 3 × last_observed_duration)``. Files
|
||||
without a cache entry keep the flat cap. This only ever *raises* the
|
||||
bound — a genuinely hung file is still killed, just with headroom
|
||||
proportional to its known-good runtime.
|
||||
"""
|
||||
if not durations:
|
||||
return file_timeout
|
||||
cached = durations.get(_format_file(file, repo_root))
|
||||
if not cached:
|
||||
return file_timeout
|
||||
return max(file_timeout, float(cached) * 3.0)
|
||||
|
||||
|
||||
def _run_one_file(
|
||||
file: Path,
|
||||
pytest_args: List[str],
|
||||
@@ -1123,12 +1153,19 @@ def main() -> int:
|
||||
_print_inline_failure(fpath, output, repo_root, pytest_passthrough)
|
||||
|
||||
with ThreadPoolExecutor(max_workers=args.jobs) as pool:
|
||||
# Duration cache for the timeout scaler: known-slow files get
|
||||
# proportional headroom instead of a false timeout-kill under
|
||||
# CI load (see _effective_file_timeout).
|
||||
timeout_durations = _load_durations(repo_root)
|
||||
futures: List[Future] = []
|
||||
for file in files:
|
||||
t0 = time.monotonic()
|
||||
fut = pool.submit(
|
||||
_run_one_file, file, pytest_passthrough, repo_root,
|
||||
args.file_timeout, args.file_retries,
|
||||
_effective_file_timeout(
|
||||
file, repo_root, args.file_timeout, timeout_durations
|
||||
),
|
||||
args.file_retries,
|
||||
)
|
||||
fut.add_done_callback(lambda f, file=file, t0=t0: _on_done(file, t0, f))
|
||||
futures.append(fut)
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Duration-aware per-file timeout scaling in scripts/run_tests_parallel.py.
|
||||
|
||||
The flat --file-timeout cap (default 300s) falsely SIGKILL'd
|
||||
known-slow large-collection files under CI load, then the automatic
|
||||
retry passed — manufacturing FLAKY reports for healthy files
|
||||
(tests/test_hermes_state.py, 2026-08-18 on main). The scaler gives a
|
||||
file max(flat_cap, 3 × last observed duration) and never lowers the cap.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
_RUNNER_PATH = REPO_ROOT / "scripts" / "run_tests_parallel.py"
|
||||
|
||||
|
||||
def _load_runner():
|
||||
spec = importlib.util.spec_from_file_location("run_tests_parallel", _RUNNER_PATH)
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
return mod
|
||||
|
||||
|
||||
def test_uncached_file_keeps_flat_cap() -> None:
|
||||
mod = _load_runner()
|
||||
f = REPO_ROOT / "tests" / "test_example.py"
|
||||
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, {}) == 300.0
|
||||
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, None) == 300.0
|
||||
|
||||
|
||||
def test_fast_file_keeps_flat_cap() -> None:
|
||||
mod = _load_runner()
|
||||
f = REPO_ROOT / "tests" / "test_fast.py"
|
||||
durations = {mod._format_file(f, REPO_ROOT): 4.2}
|
||||
# 3 × 4.2 « 300 — the flat cap stays; the scaler never lowers a bound.
|
||||
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 300.0
|
||||
|
||||
|
||||
def test_slow_file_gets_proportional_headroom() -> None:
|
||||
mod = _load_runner()
|
||||
f = REPO_ROOT / "tests" / "test_hermes_state.py"
|
||||
durations = {mod._format_file(f, REPO_ROOT): 205.0}
|
||||
# 205s last run → 615s bound: a load-dilated healthy run survives,
|
||||
# a genuine hang is still killed.
|
||||
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 615.0
|
||||
|
||||
|
||||
def test_zero_or_missing_duration_is_ignored() -> None:
|
||||
mod = _load_runner()
|
||||
f = REPO_ROOT / "tests" / "test_zero.py"
|
||||
durations = {mod._format_file(f, REPO_ROOT): 0.0}
|
||||
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 300.0
|
||||
Reference in New Issue
Block a user