fix(ci): scale per-file test timeout by cached duration to stop false FLAKY kills

The flat 300s --file-timeout SIGKILL'd known-slow large-collection
files when CI load dilated their runtime past the cap; the automatic
one-shot retry then passed, manufacturing a FLAKY report for a healthy
file. Seen 2026-08-18 on main run 32155223248's sibling PR runs:
tests/test_hermes_state.py (239 tests) killed at 300s on attempt 1,
passed in 205s on retry.

_effective_file_timeout() now gives each file
max(flat_cap, 3 x last cached duration) from test_durations.json.
The bound is only ever raised — genuinely hung files are still killed,
uncached files keep the flat cap, and --file-timeout/HERMES_TEST_FILE_TIMEOUT
semantics are unchanged.

Includes a sabotage-verified unit test (fails without the scaler).
This commit is contained in:
Teknium
2026-08-18 09:26:34 -07:00
parent 7abe9502ee
commit 9a69785790
2 changed files with 92 additions and 1 deletions
+38 -1
View File
@@ -303,6 +303,36 @@ def _kill_tree(proc: "subprocess.Popen", pgid: int | None = None) -> None:
pass
def _effective_file_timeout(
file: Path,
repo_root: Path,
file_timeout: float,
durations: dict[str, float] | None,
) -> float:
"""Scale the per-file timeout for files whose last observed runtime
approaches the flat cap.
The flat ``file_timeout`` (default 300s) is sized for the typical file,
but a handful of large-collection files (e.g. ``tests/test_hermes_state.py``,
239 tests × subprocess-per-test overhead) legitimately run 200s+ on a
quiet runner. Under CI load that dilates past the cap, the file is
SIGKILL'd mid-run, and the automatic retry then passes — a manufactured
FLAKY report for a file that was never broken (seen 2026-08-18 on main:
first attempt killed at 300s, retry passed in 205s).
Rule: a file gets ``max(flat_cap, 3 × last_observed_duration)``. Files
without a cache entry keep the flat cap. This only ever *raises* the
bound — a genuinely hung file is still killed, just with headroom
proportional to its known-good runtime.
"""
if not durations:
return file_timeout
cached = durations.get(_format_file(file, repo_root))
if not cached:
return file_timeout
return max(file_timeout, float(cached) * 3.0)
def _run_one_file(
file: Path,
pytest_args: List[str],
@@ -1123,12 +1153,19 @@ def main() -> int:
_print_inline_failure(fpath, output, repo_root, pytest_passthrough)
with ThreadPoolExecutor(max_workers=args.jobs) as pool:
# Duration cache for the timeout scaler: known-slow files get
# proportional headroom instead of a false timeout-kill under
# CI load (see _effective_file_timeout).
timeout_durations = _load_durations(repo_root)
futures: List[Future] = []
for file in files:
t0 = time.monotonic()
fut = pool.submit(
_run_one_file, file, pytest_passthrough, repo_root,
args.file_timeout, args.file_retries,
_effective_file_timeout(
file, repo_root, args.file_timeout, timeout_durations
),
args.file_retries,
)
fut.add_done_callback(lambda f, file=file, t0=t0: _on_done(file, t0, f))
futures.append(fut)
@@ -0,0 +1,54 @@
"""Duration-aware per-file timeout scaling in scripts/run_tests_parallel.py.
The flat --file-timeout cap (default 300s) falsely SIGKILL'd
known-slow large-collection files under CI load, then the automatic
retry passed — manufacturing FLAKY reports for healthy files
(tests/test_hermes_state.py, 2026-08-18 on main). The scaler gives a
file max(flat_cap, 3 × last observed duration) and never lowers the cap.
"""
from __future__ import annotations
import importlib.util
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
_RUNNER_PATH = REPO_ROOT / "scripts" / "run_tests_parallel.py"
def _load_runner():
spec = importlib.util.spec_from_file_location("run_tests_parallel", _RUNNER_PATH)
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
return mod
def test_uncached_file_keeps_flat_cap() -> None:
mod = _load_runner()
f = REPO_ROOT / "tests" / "test_example.py"
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, {}) == 300.0
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, None) == 300.0
def test_fast_file_keeps_flat_cap() -> None:
mod = _load_runner()
f = REPO_ROOT / "tests" / "test_fast.py"
durations = {mod._format_file(f, REPO_ROOT): 4.2}
# 3 × 4.2 « 300 — the flat cap stays; the scaler never lowers a bound.
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 300.0
def test_slow_file_gets_proportional_headroom() -> None:
mod = _load_runner()
f = REPO_ROOT / "tests" / "test_hermes_state.py"
durations = {mod._format_file(f, REPO_ROOT): 205.0}
# 205s last run → 615s bound: a load-dilated healthy run survives,
# a genuine hang is still killed.
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 615.0
def test_zero_or_missing_duration_is_ignored() -> None:
mod = _load_runner()
f = REPO_ROOT / "tests" / "test_zero.py"
durations = {mod._format_file(f, REPO_ROOT): 0.0}
assert mod._effective_file_timeout(f, REPO_ROOT, 300.0, durations) == 300.0