diff --git a/tests/tools/test_search_giant_line_containment.py b/tests/tools/test_search_giant_line_containment.py new file mode 100644 index 0000000000..a31c6509fa --- /dev/null +++ b/tests/tools/test_search_giant_line_containment.py @@ -0,0 +1,100 @@ +"""Giant single-line file containment in content search (cline/cline#13525 port). + +A match inside a serialized dump (multi-MB single-line JSON, minified +bundle) used to make rg/grep emit the ENTIRE matched line into stdout: +``head -n`` counts lines, so a 40MB match line crossed the transport +untruncated and was buffered whole into Python before the per-match +[:500] clamp ran (measured 42MB transport / ~180MB peak alloc for one +match). The fix bounds lines at the search-engine layer: rg gets +``--max-columns 2000 --max-columns-preview``; the grep fallbacks pipe +through ``cut -c1-2000``. + +These tests run the REAL pipelines via bash (no mocked stdout) so the +flag/pipe behavior of the installed rg/grep is what's exercised. +""" + +import os +import shutil +import subprocess + +import pytest + +from tools.file_operations import ShellFileOperations + +# Big enough to prove containment, small enough to keep the test fast. +GIANT = 5 * 1024 * 1024 # 5MB single line +# Generous ceiling: pre-fix stdout for one giant match is >= GIANT bytes. +STDOUT_CEILING = 1 * 1024 * 1024 + + +class RecordingEnv: + """Local bash executor that records the largest stdout it returned.""" + + def __init__(self, cwd): + self.cwd = cwd + self.max_stdout = 0 + + def execute(self, command, timeout=60, **kwargs): + proc = subprocess.run( + ["bash", "-c", command], + capture_output=True, text=True, errors="replace", + timeout=timeout + 30, + ) + out = proc.stdout + (proc.stderr or "") + self.max_stdout = max(self.max_stdout, len(out)) + return {"output": out, "returncode": proc.returncode} + + +@pytest.fixture() +def giant_dir(tmp_path): + (tmp_path / "trace.json").write_text( + '{"needle": "' + "x" * GIANT + '"}', encoding="utf-8" + ) + (tmp_path / "small.py").write_text("needle = 1\n", encoding="utf-8") + return tmp_path + + +def _ops(giant_dir, engine): + env = RecordingEnv(str(giant_dir)) + ops = ShellFileOperations(env) + ops._has_command = lambda cmd: cmd == engine + return ops, env + + +@pytest.mark.parametrize("engine", ["rg", "grep"]) +def test_giant_single_line_match_is_bounded(giant_dir, engine): + if shutil.which(engine) is None: + pytest.skip(f"{engine} not installed") + ops, env = _ops(giant_dir, engine) + + result = ops.search("needle", path=str(giant_dir), target="content") + + assert result.error is None + paths = {os.path.basename(m.path) for m in result.matches} + # The giant-file match must still be REPORTED (preview, not omission)... + assert paths == {"trace.json", "small.py"} + assert all(len(m.content) <= 500 for m in result.matches) + # ...but its full line must never have crossed the transport. + assert env.max_stdout < STDOUT_CEILING, ( + f"{engine} pipeline returned {env.max_stdout} bytes of stdout — " + "giant matched line was not truncated at the engine layer" + ) + + +@pytest.mark.parametrize("engine", ["rg", "grep"]) +@pytest.mark.parametrize("output_mode", ["files_only", "count"]) +def test_line_cap_skipped_for_path_and_count_modes(giant_dir, engine, output_mode): + """files_only/count lines are paths/counts — never giant, never cut.""" + if shutil.which(engine) is None: + pytest.skip(f"{engine} not installed") + ops, env = _ops(giant_dir, engine) + + result = ops.search("needle", path=str(giant_dir), target="content", + output_mode=output_mode) + + assert result.error is None + if output_mode == "files_only": + assert {os.path.basename(f) for f in result.files} == {"trace.json", "small.py"} + else: + assert {os.path.basename(k) for k in result.counts} == {"trace.json", "small.py"} + assert env.max_stdout < STDOUT_CEILING diff --git a/tools/file_operations.py b/tools/file_operations.py index f0f5fc959a..85344a312d 100644 --- a/tools/file_operations.py +++ b/tools/file_operations.py @@ -3692,6 +3692,21 @@ class ShellFileOperations(FileOperations): """Search using ripgrep.""" cmd_parts = ["rg", "--line-number", "--no-heading", "--with-filename"] + # Giant-single-line containment (ported from cline/cline#13525): a + # match inside a serialized dump (multi-MB single-line JSON/minified + # bundle) makes rg emit the ENTIRE line into stdout. `head -n` counts + # lines, so a 40MB match line sails through untruncated, gets buffered + # whole into Python, and only THEN hits the per-match [:500] clamp — + # measured 42MB across the transport / ~180MB peak alloc for one + # match on main. --max-columns bounds each printed line at the rg + # layer; --max-columns-preview keeps a truncated prefix (instead of + # omitting the match) so the model still sees the hit. 2000 cols + # comfortably exceeds the 500-char content clamp below, so no + # previously-visible content is lost. Both flags predate rg 11; the + # engine floor here is already rg 13 (--sortr). + if output_mode not in ("files_only", "count"): + cmd_parts.extend(["--max-columns", "2000", "--max-columns-preview"]) + # Auto-multiline: a regex `\n` (or a literal newline in the pattern) # cannot match in rg's default line-oriented mode — it used to hard # error ("the literal \"\\n\" is not allowed") and burn a turn. When @@ -3896,6 +3911,12 @@ class ShellFileOperations(FileOperations): # Fetch generously so we can compute total before slicing fetch_limit = limit + offset + (200 if context > 0 else 0) cmd_parts.extend(["|", "head", "-n", str(fetch_limit)]) + # grep has no --max-columns: bound giant single-line matches (see the + # rg branch's containment comment) at the pipe layer instead. Safe for + # the file:line:content parser — truncation only ever drops content + # tail. Skipped for files_only/count where lines are paths/counts. + if output_mode not in ("files_only", "count"): + cmd_parts.extend(["|", "cut", "-c1-2000"]) # `set -o pipefail` so grep's exit status propagates through `| head` # (without it the pipeline reports head's 0, masking grep's error 2). @@ -3944,9 +3965,12 @@ class ShellFileOperations(FileOperations): find_parts.extend(["-name", self._escape_shell_arg(file_glob)]) find_parts.extend(["-exec", *grep_parts, "{}", "+"]) fetch_limit = limit + offset + (200 if context > 0 else 0) + # Same giant-single-line bound as the plain grep path (grep lacks + # --max-columns); see the rg branch's containment comment. + line_cap = " | cut -c1-2000" if output_mode not in ("files_only", "count") else "" cmd = ( "set -o pipefail; " + " ".join(find_parts) - + f" 2>/dev/null | head -n {fetch_limit}" + + f" 2>/dev/null | head -n {fetch_limit}{line_cap}" ) result = self._exec(cmd, timeout=60) return self._parse_grep_search_output(result, output_mode, limit, offset, context)