Port from cline/cline#13525: bound giant single-line matches in content search

A search_files hit inside a serialized dump (multi-MB single-line JSON,
minified bundle) made rg/grep emit the entire matched line into stdout:
head -n counts lines, so a 40MB match line crossed the exec transport
untruncated and was buffered whole into Python before the per-match
[:500] clamp ran. Measured on main: 42MB transport payload and ~180MB
peak Python allocation for a single match.

Fix at the engine layer, all three pipelines:
- rg: --max-columns 2000 --max-columns-preview (preview keeps the match
  visible instead of omitting it)
- grep fallback + darwin pruned-grep fallback: | cut -c1-2000
- files_only/count modes skipped (lines are paths/counts, never giant)

2000 cols exceeds the existing 500-char content clamp, so no
previously-visible content changes.

Adapted from cline/cline#13525 (search_codebase RangeError OOM crash on
giant single-line files) — hermes buffers in Python rather than a JS
string, so the failure mode is memory/transport blowup rather than an
uncaughtException, but the class is identical.
This commit is contained in:
Teknium
2026-08-26 17:14:04 -07:00
committed by kshitij
parent cee12cc34d
commit 66414dedd3
2 changed files with 125 additions and 1 deletions
@@ -0,0 +1,100 @@
"""Giant single-line file containment in content search (cline/cline#13525 port).
A match inside a serialized dump (multi-MB single-line JSON, minified
bundle) used to make rg/grep emit the ENTIRE matched line into stdout:
``head -n`` counts lines, so a 40MB match line crossed the transport
untruncated and was buffered whole into Python before the per-match
[:500] clamp ran (measured 42MB transport / ~180MB peak alloc for one
match). The fix bounds lines at the search-engine layer: rg gets
``--max-columns 2000 --max-columns-preview``; the grep fallbacks pipe
through ``cut -c1-2000``.
These tests run the REAL pipelines via bash (no mocked stdout) so the
flag/pipe behavior of the installed rg/grep is what's exercised.
"""
import os
import shutil
import subprocess
import pytest
from tools.file_operations import ShellFileOperations
# Big enough to prove containment, small enough to keep the test fast.
GIANT = 5 * 1024 * 1024 # 5MB single line
# Generous ceiling: pre-fix stdout for one giant match is >= GIANT bytes.
STDOUT_CEILING = 1 * 1024 * 1024
class RecordingEnv:
"""Local bash executor that records the largest stdout it returned."""
def __init__(self, cwd):
self.cwd = cwd
self.max_stdout = 0
def execute(self, command, timeout=60, **kwargs):
proc = subprocess.run(
["bash", "-c", command],
capture_output=True, text=True, errors="replace",
timeout=timeout + 30,
)
out = proc.stdout + (proc.stderr or "")
self.max_stdout = max(self.max_stdout, len(out))
return {"output": out, "returncode": proc.returncode}
@pytest.fixture()
def giant_dir(tmp_path):
(tmp_path / "trace.json").write_text(
'{"needle": "' + "x" * GIANT + '"}', encoding="utf-8"
)
(tmp_path / "small.py").write_text("needle = 1\n", encoding="utf-8")
return tmp_path
def _ops(giant_dir, engine):
env = RecordingEnv(str(giant_dir))
ops = ShellFileOperations(env)
ops._has_command = lambda cmd: cmd == engine
return ops, env
@pytest.mark.parametrize("engine", ["rg", "grep"])
def test_giant_single_line_match_is_bounded(giant_dir, engine):
if shutil.which(engine) is None:
pytest.skip(f"{engine} not installed")
ops, env = _ops(giant_dir, engine)
result = ops.search("needle", path=str(giant_dir), target="content")
assert result.error is None
paths = {os.path.basename(m.path) for m in result.matches}
# The giant-file match must still be REPORTED (preview, not omission)...
assert paths == {"trace.json", "small.py"}
assert all(len(m.content) <= 500 for m in result.matches)
# ...but its full line must never have crossed the transport.
assert env.max_stdout < STDOUT_CEILING, (
f"{engine} pipeline returned {env.max_stdout} bytes of stdout — "
"giant matched line was not truncated at the engine layer"
)
@pytest.mark.parametrize("engine", ["rg", "grep"])
@pytest.mark.parametrize("output_mode", ["files_only", "count"])
def test_line_cap_skipped_for_path_and_count_modes(giant_dir, engine, output_mode):
"""files_only/count lines are paths/counts — never giant, never cut."""
if shutil.which(engine) is None:
pytest.skip(f"{engine} not installed")
ops, env = _ops(giant_dir, engine)
result = ops.search("needle", path=str(giant_dir), target="content",
output_mode=output_mode)
assert result.error is None
if output_mode == "files_only":
assert {os.path.basename(f) for f in result.files} == {"trace.json", "small.py"}
else:
assert {os.path.basename(k) for k in result.counts} == {"trace.json", "small.py"}
assert env.max_stdout < STDOUT_CEILING
+25 -1
View File
@@ -3692,6 +3692,21 @@ class ShellFileOperations(FileOperations):
"""Search using ripgrep."""
cmd_parts = ["rg", "--line-number", "--no-heading", "--with-filename"]
# Giant-single-line containment (ported from cline/cline#13525): a
# match inside a serialized dump (multi-MB single-line JSON/minified
# bundle) makes rg emit the ENTIRE line into stdout. `head -n` counts
# lines, so a 40MB match line sails through untruncated, gets buffered
# whole into Python, and only THEN hits the per-match [:500] clamp —
# measured 42MB across the transport / ~180MB peak alloc for one
# match on main. --max-columns bounds each printed line at the rg
# layer; --max-columns-preview keeps a truncated prefix (instead of
# omitting the match) so the model still sees the hit. 2000 cols
# comfortably exceeds the 500-char content clamp below, so no
# previously-visible content is lost. Both flags predate rg 11; the
# engine floor here is already rg 13 (--sortr).
if output_mode not in ("files_only", "count"):
cmd_parts.extend(["--max-columns", "2000", "--max-columns-preview"])
# Auto-multiline: a regex `\n` (or a literal newline in the pattern)
# cannot match in rg's default line-oriented mode — it used to hard
# error ("the literal \"\\n\" is not allowed") and burn a turn. When
@@ -3896,6 +3911,12 @@ class ShellFileOperations(FileOperations):
# Fetch generously so we can compute total before slicing
fetch_limit = limit + offset + (200 if context > 0 else 0)
cmd_parts.extend(["|", "head", "-n", str(fetch_limit)])
# grep has no --max-columns: bound giant single-line matches (see the
# rg branch's containment comment) at the pipe layer instead. Safe for
# the file:line:content parser — truncation only ever drops content
# tail. Skipped for files_only/count where lines are paths/counts.
if output_mode not in ("files_only", "count"):
cmd_parts.extend(["|", "cut", "-c1-2000"])
# `set -o pipefail` so grep's exit status propagates through `| head`
# (without it the pipeline reports head's 0, masking grep's error 2).
@@ -3944,9 +3965,12 @@ class ShellFileOperations(FileOperations):
find_parts.extend(["-name", self._escape_shell_arg(file_glob)])
find_parts.extend(["-exec", *grep_parts, "{}", "+"])
fetch_limit = limit + offset + (200 if context > 0 else 0)
# Same giant-single-line bound as the plain grep path (grep lacks
# --max-columns); see the rg branch's containment comment.
line_cap = " | cut -c1-2000" if output_mode not in ("files_only", "count") else ""
cmd = (
"set -o pipefail; " + " ".join(find_parts)
+ f" 2>/dev/null | head -n {fetch_limit}"
+ f" 2>/dev/null | head -n {fetch_limit}{line_cap}"
)
result = self._exec(cmd, timeout=60)
return self._parse_grep_search_output(result, output_mode, limit, offset, context)