c683f6e739
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
467 lines
16 KiB
Python
467 lines
16 KiB
Python
"""Bounded document extraction and non-text file policy for workspaces."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
import zipfile
|
|
from pathlib import Path
|
|
from xml.etree import ElementTree as ET
|
|
|
|
MAX_DOCUMENT_BYTES = 50 * 1024 * 1024
|
|
MAX_DOCUMENT_RESULT_CHARS = 50_000
|
|
MAX_CONVERTED_DOCUMENT_BYTES = 10 * 1024 * 1024
|
|
DOCUMENT_CONVERSION_TIMEOUT_SECONDS = 60
|
|
MAX_IMAGE_BYTES = 25 * 1024 * 1024
|
|
MAX_IMAGE_EDGE = 2048
|
|
MAX_IMAGE_PIXELS = 40_000_000
|
|
MAX_OOXML_MEMBERS = 10_000
|
|
MAX_OOXML_MEMBER_BYTES = 50 * 1024 * 1024
|
|
MAX_OOXML_EXPANDED_BYTES = 200 * 1024 * 1024
|
|
MAX_OOXML_COMPRESSION_RATIO = 100
|
|
|
|
IMAGE_EXTENSIONS = frozenset(
|
|
{".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".png", ".tif", ".tiff", ".webp"}
|
|
)
|
|
DOCUMENT_EXTENSIONS = frozenset(
|
|
{
|
|
".doc",
|
|
".docm",
|
|
".docx",
|
|
".epub",
|
|
".odp",
|
|
".ods",
|
|
".odt",
|
|
".pdf",
|
|
".pot",
|
|
".pps",
|
|
".ppsm",
|
|
".ppsx",
|
|
".ppt",
|
|
".pptm",
|
|
".pptx",
|
|
".rtf",
|
|
".xls",
|
|
".xlsb",
|
|
".xlsm",
|
|
".xlsx",
|
|
}
|
|
)
|
|
ARCHIVE_EXTENSIONS = frozenset(
|
|
{".7z", ".bz2", ".gz", ".rar", ".tar", ".tgz", ".xz", ".zip"}
|
|
)
|
|
DATABASE_EXTENSIONS = frozenset({".db", ".sqlite", ".sqlite3"})
|
|
EXECUTABLE_EXTENSIONS = frozenset(
|
|
{".app", ".deb", ".dll", ".dylib", ".elf", ".exe", ".msi", ".rpm", ".so"}
|
|
)
|
|
DATASET_EXTENSIONS = frozenset(
|
|
{".arrow", ".feather", ".h5", ".hdf5", ".npy", ".npz", ".parquet"}
|
|
)
|
|
MEDIA_EXTENSIONS = frozenset(
|
|
{".aac", ".avi", ".flac", ".m4a", ".mkv", ".mov", ".mp3", ".mp4", ".ogg", ".wav", ".webm"}
|
|
)
|
|
|
|
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
_A = "http://schemas.openxmlformats.org/drawingml/2006/main"
|
|
_S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
|
|
_R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
|
|
|
|
|
|
class DocumentExtractionError(RuntimeError):
|
|
"""A supported document could not be converted to bounded text."""
|
|
|
|
|
|
def classify_file(path: str, head: bytes = b"") -> str:
|
|
"""Classify a workspace file into one policy category."""
|
|
|
|
extension = Path(path).suffix.lower()
|
|
if extension in IMAGE_EXTENSIONS:
|
|
return "image"
|
|
if extension in DOCUMENT_EXTENSIONS:
|
|
return "document"
|
|
if extension in ARCHIVE_EXTENSIONS:
|
|
return "archive"
|
|
if extension in DATABASE_EXTENSIONS or head.startswith(b"SQLite format 3\x00"):
|
|
return "database"
|
|
if extension in EXECUTABLE_EXTENSIONS or head.startswith((b"MZ", b"\x7fELF")):
|
|
return "executable"
|
|
if extension in DATASET_EXTENSIONS:
|
|
return "dataset"
|
|
if extension in MEDIA_EXTENSIONS:
|
|
return "media"
|
|
return "unknown"
|
|
|
|
|
|
def binary_processing_guidance(path: str, kind: str, size_bytes: int) -> str:
|
|
"""Return bounded, actionable JSON-like guidance for Agent-side programming."""
|
|
|
|
import json
|
|
|
|
common = {
|
|
"code": "BINARY_PROCESSING_REQUIRED"
|
|
if kind != "binary"
|
|
else "UNSUPPORTED_BINARY_FILE",
|
|
"path": path,
|
|
"kind": kind,
|
|
"size_bytes": size_bytes,
|
|
}
|
|
if kind == "archive":
|
|
common.update(
|
|
action=(
|
|
"Use execute with Python to list and validate archive members before "
|
|
"selective extraction; never use extractall."
|
|
),
|
|
constraints={
|
|
"list_before_extract": True,
|
|
"max_members": 2000,
|
|
"max_total_uncompressed_bytes": 500 * 1024 * 1024,
|
|
"max_member_bytes": 100 * 1024 * 1024,
|
|
"max_compression_ratio": 100,
|
|
"reject_absolute_or_parent_paths": True,
|
|
"do_not_execute_members": True,
|
|
},
|
|
)
|
|
elif kind == "database":
|
|
common.update(
|
|
action=(
|
|
"Use execute with Python sqlite3 in read-only mode: "
|
|
"file:<path>?mode=ro&immutable=1; set PRAGMA query_only=ON; "
|
|
"inspect schema, run bounded SELECT queries with LIMIT, and write "
|
|
"large results under artifacts/."
|
|
),
|
|
constraints={
|
|
"read_only": True,
|
|
"mode": "mode=ro",
|
|
"query_only": True,
|
|
"max_rows": 1000,
|
|
"forbid_attach_database": True,
|
|
"forbid_load_extension": True,
|
|
},
|
|
)
|
|
elif kind == "executable":
|
|
common.update(
|
|
action=(
|
|
"Use execute only for bounded static metadata inspection (hash, file "
|
|
"headers, signature, imports, strings); this file must not be executed."
|
|
),
|
|
constraints={"must_not_be_executed": True, "static_analysis_only": True},
|
|
)
|
|
elif kind == "dataset":
|
|
common.update(
|
|
action=(
|
|
"Use execute with the appropriate library to inspect schema, dimensions, "
|
|
"statistics, and a bounded sample; do not serialize the whole dataset."
|
|
)
|
|
)
|
|
elif kind == "media":
|
|
common.update(
|
|
action=(
|
|
"Use execute with ffprobe/ffmpeg or an available transcription workflow "
|
|
"to inspect metadata and selected ranges; do not inline the complete file."
|
|
)
|
|
)
|
|
else:
|
|
common.update(
|
|
kind="binary",
|
|
action=(
|
|
"This is an unsupported binary file. Use execute only for bounded static "
|
|
"inspection; do not execute it or inline its bytes."
|
|
),
|
|
)
|
|
return json.dumps(common, ensure_ascii=False, sort_keys=True)
|
|
|
|
|
|
def prepare_image_bytes(data: bytes, path: str) -> bytes:
|
|
"""Validate and downsample an image before it becomes a model media block."""
|
|
|
|
import io
|
|
|
|
if len(data) > MAX_IMAGE_BYTES:
|
|
raise DocumentExtractionError(
|
|
f"IMAGE_TOO_LARGE: {len(data)} bytes exceeds {MAX_IMAGE_BYTES}"
|
|
)
|
|
try:
|
|
from PIL import Image
|
|
|
|
image = Image.open(io.BytesIO(data))
|
|
width, height = image.size
|
|
if width * height > MAX_IMAGE_PIXELS:
|
|
raise DocumentExtractionError(
|
|
f"IMAGE_PIXEL_BUDGET_EXCEEDED: {width}x{height} exceeds "
|
|
f"{MAX_IMAGE_PIXELS} pixels"
|
|
)
|
|
image.load()
|
|
except DocumentExtractionError:
|
|
raise
|
|
except Exception as exc:
|
|
raise DocumentExtractionError(
|
|
f"IMAGE_PROCESSING_FAILED: {path}: {type(exc).__name__}: {exc}"
|
|
) from exc
|
|
|
|
frame_count = int(getattr(image, "n_frames", 1) or 1)
|
|
if frame_count > 1:
|
|
image.seek(0)
|
|
work = image.convert("RGBA" if image.mode in {"RGBA", "LA"} else "RGB")
|
|
else:
|
|
work = image.copy()
|
|
|
|
if max(work.size) <= MAX_IMAGE_EDGE and frame_count == 1:
|
|
return data
|
|
|
|
work.thumbnail((MAX_IMAGE_EDGE, MAX_IMAGE_EDGE), Image.Resampling.LANCZOS)
|
|
has_alpha = work.mode in {"RGBA", "LA"} or (
|
|
work.mode == "P" and "transparency" in work.info
|
|
)
|
|
output = io.BytesIO()
|
|
if has_alpha:
|
|
if work.mode == "P":
|
|
work = work.convert("RGBA")
|
|
work.save(output, "PNG", optimize=True)
|
|
else:
|
|
if work.mode != "RGB":
|
|
work = work.convert("RGB")
|
|
work.save(output, "JPEG", quality=85, optimize=True)
|
|
return output.getvalue()
|
|
|
|
|
|
def extract_document_bytes(data: bytes, path: str) -> str:
|
|
"""Extract readable text from a supported document without exposing bytes."""
|
|
|
|
if len(data) > MAX_DOCUMENT_BYTES:
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_TOO_LARGE: {len(data)} bytes exceeds {MAX_DOCUMENT_BYTES}"
|
|
)
|
|
extension = Path(path).suffix.lower()
|
|
try:
|
|
if extension == ".docx":
|
|
return _extract_docx(data)
|
|
if extension == ".pptx":
|
|
return _extract_pptx(data)
|
|
if extension == ".xlsx":
|
|
return _extract_xlsx(data)
|
|
return _extract_anydoc(data, extension)
|
|
except DocumentExtractionError:
|
|
raise
|
|
except Exception as exc:
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
|
|
) from exc
|
|
|
|
|
|
def paginate_document_text(text: str, *, offset: int, limit: int) -> str:
|
|
"""Apply line and character budgets to extracted document text."""
|
|
|
|
lines = text.splitlines(keepends=True)
|
|
if not lines:
|
|
return "(document contains no extractable text)"
|
|
if offset >= len(lines):
|
|
raise DocumentExtractionError(
|
|
f"Line offset {offset} exceeds extracted document length ({len(lines)} lines)"
|
|
)
|
|
selected = "".join(lines[offset : offset + limit])
|
|
if len(selected) <= MAX_DOCUMENT_RESULT_CHARS:
|
|
return selected
|
|
trimmed = selected[:MAX_DOCUMENT_RESULT_CHARS]
|
|
boundary = trimmed.rfind("\n")
|
|
if boundary > 0:
|
|
trimmed = trimmed[: boundary + 1]
|
|
consumed = max(1, len(trimmed.splitlines()))
|
|
return (
|
|
trimmed
|
|
+ f"\n[DOCUMENT_OUTPUT_TRUNCATED: use offset={offset + consumed} to continue; "
|
|
+ f"single-read limit is {MAX_DOCUMENT_RESULT_CHARS} characters]\n"
|
|
)
|
|
|
|
|
|
def _validated_ooxml_archive(data: bytes) -> zipfile.ZipFile:
|
|
try:
|
|
archive = zipfile.ZipFile(_bytes_path(data))
|
|
members = archive.infolist()
|
|
except zipfile.BadZipFile as exc:
|
|
raise DocumentExtractionError(
|
|
"DOCUMENT_EXTRACTION_FAILED: invalid OOXML container"
|
|
) from exc
|
|
expanded = 0
|
|
names: set[str] = set()
|
|
if len(members) > MAX_OOXML_MEMBERS:
|
|
archive.close()
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_RESOURCE_LIMIT: OOXML has {len(members)} members; "
|
|
f"limit is {MAX_OOXML_MEMBERS}"
|
|
)
|
|
for member in members:
|
|
normalized = member.filename.replace("\\", "/")
|
|
parts = tuple(part for part in normalized.split("/") if part)
|
|
if (
|
|
normalized.startswith("/")
|
|
or ".." in parts
|
|
or member.filename in names
|
|
or bool(member.flag_bits & 0x1)
|
|
):
|
|
archive.close()
|
|
raise DocumentExtractionError(
|
|
"DOCUMENT_RESOURCE_LIMIT: OOXML contains an unsafe, duplicate, "
|
|
"or encrypted member"
|
|
)
|
|
names.add(member.filename)
|
|
expanded += member.file_size
|
|
ratio = member.file_size / max(member.compress_size, 1)
|
|
if (
|
|
member.file_size > MAX_OOXML_MEMBER_BYTES
|
|
or expanded > MAX_OOXML_EXPANDED_BYTES
|
|
or ratio > MAX_OOXML_COMPRESSION_RATIO
|
|
):
|
|
archive.close()
|
|
raise DocumentExtractionError(
|
|
"DOCUMENT_RESOURCE_LIMIT: OOXML member expansion exceeds safety limits"
|
|
)
|
|
return archive
|
|
|
|
|
|
def _zip_xml(data: bytes, member: str) -> ET.Element:
|
|
try:
|
|
with _validated_ooxml_archive(data) as archive:
|
|
raw = archive.read(member)
|
|
except KeyError as exc:
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_EXTRACTION_FAILED: missing {member}"
|
|
) from exc
|
|
return ET.fromstring(raw)
|
|
|
|
|
|
def _bytes_path(data: bytes):
|
|
import io
|
|
|
|
return io.BytesIO(data)
|
|
|
|
|
|
def _ooxml_part_number(name: str) -> int:
|
|
stem = Path(name).stem
|
|
digits = "".join(character for character in stem if character.isdigit())
|
|
return int(digits) if digits else 0
|
|
|
|
|
|
def _extract_docx(data: bytes) -> str:
|
|
root = _zip_xml(data, "word/document.xml")
|
|
paragraphs: list[str] = []
|
|
for paragraph in root.iter(f"{{{_W}}}p"):
|
|
text = "".join(node.text or "" for node in paragraph.iter(f"{{{_W}}}t"))
|
|
if text:
|
|
paragraphs.append(text)
|
|
if not paragraphs:
|
|
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: DOCX has no text")
|
|
return "\n".join(paragraphs) + "\n"
|
|
|
|
|
|
def _extract_pptx(data: bytes) -> str:
|
|
try:
|
|
with _validated_ooxml_archive(data) as archive:
|
|
names = sorted(
|
|
name
|
|
for name in archive.namelist()
|
|
if name.startswith("ppt/slides/slide") and name.endswith(".xml")
|
|
)
|
|
slides: list[str] = []
|
|
for index, name in enumerate(sorted(names, key=_ooxml_part_number), 1):
|
|
root = ET.fromstring(archive.read(name))
|
|
texts = [node.text or "" for node in root.iter(f"{{{_A}}}t")]
|
|
slides.append(f"## Slide {index}\n" + "\n".join(t for t in texts if t))
|
|
except zipfile.BadZipFile as exc:
|
|
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid PPTX") from exc
|
|
if not slides:
|
|
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: PPTX has no slides")
|
|
return "\n\n".join(slides) + "\n"
|
|
|
|
|
|
def _extract_xlsx(data: bytes) -> str:
|
|
try:
|
|
with _validated_ooxml_archive(data) as archive:
|
|
shared: list[str] = []
|
|
if "xl/sharedStrings.xml" in archive.namelist():
|
|
root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
|
|
shared = [
|
|
"".join(node.text or "" for node in item.iter(f"{{{_S}}}t"))
|
|
for item in root.iter(f"{{{_S}}}si")
|
|
]
|
|
sheets = sorted(
|
|
name
|
|
for name in archive.namelist()
|
|
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
|
|
)
|
|
output: list[str] = []
|
|
for index, name in enumerate(sheets, 1):
|
|
root = ET.fromstring(archive.read(name))
|
|
output.append(f"## Sheet {index}")
|
|
for row in root.iter(f"{{{_S}}}row"):
|
|
values: list[str] = []
|
|
for cell in row.iter(f"{{{_S}}}c"):
|
|
value_node = cell.find(f"{{{_S}}}v")
|
|
value = value_node.text if value_node is not None else ""
|
|
if cell.get("t") == "s" and value and value.isdigit():
|
|
shared_index = int(value)
|
|
value = shared[shared_index] if shared_index < len(shared) else value
|
|
values.append(value or "")
|
|
output.append("\t".join(values))
|
|
except zipfile.BadZipFile as exc:
|
|
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid XLSX") from exc
|
|
if len(output) <= 1:
|
|
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: XLSX has no sheets")
|
|
return "\n".join(output) + "\n"
|
|
|
|
|
|
def _extract_anydoc(data: bytes, extension: str) -> str:
|
|
source = ""
|
|
output = ""
|
|
try:
|
|
with tempfile.NamedTemporaryFile(suffix=extension, delete=False) as handle:
|
|
handle.write(data)
|
|
source = handle.name
|
|
with tempfile.NamedTemporaryFile(suffix=".md", delete=False) as handle:
|
|
output = handle.name
|
|
script = (
|
|
"import pathlib,sys; import anydoc; "
|
|
"text=anydoc.to_markdown(sys.argv[1]); "
|
|
"pathlib.Path(sys.argv[2]).write_text(text, encoding='utf-8')"
|
|
)
|
|
subprocess.run(
|
|
[sys.executable, "-c", script, source, output],
|
|
check=True,
|
|
capture_output=True,
|
|
timeout=DOCUMENT_CONVERSION_TIMEOUT_SECONDS,
|
|
)
|
|
output_path = Path(output)
|
|
if output_path.stat().st_size > MAX_CONVERTED_DOCUMENT_BYTES:
|
|
raise DocumentExtractionError(
|
|
"DOCUMENT_RESOURCE_LIMIT: converted document exceeds output budget"
|
|
)
|
|
text = output_path.read_text(encoding="utf-8")
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_CONVERSION_TIMEOUT: exceeded {DOCUMENT_CONVERSION_TIMEOUT_SECONDS}s"
|
|
) from exc
|
|
except subprocess.CalledProcessError as exc:
|
|
detail = exc.stderr.decode("utf-8", errors="replace")[-1000:]
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_EXTRACTION_FAILED: converter exited {exc.returncode}: {detail}"
|
|
) from exc
|
|
except Exception as exc:
|
|
if isinstance(exc, DocumentExtractionError):
|
|
raise
|
|
raise DocumentExtractionError(
|
|
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
|
|
) from exc
|
|
finally:
|
|
for temporary in (source, output):
|
|
if temporary:
|
|
try:
|
|
os.unlink(temporary)
|
|
except OSError:
|
|
pass
|
|
if not isinstance(text, str) or not text.strip():
|
|
raise DocumentExtractionError(
|
|
"DOCUMENT_EXTRACTION_FAILED: document contains no extractable text"
|
|
)
|
|
return text.rstrip("\n") + "\n"
|