Files
EvoScientist-Multi/EvoScientist/document_extract.py
T
m4 c683f6e739
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
feat: prepare EvoScientist 0.3.0
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
2026-09-03 06:55:56 +08:00

467 lines
16 KiB
Python

"""Bounded document extraction and non-text file policy for workspaces."""
from __future__ import annotations
import os
import subprocess
import sys
import tempfile
import zipfile
from pathlib import Path
from xml.etree import ElementTree as ET
MAX_DOCUMENT_BYTES = 50 * 1024 * 1024
MAX_DOCUMENT_RESULT_CHARS = 50_000
MAX_CONVERTED_DOCUMENT_BYTES = 10 * 1024 * 1024
DOCUMENT_CONVERSION_TIMEOUT_SECONDS = 60
MAX_IMAGE_BYTES = 25 * 1024 * 1024
MAX_IMAGE_EDGE = 2048
MAX_IMAGE_PIXELS = 40_000_000
MAX_OOXML_MEMBERS = 10_000
MAX_OOXML_MEMBER_BYTES = 50 * 1024 * 1024
MAX_OOXML_EXPANDED_BYTES = 200 * 1024 * 1024
MAX_OOXML_COMPRESSION_RATIO = 100
IMAGE_EXTENSIONS = frozenset(
{".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".png", ".tif", ".tiff", ".webp"}
)
DOCUMENT_EXTENSIONS = frozenset(
{
".doc",
".docm",
".docx",
".epub",
".odp",
".ods",
".odt",
".pdf",
".pot",
".pps",
".ppsm",
".ppsx",
".ppt",
".pptm",
".pptx",
".rtf",
".xls",
".xlsb",
".xlsm",
".xlsx",
}
)
ARCHIVE_EXTENSIONS = frozenset(
{".7z", ".bz2", ".gz", ".rar", ".tar", ".tgz", ".xz", ".zip"}
)
DATABASE_EXTENSIONS = frozenset({".db", ".sqlite", ".sqlite3"})
EXECUTABLE_EXTENSIONS = frozenset(
{".app", ".deb", ".dll", ".dylib", ".elf", ".exe", ".msi", ".rpm", ".so"}
)
DATASET_EXTENSIONS = frozenset(
{".arrow", ".feather", ".h5", ".hdf5", ".npy", ".npz", ".parquet"}
)
MEDIA_EXTENSIONS = frozenset(
{".aac", ".avi", ".flac", ".m4a", ".mkv", ".mov", ".mp3", ".mp4", ".ogg", ".wav", ".webm"}
)
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
_A = "http://schemas.openxmlformats.org/drawingml/2006/main"
_S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
_R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
class DocumentExtractionError(RuntimeError):
"""A supported document could not be converted to bounded text."""
def classify_file(path: str, head: bytes = b"") -> str:
"""Classify a workspace file into one policy category."""
extension = Path(path).suffix.lower()
if extension in IMAGE_EXTENSIONS:
return "image"
if extension in DOCUMENT_EXTENSIONS:
return "document"
if extension in ARCHIVE_EXTENSIONS:
return "archive"
if extension in DATABASE_EXTENSIONS or head.startswith(b"SQLite format 3\x00"):
return "database"
if extension in EXECUTABLE_EXTENSIONS or head.startswith((b"MZ", b"\x7fELF")):
return "executable"
if extension in DATASET_EXTENSIONS:
return "dataset"
if extension in MEDIA_EXTENSIONS:
return "media"
return "unknown"
def binary_processing_guidance(path: str, kind: str, size_bytes: int) -> str:
"""Return bounded, actionable JSON-like guidance for Agent-side programming."""
import json
common = {
"code": "BINARY_PROCESSING_REQUIRED"
if kind != "binary"
else "UNSUPPORTED_BINARY_FILE",
"path": path,
"kind": kind,
"size_bytes": size_bytes,
}
if kind == "archive":
common.update(
action=(
"Use execute with Python to list and validate archive members before "
"selective extraction; never use extractall."
),
constraints={
"list_before_extract": True,
"max_members": 2000,
"max_total_uncompressed_bytes": 500 * 1024 * 1024,
"max_member_bytes": 100 * 1024 * 1024,
"max_compression_ratio": 100,
"reject_absolute_or_parent_paths": True,
"do_not_execute_members": True,
},
)
elif kind == "database":
common.update(
action=(
"Use execute with Python sqlite3 in read-only mode: "
"file:<path>?mode=ro&immutable=1; set PRAGMA query_only=ON; "
"inspect schema, run bounded SELECT queries with LIMIT, and write "
"large results under artifacts/."
),
constraints={
"read_only": True,
"mode": "mode=ro",
"query_only": True,
"max_rows": 1000,
"forbid_attach_database": True,
"forbid_load_extension": True,
},
)
elif kind == "executable":
common.update(
action=(
"Use execute only for bounded static metadata inspection (hash, file "
"headers, signature, imports, strings); this file must not be executed."
),
constraints={"must_not_be_executed": True, "static_analysis_only": True},
)
elif kind == "dataset":
common.update(
action=(
"Use execute with the appropriate library to inspect schema, dimensions, "
"statistics, and a bounded sample; do not serialize the whole dataset."
)
)
elif kind == "media":
common.update(
action=(
"Use execute with ffprobe/ffmpeg or an available transcription workflow "
"to inspect metadata and selected ranges; do not inline the complete file."
)
)
else:
common.update(
kind="binary",
action=(
"This is an unsupported binary file. Use execute only for bounded static "
"inspection; do not execute it or inline its bytes."
),
)
return json.dumps(common, ensure_ascii=False, sort_keys=True)
def prepare_image_bytes(data: bytes, path: str) -> bytes:
"""Validate and downsample an image before it becomes a model media block."""
import io
if len(data) > MAX_IMAGE_BYTES:
raise DocumentExtractionError(
f"IMAGE_TOO_LARGE: {len(data)} bytes exceeds {MAX_IMAGE_BYTES}"
)
try:
from PIL import Image
image = Image.open(io.BytesIO(data))
width, height = image.size
if width * height > MAX_IMAGE_PIXELS:
raise DocumentExtractionError(
f"IMAGE_PIXEL_BUDGET_EXCEEDED: {width}x{height} exceeds "
f"{MAX_IMAGE_PIXELS} pixels"
)
image.load()
except DocumentExtractionError:
raise
except Exception as exc:
raise DocumentExtractionError(
f"IMAGE_PROCESSING_FAILED: {path}: {type(exc).__name__}: {exc}"
) from exc
frame_count = int(getattr(image, "n_frames", 1) or 1)
if frame_count > 1:
image.seek(0)
work = image.convert("RGBA" if image.mode in {"RGBA", "LA"} else "RGB")
else:
work = image.copy()
if max(work.size) <= MAX_IMAGE_EDGE and frame_count == 1:
return data
work.thumbnail((MAX_IMAGE_EDGE, MAX_IMAGE_EDGE), Image.Resampling.LANCZOS)
has_alpha = work.mode in {"RGBA", "LA"} or (
work.mode == "P" and "transparency" in work.info
)
output = io.BytesIO()
if has_alpha:
if work.mode == "P":
work = work.convert("RGBA")
work.save(output, "PNG", optimize=True)
else:
if work.mode != "RGB":
work = work.convert("RGB")
work.save(output, "JPEG", quality=85, optimize=True)
return output.getvalue()
def extract_document_bytes(data: bytes, path: str) -> str:
"""Extract readable text from a supported document without exposing bytes."""
if len(data) > MAX_DOCUMENT_BYTES:
raise DocumentExtractionError(
f"DOCUMENT_TOO_LARGE: {len(data)} bytes exceeds {MAX_DOCUMENT_BYTES}"
)
extension = Path(path).suffix.lower()
try:
if extension == ".docx":
return _extract_docx(data)
if extension == ".pptx":
return _extract_pptx(data)
if extension == ".xlsx":
return _extract_xlsx(data)
return _extract_anydoc(data, extension)
except DocumentExtractionError:
raise
except Exception as exc:
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
) from exc
def paginate_document_text(text: str, *, offset: int, limit: int) -> str:
"""Apply line and character budgets to extracted document text."""
lines = text.splitlines(keepends=True)
if not lines:
return "(document contains no extractable text)"
if offset >= len(lines):
raise DocumentExtractionError(
f"Line offset {offset} exceeds extracted document length ({len(lines)} lines)"
)
selected = "".join(lines[offset : offset + limit])
if len(selected) <= MAX_DOCUMENT_RESULT_CHARS:
return selected
trimmed = selected[:MAX_DOCUMENT_RESULT_CHARS]
boundary = trimmed.rfind("\n")
if boundary > 0:
trimmed = trimmed[: boundary + 1]
consumed = max(1, len(trimmed.splitlines()))
return (
trimmed
+ f"\n[DOCUMENT_OUTPUT_TRUNCATED: use offset={offset + consumed} to continue; "
+ f"single-read limit is {MAX_DOCUMENT_RESULT_CHARS} characters]\n"
)
def _validated_ooxml_archive(data: bytes) -> zipfile.ZipFile:
try:
archive = zipfile.ZipFile(_bytes_path(data))
members = archive.infolist()
except zipfile.BadZipFile as exc:
raise DocumentExtractionError(
"DOCUMENT_EXTRACTION_FAILED: invalid OOXML container"
) from exc
expanded = 0
names: set[str] = set()
if len(members) > MAX_OOXML_MEMBERS:
archive.close()
raise DocumentExtractionError(
f"DOCUMENT_RESOURCE_LIMIT: OOXML has {len(members)} members; "
f"limit is {MAX_OOXML_MEMBERS}"
)
for member in members:
normalized = member.filename.replace("\\", "/")
parts = tuple(part for part in normalized.split("/") if part)
if (
normalized.startswith("/")
or ".." in parts
or member.filename in names
or bool(member.flag_bits & 0x1)
):
archive.close()
raise DocumentExtractionError(
"DOCUMENT_RESOURCE_LIMIT: OOXML contains an unsafe, duplicate, "
"or encrypted member"
)
names.add(member.filename)
expanded += member.file_size
ratio = member.file_size / max(member.compress_size, 1)
if (
member.file_size > MAX_OOXML_MEMBER_BYTES
or expanded > MAX_OOXML_EXPANDED_BYTES
or ratio > MAX_OOXML_COMPRESSION_RATIO
):
archive.close()
raise DocumentExtractionError(
"DOCUMENT_RESOURCE_LIMIT: OOXML member expansion exceeds safety limits"
)
return archive
def _zip_xml(data: bytes, member: str) -> ET.Element:
try:
with _validated_ooxml_archive(data) as archive:
raw = archive.read(member)
except KeyError as exc:
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: missing {member}"
) from exc
return ET.fromstring(raw)
def _bytes_path(data: bytes):
import io
return io.BytesIO(data)
def _ooxml_part_number(name: str) -> int:
stem = Path(name).stem
digits = "".join(character for character in stem if character.isdigit())
return int(digits) if digits else 0
def _extract_docx(data: bytes) -> str:
root = _zip_xml(data, "word/document.xml")
paragraphs: list[str] = []
for paragraph in root.iter(f"{{{_W}}}p"):
text = "".join(node.text or "" for node in paragraph.iter(f"{{{_W}}}t"))
if text:
paragraphs.append(text)
if not paragraphs:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: DOCX has no text")
return "\n".join(paragraphs) + "\n"
def _extract_pptx(data: bytes) -> str:
try:
with _validated_ooxml_archive(data) as archive:
names = sorted(
name
for name in archive.namelist()
if name.startswith("ppt/slides/slide") and name.endswith(".xml")
)
slides: list[str] = []
for index, name in enumerate(sorted(names, key=_ooxml_part_number), 1):
root = ET.fromstring(archive.read(name))
texts = [node.text or "" for node in root.iter(f"{{{_A}}}t")]
slides.append(f"## Slide {index}\n" + "\n".join(t for t in texts if t))
except zipfile.BadZipFile as exc:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid PPTX") from exc
if not slides:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: PPTX has no slides")
return "\n\n".join(slides) + "\n"
def _extract_xlsx(data: bytes) -> str:
try:
with _validated_ooxml_archive(data) as archive:
shared: list[str] = []
if "xl/sharedStrings.xml" in archive.namelist():
root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
shared = [
"".join(node.text or "" for node in item.iter(f"{{{_S}}}t"))
for item in root.iter(f"{{{_S}}}si")
]
sheets = sorted(
name
for name in archive.namelist()
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
)
output: list[str] = []
for index, name in enumerate(sheets, 1):
root = ET.fromstring(archive.read(name))
output.append(f"## Sheet {index}")
for row in root.iter(f"{{{_S}}}row"):
values: list[str] = []
for cell in row.iter(f"{{{_S}}}c"):
value_node = cell.find(f"{{{_S}}}v")
value = value_node.text if value_node is not None else ""
if cell.get("t") == "s" and value and value.isdigit():
shared_index = int(value)
value = shared[shared_index] if shared_index < len(shared) else value
values.append(value or "")
output.append("\t".join(values))
except zipfile.BadZipFile as exc:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid XLSX") from exc
if len(output) <= 1:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: XLSX has no sheets")
return "\n".join(output) + "\n"
def _extract_anydoc(data: bytes, extension: str) -> str:
source = ""
output = ""
try:
with tempfile.NamedTemporaryFile(suffix=extension, delete=False) as handle:
handle.write(data)
source = handle.name
with tempfile.NamedTemporaryFile(suffix=".md", delete=False) as handle:
output = handle.name
script = (
"import pathlib,sys; import anydoc; "
"text=anydoc.to_markdown(sys.argv[1]); "
"pathlib.Path(sys.argv[2]).write_text(text, encoding='utf-8')"
)
subprocess.run(
[sys.executable, "-c", script, source, output],
check=True,
capture_output=True,
timeout=DOCUMENT_CONVERSION_TIMEOUT_SECONDS,
)
output_path = Path(output)
if output_path.stat().st_size > MAX_CONVERTED_DOCUMENT_BYTES:
raise DocumentExtractionError(
"DOCUMENT_RESOURCE_LIMIT: converted document exceeds output budget"
)
text = output_path.read_text(encoding="utf-8")
except subprocess.TimeoutExpired as exc:
raise DocumentExtractionError(
f"DOCUMENT_CONVERSION_TIMEOUT: exceeded {DOCUMENT_CONVERSION_TIMEOUT_SECONDS}s"
) from exc
except subprocess.CalledProcessError as exc:
detail = exc.stderr.decode("utf-8", errors="replace")[-1000:]
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: converter exited {exc.returncode}: {detail}"
) from exc
except Exception as exc:
if isinstance(exc, DocumentExtractionError):
raise
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
) from exc
finally:
for temporary in (source, output):
if temporary:
try:
os.unlink(temporary)
except OSError:
pass
if not isinstance(text, str) or not text.strip():
raise DocumentExtractionError(
"DOCUMENT_EXTRACTION_FAILED: document contains no extractable text"
)
return text.rstrip("\n") + "\n"