Files
hermes-agent/skills/productivity/docx/scripts/docx_validate.py
T
Teknium fad88cf130 feat: extend clean-room office skills toward full parity
Same clean-room discipline as the initial rewrite (isolated subagents,
functional specs only, predecessor content banned including via git
history; transcripts retained). All additions test-proven.

docx (13->29 tests):
- docx_revisions.py: tracked changes list/accept/reject (all or by id),
  incl. tables and headers/footers, via direct oxml manipulation
- docx_comments.py: list/add/delete comments (native python-docx >=1.2
  API with XML fallback), anchored-text extraction
- docx_validate.py: package health check (rels, images, styles, CRC)
  with JSON severity report — explicitly not XSD validation
- docx_edit.py: run normalization; TOC + PAGE/NUMPAGES field insertion

xlsx (5->12 tests):
- xlsx_restructure.py: reference-aware insert/delete rows/cols —
  rewrites formulas on all sheets (absolute refs, ranges, cross-sheet,
  quoted names), shifts merges/autofilter/freeze/validation/CF ranges,
  tables, defined names; JSON report incl. honest not_shifted list
- native Excel tables, named ranges, hyperlinks, cell notes,
  sheet protection (documented as strippable, not security)
- xlsx_recalc.py: headless LibreOffice recalc with graceful degrade

powerpoint (11->21 tests):
- pptx_render.py: all slides -> PNGs (soffice + pdftoppm/pdftocairo),
  wired to vision_analyze review loop in SKILL.md
- run-merge normalize before replace (identical-format splits lossless)
- surgical chart ops (series/category/title) wrapping replace_data
- slide duplication with rel remap (clean refusal on chart slides)
- backgrounds, hyperlinks, slide numbers, footers, notes editing

pdf (8->21 tests):
- pdf_make_form.py: JSON spec -> AcroForm (text/checkbox/radio/dropdown)
- pdf_form_layout.py: pre-build layout lint (bounds/overlap/pairing)
  + rendered box overlay for vision_analyze review
- pdf_page_image.py + shared _raster.py: pypdfium2 -> pdftoppm chain,
  graceful degrade; connected to scanned-PDF triage flow
- pdf_stamp.py: text/image stamps at coordinates (rotation/opacity)
- pdf_meta.py: DocInfo metadata + attachments round-trip

Gates re-verified independently: 83 skill tests green under LC_ALL=C,
repo invariant suite 29/29, SkillEvaluator pii+unicode+lint 3/3 x4.
2026-08-08 10:46:20 -07:00

157 lines
6.1 KiB
Python

#!/usr/bin/env python3
# MIT License. Part of the Hermes docx skill.
"""Health-check a .docx package and report issues as JSON.
Usage: docx_validate.py file.docx
Checks (health-check tier, NOT full XSD schema validation):
- the file is a readable zip and python-docx can open it
- required package parts exist ([Content_Types].xml, document.xml)
- every relationship in every .rels file resolves to a part in the
package (dangling image/hyperlink/etc. rels are reported; external
targets such as hyperlinks are skipped)
- r:embed / r:id references in document.xml resolve to relationships
- embedded images are non-empty and start with known magic bytes
(PNG/JPEG/GIF/BMP/TIFF/EMF/WMF/SVG); no PIL required
- paragraph and run style ids referenced by the document exist in
styles.xml
Output: {"ok": bool, "issues": [{"severity": "error"|"warning", ...}]}
Exit code 1 when any error-severity issue is found (warnings exit 0).
"""
from __future__ import annotations
import argparse
import json
import posixpath
import sys
import zipfile
from lxml import etree
W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
PR = "http://schemas.openxmlformats.org/package/2006/relationships"
IMAGE_MAGIC = (
b"\x89PNG\r\n\x1a\n", b"\xff\xd8\xff", b"GIF87a", b"GIF89a",
b"BM", b"II*\x00", b"MM\x00*",
b"\x01\x00\x00\x00", # EMF
b"\xd7\xcd\xc6\x9a", b"\x01\x00\x09\x00", # WMF variants
b"<?xml", b"<svg",
)
def _issue(issues, severity, code, detail):
issues.append({"severity": severity, "code": code, "detail": detail})
def _rel_target(base_part: str, target: str) -> str:
base_dir = posixpath.dirname(base_part)
return posixpath.normpath(posixpath.join(base_dir, target)).lstrip("/")
def validate(path: str) -> dict:
issues: list[dict] = []
try:
zf = zipfile.ZipFile(path)
except (OSError, zipfile.BadZipFile) as exc:
_issue(issues, "error", "not-a-zip", str(exc))
return {"ok": False, "issues": issues}
names = set(zf.namelist())
bad = zf.testzip()
if bad is not None:
_issue(issues, "error", "corrupt-member", f"CRC check failed: {bad}")
for required in ("[Content_Types].xml", "word/document.xml"):
if required not in names:
_issue(issues, "error", "missing-part",
f"required part absent: {required}")
if issues and any(i["severity"] == "error" for i in issues):
return {"ok": False, "issues": issues}
# --- relationships resolve ------------------------------------------
rel_ids_by_source: dict[str, dict] = {}
for rels_name in [n for n in names if n.endswith(".rels")]:
try:
root = etree.fromstring(zf.read(rels_name))
except etree.XMLSyntaxError as exc:
_issue(issues, "error", "bad-rels-xml", f"{rels_name}: {exc}")
continue
source_part = posixpath.normpath(
posixpath.join(posixpath.dirname(rels_name), ".."))
source_part = "" if source_part == "." else source_part
ids = {}
for rel in root.iter(f"{{{PR}}}Relationship"):
rid, target = rel.get("Id"), rel.get("Target", "")
mode = rel.get("TargetMode", "Internal")
ids[rid] = target
if mode == "External":
continue
resolved = _rel_target(source_part + "/x" if source_part
else "x", target)
if resolved not in names:
_issue(issues, "error", "dangling-rel",
f"{rels_name}: {rid} -> {target} (missing part)")
rel_ids_by_source[source_part or "_package"] = ids
# --- r:id / r:embed references in document.xml -----------------------
doc_root = etree.fromstring(zf.read("word/document.xml"))
doc_rels = rel_ids_by_source.get("word", {})
for el in doc_root.iter():
for attr in (f"{{{R}}}id", f"{{{R}}}embed", f"{{{R}}}link"):
rid = el.get(attr)
if rid and rid not in doc_rels:
_issue(issues, "error", "unresolved-reference",
f"document.xml references {rid} with no relationship")
# --- embedded images decode ------------------------------------------
for name in [n for n in names if n.startswith("word/media/")]:
data = zf.read(name)
if not data:
_issue(issues, "error", "empty-image", name)
elif not any(data.startswith(m) for m in IMAGE_MAGIC):
_issue(issues, "warning", "unknown-image-format",
f"{name}: unrecognized magic bytes")
# --- styles referenced exist ------------------------------------------
defined = set()
if "word/styles.xml" in names:
styles_root = etree.fromstring(zf.read("word/styles.xml"))
defined = {s.get(f"{{{W}}}styleId")
for s in styles_root.iter(f"{{{W}}}style")}
for tag, attr in ((f"{{{W}}}pStyle", f"{{{W}}}val"),
(f"{{{W}}}rStyle", f"{{{W}}}val"),
(f"{{{W}}}tblStyle", f"{{{W}}}val")):
for el in doc_root.iter(tag):
sid = el.get(attr)
if sid and sid not in defined:
_issue(issues, "error", "missing-style",
f"style id referenced but not defined: {sid}")
# --- python-docx can open it ------------------------------------------
try:
from docx import Document
Document(path)
except Exception as exc: # noqa: BLE001 - triage tool, report anything
_issue(issues, "error", "python-docx-open-failed", str(exc))
ok = not any(i["severity"] == "error" for i in issues)
return {"ok": ok, "issues": issues}
def main() -> int:
ap = argparse.ArgumentParser(
description="Health-check a .docx (not XSD schema validation).")
ap.add_argument("path", help="the .docx file to check")
args = ap.parse_args()
report = validate(args.path)
print(json.dumps(report, ensure_ascii=False))
return 0 if report["ok"] else 1
if __name__ == "__main__":
sys.exit(main())