Update docx, pptx, and xlsx skills (#1447)

Add support for template formats (.dotx, .potx, .xltx) across validation
and the helper scripts.

Consolidate the shared office helpers into a single module, replacing the
pack/unpack pipeline with explicit zip/unzip steps. Extraction now rejects
symlink and path-traversal archive entries. Move run merging to a
standalone docx script, since it only ever applied to Word documents.

Fix the redlining validator so it compares against the original even when
a document has no tracked changes, which is when an untracked edit would
otherwise go unreported.

Provision a LibreOffice user profile per invocation so conversions work in
sandboxed environments.

Trim the skill docs to the guidance that earns its place.
This commit is contained in:
Peter Lai
2026-07-16 19:47:37 -07:00
committed by GitHub
parent 9d2f1ae187
commit fa0fa64bdc
53 changed files with 4147 additions and 4442 deletions
+173 -7
View File
@@ -3,6 +3,9 @@ Validator for PowerPoint presentation XML files against XSD schemas.
"""
import re
from pathlib import Path
from helpers import opc_target, rels_source_part, safe_extract
from .base import BaseSchemaValidator
@@ -57,8 +60,171 @@ class PPTXSchemaValidator(BaseSchemaValidator):
if not self.validate_no_duplicate_slide_layouts():
all_valid = False
if not self.validate_master_theme_uniqueness():
all_valid = False
if not self.validate_charts():
all_valid = False
if not self.validate_slides():
all_valid = False
return all_valid
def _package_map(self) -> dict:
wanted = []
wanted += list(self.unpacked_dir.glob("[[]Content_Types[]].xml"))
wanted += list(self.unpacked_dir.glob("ppt/presentation.xml"))
wanted += list(self.unpacked_dir.glob("ppt/theme/*.xml"))
wanted += list(self.unpacked_dir.glob("ppt/theme/_rels/*.rels"))
wanted += list(self.unpacked_dir.glob("ppt/charts/chart*.xml"))
for group in ("slideMasters", "notesMasters", "handoutMasters"):
wanted += list(self.unpacked_dir.glob(f"ppt/{group}/*.xml"))
wanted += list(self.unpacked_dir.glob(f"ppt/{group}/_rels/*.rels"))
return {
p.relative_to(self.unpacked_dir).as_posix(): p.read_bytes()
for p in wanted
if p.is_file()
}
def validate_master_theme_uniqueness(self):
from helpers.pptx_theme import _NOTES_MASTERS, live_shared_master_themes
shared = live_shared_master_themes(self._package_map())
if shared:
print(f"FAILED - Found {len(shared)} master(s) sharing a theme part:")
for message in shared:
print(f" {message}")
if any(m.startswith(_NOTES_MASTERS) for m in shared):
print(" Fix: in ppt/presentation.xml, move <p:notesMasterIdLst> back to "
"directly after <p:sldIdLst>. PowerPoint reads that happily.")
else:
print(" Fix: give each master its own theme part.")
return False
if self.verbose:
print("PASSED - No master shares a theme part in a way PowerPoint refuses")
return True
def validate_charts(self):
from helpers.pptx_chart import find_chart_problems
problems = find_chart_problems(self._package_map())
if problems:
print(f"FAILED - Found {len(problems)} chart problem(s) PowerPoint rejects:")
for message in problems:
print(f" {message}")
return False
if self.verbose:
print("PASSED - Charts satisfy the constraints PowerPoint enforces")
return True
def _original_slide_defects(self, schema) -> set[str]:
import tempfile
import zipfile
from helpers.pptx_slide import SLIDE_PART_RE, fatal_slide_errors
if self.original_file is None:
return set()
found: set[str] = set()
with tempfile.TemporaryDirectory() as temp_dir:
temp_path = Path(temp_dir)
try:
with zipfile.ZipFile(self.original_file, "r") as zf:
safe_extract(zf, temp_path)
except (zipfile.BadZipFile, ValueError, OSError):
return set()
for part in sorted(temp_path.rglob("*.xml")):
relative = part.relative_to(temp_path).as_posix()
if not SLIDE_PART_RE.fullmatch(relative):
continue
ok, errors = self._validate_single_file_xsd(
part.resolve(), temp_path.resolve(), schema_path=schema
)
if ok is None or ok or not errors:
continue
found |= set(fatal_slide_errors(set(errors)))
return found
def validate_slides(self):
from helpers.pptx_slide import (
SLIDE_PART_RE,
fatal_slide_errors,
is_schema_verdict,
)
schema = self.schemas_dir / self.SCHEMA_MAPPINGS["ppt"]
inherited = self._original_slide_defects(schema)
problems: list[str] = []
broken: list[str] = []
for xml_file in self.xml_files:
relative = xml_file.relative_to(self.unpacked_dir).as_posix()
if not SLIDE_PART_RE.fullmatch(relative):
continue
ok, errors = self._validate_single_file_xsd(
xml_file.resolve(), self.unpacked_dir.resolve(), schema_path=schema
)
if ok is None or not errors:
continue
unreadable = [f"{relative}: {e}" for e in errors if not is_schema_verdict(e)]
if unreadable:
broken.extend(unreadable)
continue
if ok:
continue
for message in fatal_slide_errors(set(errors)):
if message in inherited:
continue
problems.append(f"{relative}: {message}")
if broken:
print(f"FAILED - Could not check {len(broken)} slide part(s):")
for message in sorted(broken):
print(f" {message[:240]}")
if problems:
print(f"FAILED - Found {len(problems)} slide problem(s) PowerPoint rejects:")
for message in sorted(problems):
print(f" {message[:240]}")
if broken or problems:
return False
if self.verbose:
print("PASSED - Slide XML has none of the defects PowerPoint refuses")
return True
def _get_schema_path(self, xml_file):
if xml_file.parent.name == "charts" and xml_file.name.startswith("chart"):
return None
return super()._get_schema_path(xml_file)
def _preprocess_for_schema(self, xml_doc, relative_path):
if relative_path.as_posix() != "ppt/presentation.xml":
return xml_doc
root = xml_doc.getroot()
ns = f"{{{self.PRESENTATIONML_NAMESPACE}}}"
notes = root.find(f"{ns}notesMasterIdLst")
slides = root.find(f"{ns}sldIdLst")
if notes is None or slides is None:
return xml_doc
children = list(root)
if children.index(notes) < children.index(slides):
return xml_doc
root.remove(notes)
root.insert(list(root).index(slides), notes)
return xml_doc
def validate_uuid_ids(self):
import lxml.etree
@@ -229,17 +395,17 @@ class PPTXSchemaValidator(BaseSchemaValidator):
):
rel_type = rel.get("Type", "")
if "notesSlide" in rel_type:
target = rel.get("Target", "")
if target:
normalized_target = target.replace("../", "")
part = opc_target(
rel.get("Target", ""),
rels_source_part(rels_file, self.unpacked_dir),
rel.get("TargetMode", ""),
)
if part:
slide_name = rels_file.stem.replace(
".xml", ""
)
if normalized_target not in notes_slide_references:
notes_slide_references[normalized_target] = []
notes_slide_references[normalized_target].append(
notes_slide_references.setdefault(part, []).append(
(slide_name, rels_file)
)