Update docx, pptx, and xlsx skills (#1447)

Add support for template formats (.dotx, .potx, .xltx) across validation
and the helper scripts.

Consolidate the shared office helpers into a single module, replacing the
pack/unpack pipeline with explicit zip/unzip steps. Extraction now rejects
symlink and path-traversal archive entries. Move run merging to a
standalone docx script, since it only ever applied to Word documents.

Fix the redlining validator so it compares against the original even when
a document has no tracked changes, which is when an untracked edit would
otherwise go unreported.

Provision a LibreOffice user profile per invocation so conversions work in
sandboxed environments.

Trim the skill docs to the guidance that earns its place.
This commit is contained in:
Peter Lai
2026-07-16 19:47:37 -07:00
committed by GitHub
parent 9d2f1ae187
commit fa0fa64bdc
53 changed files with 4147 additions and 4442 deletions
+56 -36
View File
@@ -6,10 +6,13 @@ import random
import re
import tempfile
import zipfile
from pathlib import Path
import defusedxml.minidom
import lxml.etree
from helpers import safe_extract
from .base import BaseSchemaValidator
@@ -186,7 +189,7 @@ class DOCXSchemaValidator(BaseSchemaValidator):
try:
with tempfile.TemporaryDirectory() as temp_dir:
with zipfile.ZipFile(original, "r") as zip_ref:
zip_ref.extractall(temp_dir)
safe_extract(zip_ref, Path(temp_dir))
doc_xml_path = temp_dir + "/word/document.xml"
root = lxml.etree.parse(doc_xml_path).getroot()
@@ -241,9 +244,12 @@ class DOCXSchemaValidator(BaseSchemaValidator):
return True
def compare_paragraph_counts(self):
original_count = self.count_paragraphs_in_original()
new_count = self.count_paragraphs_in_unpacked()
if self.original_file is None:
print(f"\nParagraphs: {new_count}")
return
original_count = self.count_paragraphs_in_original()
diff = new_count - original_count
diff_str = f"+{diff}" if diff > 0 else str(diff)
print(f"\nParagraphs: {original_count}{new_count} ({diff_str})")
@@ -260,9 +266,15 @@ class DOCXSchemaValidator(BaseSchemaValidator):
try:
for elem in lxml.etree.parse(str(xml_file)).iter():
if val := elem.get(para_id_attr):
if self._parse_id_value(val, base=16) >= 0x80000000:
try:
if self._parse_id_value(val, base=16) >= 0x80000000:
errors.append(
f" {xml_file.name}:{elem.sourceline}: paraId={val} >= 0x80000000"
)
except ValueError:
errors.append(
f" {xml_file.name}:{elem.sourceline}: paraId={val} >= 0x80000000"
f" {xml_file.name}:{elem.sourceline}: "
f"paraId={val} is not valid hex"
)
if val := elem.get(durable_id_attr):
@@ -279,13 +291,19 @@ class DOCXSchemaValidator(BaseSchemaValidator):
f"durableId={val} must be decimal in numbering.xml"
)
else:
if self._parse_id_value(val, base=16) >= 0x7FFFFFFF:
try:
if self._parse_id_value(val, base=16) >= 0x7FFFFFFF:
errors.append(
f" {xml_file.name}:{elem.sourceline}: "
f"durableId={val} >= 0x7FFFFFFF"
)
except ValueError:
errors.append(
f" {xml_file.name}:{elem.sourceline}: "
f"durableId={val} >= 0x7FFFFFFF"
f"durableId={val} is not valid hex"
)
except Exception:
pass
except lxml.etree.XMLSyntaxError:
continue
if errors:
print(f"FAILED - {len(errors)} ID constraint violations:")
@@ -389,52 +407,54 @@ class DOCXSchemaValidator(BaseSchemaValidator):
return repairs
def repair_durableId(self) -> int:
DURABLE_ID_ATTRS = ("w16cid:durableId", "w16cex:durableId")
repairs = 0
renames: dict = {}
for xml_file in self.xml_files:
try:
content = xml_file.read_text(encoding="utf-8")
dom = defusedxml.minidom.parseString(content)
is_numbering = xml_file.name == "numbering.xml"
base = 10 if is_numbering else 16
pending = []
seen_in_file = set()
modified = False
for elem in dom.getElementsByTagName("*"):
if not elem.hasAttribute("w16cid:durableId"):
continue
for attr_name in DURABLE_ID_ATTRS:
if not elem.hasAttribute(attr_name):
continue
durable_id = elem.getAttribute("w16cid:durableId")
needs_repair = False
if xml_file.name == "numbering.xml":
durable_id = elem.getAttribute(attr_name)
try:
needs_repair = (
self._parse_id_value(durable_id, base=10) >= 0x7FFFFFFF
)
except ValueError:
needs_repair = True
else:
try:
needs_repair = (
self._parse_id_value(durable_id, base=16) >= 0x7FFFFFFF
)
key = self._parse_id_value(durable_id, base=base)
needs_repair = key >= 0x7FFFFFFF
except ValueError:
key = durable_id
needs_repair = True
if needs_repair:
value = random.randint(1, 0x7FFFFFFE)
if xml_file.name == "numbering.xml":
new_id = str(value)
else:
new_id = f"{value:08X}"
if needs_repair:
if key in seen_in_file:
value = random.randint(1, 0x7FFFFFFE)
else:
seen_in_file.add(key)
if key not in renames:
renames[key] = random.randint(1, 0x7FFFFFFE)
value = renames[key]
new_id = str(value) if is_numbering else f"{value:08X}"
elem.setAttribute("w16cid:durableId", new_id)
print(
f" Repaired: {xml_file.name}: durableId {durable_id}{new_id}"
)
repairs += 1
modified = True
elem.setAttribute(attr_name, new_id)
pending.append(
f" Repaired: {xml_file.name}: durableId {durable_id}{new_id}"
)
modified = True
if modified:
xml_file.write_bytes(dom.toxml(encoding="UTF-8"))
for message in pending:
print(message)
repairs += len(pending)
except Exception:
pass