mirror of
https://github.com/anthropics/skills.git
synced 2026-08-02 21:15:27 +08:00
fa0fa64bdc
Add support for template formats (.dotx, .potx, .xltx) across validation and the helper scripts. Consolidate the shared office helpers into a single module, replacing the pack/unpack pipeline with explicit zip/unzip steps. Extraction now rejects symlink and path-traversal archive entries. Move run merging to a standalone docx script, since it only ever applied to Word documents. Fix the redlining validator so it compares against the original even when a document has no tracked changes, which is when an untracked edit would otherwise go unreported. Provision a LibreOffice user profile per invocation so conversions work in sandboxed environments. Trim the skill docs to the guidance that earns its place.
311 lines
9.2 KiB
Python
Executable File
311 lines
9.2 KiB
Python
Executable File
"""Merge adjacent identically-formatted runs in a DOCX.
|
|
|
|
Word fragments paragraph text across many <w:r> elements (revision ids,
|
|
spell-check markers, editing history), which makes find-and-replace on
|
|
word/document.xml unreliable — the string you're looking for is split
|
|
across runs. This coalesces adjacent runs whose formatting (<w:rPr>) is
|
|
identical, strips rsid attributes and proofErr markers, and consolidates the
|
|
text elements — <w:t>, and <w:delText> for text inside a tracked deletion.
|
|
|
|
Rendering is unchanged. The text you search is what Word draws, which is not
|
|
always the bytes in the file: an element without xml:space="preserve" has its
|
|
edge whitespace trimmed before it reaches the page, so `<w:t>Hello </w:t>`
|
|
followed by `<w:t>world</w:t>` reads "Helloworld" and merges to exactly that.
|
|
|
|
Runs in two different <w:ins>/<w:del> wrappers are never merged: that would
|
|
rewrite tracked-change structure, collapsing separate revisions into one.
|
|
|
|
Only word/document.xml is processed (not headers, footers, or footnotes).
|
|
|
|
Usage:
|
|
python merge_runs.py unpacked/ # after unzip, before editing
|
|
python merge_runs.py document.docx # rewrite in place
|
|
python merge_runs.py document.docx -o out.docx
|
|
"""
|
|
|
|
|
|
import argparse
|
|
import sys
|
|
import tempfile
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
import defusedxml.minidom
|
|
|
|
from office.helpers import XML_SPACE, rendered_text, rezip, safe_extract
|
|
|
|
WORDML_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
|
|
|
|
def merge_runs(input_dir: str) -> tuple[int, str]:
|
|
doc_xml = Path(input_dir) / "word" / "document.xml"
|
|
|
|
if not doc_xml.exists():
|
|
return 0, f"Error: {doc_xml} not found"
|
|
|
|
try:
|
|
dom = defusedxml.minidom.parseString(doc_xml.read_text(encoding="utf-8"))
|
|
root = dom.documentElement
|
|
run_names = _run_tag_names(root)
|
|
|
|
_remove_elements(root, "proofErr")
|
|
|
|
runs = _find_runs(root, run_names)
|
|
_strip_rsid_attrs(runs)
|
|
|
|
merge_count = 0
|
|
for container in {run.parentNode for run in runs}:
|
|
merge_count += _merge_runs_in(container, run_names)
|
|
|
|
doc_xml.write_bytes(dom.toxml(encoding="UTF-8"))
|
|
return merge_count, f"Merged {merge_count} runs"
|
|
|
|
except Exception as e:
|
|
return 0, f"Error: {e}"
|
|
|
|
|
|
|
|
|
|
def _is_element(node, tag: str) -> bool:
|
|
name = node.localName or node.tagName
|
|
return name == tag or name.endswith(f":{tag}")
|
|
|
|
|
|
def _run_tag_names(root) -> set[str]:
|
|
names = set()
|
|
for attr in root.attributes.values():
|
|
if attr.value == WORDML_NS:
|
|
if attr.name == "xmlns":
|
|
names.add("r")
|
|
elif attr.name.startswith("xmlns:"):
|
|
names.add(attr.name.split(":", 1)[1] + ":r")
|
|
return names or {"w:r", "r"}
|
|
|
|
|
|
def _find_elements(root, tag: str) -> list:
|
|
results = []
|
|
|
|
def traverse(node):
|
|
if node.nodeType == node.ELEMENT_NODE:
|
|
if _is_element(node, tag):
|
|
results.append(node)
|
|
for child in node.childNodes:
|
|
traverse(child)
|
|
|
|
traverse(root)
|
|
return results
|
|
|
|
|
|
def _find_runs(root, run_names: set[str]) -> list:
|
|
return [e for e in _find_elements(root, "r") if _is_run(e, run_names)]
|
|
|
|
|
|
def _get_child(parent, tag: str):
|
|
return next(iter(_get_children(parent, tag)), None)
|
|
|
|
|
|
def _get_children(parent, tag: str) -> list:
|
|
return [
|
|
child
|
|
for child in parent.childNodes
|
|
if child.nodeType == child.ELEMENT_NODE and _is_element(child, tag)
|
|
]
|
|
|
|
|
|
def _is_adjacent(elem1, elem2) -> bool:
|
|
node = elem1.nextSibling
|
|
while node:
|
|
if node == elem2:
|
|
return True
|
|
if node.nodeType == node.ELEMENT_NODE:
|
|
return False
|
|
if node.nodeType == node.TEXT_NODE and node.data.strip(XML_SPACE):
|
|
return False
|
|
node = node.nextSibling
|
|
return False
|
|
|
|
|
|
|
|
|
|
def _remove_elements(root, tag: str):
|
|
for elem in _find_elements(root, tag):
|
|
if elem.parentNode:
|
|
elem.parentNode.removeChild(elem)
|
|
|
|
|
|
def _strip_rsid_attrs(runs: list):
|
|
for run in runs:
|
|
for attr in list(run.attributes.values()):
|
|
if "rsid" in attr.name.lower():
|
|
run.removeAttribute(attr.name)
|
|
|
|
|
|
|
|
|
|
def _merge_runs_in(container, run_names: set[str]) -> int:
|
|
merge_count = 0
|
|
run = _first_child_run(container, run_names)
|
|
|
|
while run:
|
|
while True:
|
|
next_elem = _next_element_sibling(run)
|
|
if next_elem and _is_run(next_elem, run_names) and _can_merge(run, next_elem):
|
|
_merge_run_content(run, next_elem)
|
|
container.removeChild(next_elem)
|
|
merge_count += 1
|
|
else:
|
|
break
|
|
|
|
_consolidate_text(run)
|
|
run = _next_sibling_run(run, run_names)
|
|
|
|
return merge_count
|
|
|
|
|
|
def _first_child_run(container, run_names: set[str]):
|
|
for child in container.childNodes:
|
|
if child.nodeType == child.ELEMENT_NODE and _is_run(child, run_names):
|
|
return child
|
|
return None
|
|
|
|
|
|
def _next_element_sibling(node):
|
|
sibling = node.nextSibling
|
|
while sibling:
|
|
if sibling.nodeType == sibling.ELEMENT_NODE:
|
|
return sibling
|
|
sibling = sibling.nextSibling
|
|
return None
|
|
|
|
|
|
def _next_sibling_run(node, run_names: set[str]):
|
|
sibling = node.nextSibling
|
|
while sibling:
|
|
if sibling.nodeType == sibling.ELEMENT_NODE:
|
|
if _is_run(sibling, run_names):
|
|
return sibling
|
|
sibling = sibling.nextSibling
|
|
return None
|
|
|
|
|
|
def _is_run(node, run_names: set[str]) -> bool:
|
|
return node.tagName in run_names
|
|
|
|
|
|
def _can_merge(run1, run2) -> bool:
|
|
rpr1 = _get_child(run1, "rPr")
|
|
rpr2 = _get_child(run2, "rPr")
|
|
|
|
if (rpr1 is None) != (rpr2 is None):
|
|
return False
|
|
if rpr1 is None:
|
|
return True
|
|
return rpr1.toxml() == rpr2.toxml()
|
|
|
|
|
|
def _merge_run_content(target, source):
|
|
for child in list(source.childNodes):
|
|
if child.nodeType == child.ELEMENT_NODE:
|
|
name = child.localName or child.tagName
|
|
if name != "rPr" and not name.endswith(":rPr"):
|
|
target.appendChild(child)
|
|
|
|
|
|
def _element_text(elem) -> str:
|
|
return "".join(
|
|
child.data
|
|
for child in elem.childNodes
|
|
if child.nodeType in (child.TEXT_NODE, child.CDATA_SECTION_NODE)
|
|
)
|
|
|
|
|
|
def _has_preserve(elem) -> bool:
|
|
return elem.getAttribute("xml:space") == "preserve"
|
|
|
|
|
|
def _rendered_text(elem) -> str:
|
|
return rendered_text(_element_text(elem), _has_preserve(elem))
|
|
|
|
|
|
def _consolidate_text(run):
|
|
for tag in ("t", "delText"):
|
|
_consolidate_text_elements(run, tag)
|
|
|
|
|
|
def _consolidate_text_elements(run, tag: str):
|
|
t_elements = _get_children(run, tag)
|
|
|
|
for i in range(len(t_elements) - 1, 0, -1):
|
|
curr, prev = t_elements[i], t_elements[i - 1]
|
|
|
|
if _is_adjacent(prev, curr):
|
|
merged = _rendered_text(prev) + _rendered_text(curr)
|
|
had_preserve = _has_preserve(prev) or _has_preserve(curr)
|
|
|
|
new_text = run.ownerDocument.createTextNode(merged)
|
|
for node in list(prev.childNodes):
|
|
if node.nodeType in (node.TEXT_NODE, node.CDATA_SECTION_NODE):
|
|
prev.removeChild(node)
|
|
else:
|
|
run.insertBefore(node, curr)
|
|
prev.appendChild(new_text)
|
|
for node in list(curr.childNodes):
|
|
if node.nodeType not in (node.TEXT_NODE, node.CDATA_SECTION_NODE):
|
|
run.insertBefore(node, curr)
|
|
|
|
if merged != merged.strip(XML_SPACE) or had_preserve:
|
|
prev.setAttribute("xml:space", "preserve")
|
|
elif prev.hasAttribute("xml:space"):
|
|
prev.removeAttribute("xml:space")
|
|
|
|
run.removeChild(curr)
|
|
|
|
|
|
|
|
|
|
def _merge_or_die(path: Path) -> str:
|
|
_, msg = merge_runs(str(path))
|
|
if msg.startswith("Error"):
|
|
print(msg, file=sys.stderr)
|
|
sys.exit(1)
|
|
return msg
|
|
|
|
|
|
def main() -> None:
|
|
p = argparse.ArgumentParser(
|
|
description="Merge adjacent identically-formatted runs in a DOCX (directory or .docx file)."
|
|
)
|
|
p.add_argument("input", help="Unpacked DOCX directory OR a .docx/.dotx file")
|
|
p.add_argument(
|
|
"-o", "--output",
|
|
help="Output .docx path (only valid when input is a .docx; default: overwrite input)",
|
|
)
|
|
args = p.parse_args()
|
|
|
|
src = Path(args.input)
|
|
|
|
try:
|
|
if src.is_dir():
|
|
if args.output:
|
|
p.error("--output is only valid for .docx input; directory input is modified in place")
|
|
print(_merge_or_die(src))
|
|
elif src.is_file() and src.suffix.lower() in (".docx", ".dotx"):
|
|
out = Path(args.output) if args.output else src
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
tmp_path = Path(tmp)
|
|
with zipfile.ZipFile(src) as zf:
|
|
safe_extract(zf, tmp_path)
|
|
msg = _merge_or_die(tmp_path)
|
|
rezip(tmp_path, out)
|
|
print(f"{msg}; wrote {out}")
|
|
else:
|
|
print(f"Error: {src} is neither a directory nor a .docx/.dotx file", file=sys.stderr)
|
|
sys.exit(1)
|
|
except (OSError, ValueError, zipfile.BadZipFile) as e:
|
|
print(f"Error: {e}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|