""" Validator for tracked changes in Word documents. Detects untracked edits in word/document.xml: text that differs from the original without a / wrapper recording it. The tracked changes that are new relative to the original are undone, and the result is compared against the original; whatever text still differs was edited without being tracked. Only the document body is compared. Headers, footers, footnotes and endnotes are separate parts and are not checked. """ import subprocess import tempfile import zipfile from pathlib import Path import defusedxml.ElementTree as ET from defusedxml.common import DefusedXmlException from helpers import rendered_text, safe_extract class RedliningValidator: def __init__(self, unpacked_dir, original_docx, verbose=False): self.unpacked_dir = Path(unpacked_dir) self.original_docx = Path(original_docx) self.verbose = verbose self.namespaces = { "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main" } def repair(self) -> int: return 0 def validate(self): modified_file = self.unpacked_dir / "word" / "document.xml" if not modified_file.exists(): print(f"FAILED - Modified document.xml not found at {modified_file}") return False with tempfile.TemporaryDirectory() as temp_dir: temp_path = Path(temp_dir) try: with zipfile.ZipFile(self.original_docx, "r") as zip_ref: safe_extract(zip_ref, temp_path) except Exception as e: print(f"FAILED - Error unpacking original docx: {e}") return False original_file = temp_path / "word" / "document.xml" if not original_file.exists(): print( f"FAILED - Original document.xml not found in {self.original_docx}" ) return False try: modified_tree = ET.parse(modified_file) modified_root = modified_tree.getroot() original_tree = ET.parse(original_file) original_root = original_tree.getroot() except (ET.ParseError, DefusedXmlException) as e: print(f"FAILED - Error parsing XML files: {e}") return False new_changes = self._new_tracked_changes(original_root, modified_root) self._remove_tracked_changes(modified_root, new_changes) modified_text = self._extract_text_content(modified_root) original_text = self._extract_text_content(original_root) if modified_text != original_text: error_message = self._generate_detailed_diff( original_text, modified_text ) print(error_message) return False if self.verbose: print( f"PASSED - All {len(new_changes)} change(s) against the original " "are properly tracked" ) return True def _tracked_change_elements(self, root): ins_tag = f"{{{self.namespaces['w']}}}ins" del_tag = f"{{{self.namespaces['w']}}}del" return [elem for elem in root.iter() if elem.tag in (ins_tag, del_tag)] def _rendered_text(self, elem): preserve = elem.get("{http://www.w3.org/XML/1998/namespace}space") == "preserve" return rendered_text(elem.text or "", preserve) def _text_elements(self, elem): w = self.namespaces["w"] return [ node for node in elem.iter() if node.tag in (f"{{{w}}}t", f"{{{w}}}delText") ] def _tracked_change_key(self, elem): w = self.namespaces["w"] text = "".join(self._rendered_text(node) for node in self._text_elements(elem)) return (elem.tag, elem.get(f"{{{w}}}author"), elem.get(f"{{{w}}}date"), text) def _new_tracked_changes(self, original_root, modified_root): original = self._tracked_change_elements(original_root) modified = self._tracked_change_elements(modified_root) pool = {} for elem in original: pool.setdefault(self._tracked_change_key(elem), []).append(elem) matched, leftover = set(), [] for elem in modified: bucket = pool.get(self._tracked_change_key(elem)) if bucket: matched.add(bucket.pop()) else: leftover.append(elem) def group(elem): return self._tracked_change_key(elem)[:3] def text_of(elems): return "".join(self._tracked_change_key(e)[3] for e in elems) unmatched_original = {} for elem in original: if elem not in matched: unmatched_original.setdefault(group(elem), []).append(elem) by_group = {} for elem in leftover: by_group.setdefault(group(elem), []).append(elem) new = set() for key, elems in by_group.items(): rebuilt = text_of(elems) if rebuilt and rebuilt == text_of(unmatched_original.get(key, [])): continue new.update(elems) return new def _generate_detailed_diff(self, original_text, modified_text): error_parts = [ "FAILED - Document text doesn't match after removing the tracked changes", "", "Likely causes:", " 1. Modified text inside another author's or tags", " 2. Made edits without proper tracked changes", " 3. Didn't nest inside when deleting another's insertion", " 4. Rewrote another author's / and changed its text on", " the way. A tracked change from the original is recognised by its", " author, date and text; anything that doesn't reproduce one exactly", " reads as new, and the text it carried is reported missing.", "", "For pre-redlined documents, use correct patterns:", " - To reject another's INSERTION: Nest inside their ", " - To reject PART of one: nest around only the runs you reject.", " Their may be split around it, so long as the pieces keep", " their author and date and still spell out the same text.", " - To restore another's DELETION: Add new AFTER their ", "", ] git_diff = self._get_git_word_diff(original_text, modified_text) if git_diff: error_parts.extend(["Differences:", "============", git_diff]) else: error_parts.append("Unable to generate word diff (git not available)") return "\n".join(error_parts) def _get_git_word_diff(self, original_text, modified_text): try: with tempfile.TemporaryDirectory() as temp_dir: temp_path = Path(temp_dir) original_file = temp_path / "original.txt" modified_file = temp_path / "modified.txt" original_file.write_text(original_text, encoding="utf-8") modified_file.write_text(modified_text, encoding="utf-8") result = subprocess.run( [ "git", "diff", "--word-diff=plain", "--word-diff-regex=.", "-U0", "--no-index", str(original_file), str(modified_file), ], capture_output=True, text=True, ) if result.stdout.strip(): lines = result.stdout.split("\n") content_lines = [] in_content = False for line in lines: if line.startswith("@@"): in_content = True continue if in_content and line.strip(): content_lines.append(line) if content_lines: return "\n".join(content_lines) result = subprocess.run( [ "git", "diff", "--word-diff=plain", "-U0", "--no-index", str(original_file), str(modified_file), ], capture_output=True, text=True, ) if result.stdout.strip(): lines = result.stdout.split("\n") content_lines = [] in_content = False for line in lines: if line.startswith("@@"): in_content = True continue if in_content and line.strip(): content_lines.append(line) return "\n".join(content_lines) except (subprocess.CalledProcessError, FileNotFoundError, Exception): pass return None def _remove_tracked_changes(self, root, targets): ins_tag = f"{{{self.namespaces['w']}}}ins" del_tag = f"{{{self.namespaces['w']}}}del" for parent in root.iter(): to_remove = [] for child in parent: if child.tag == ins_tag and child in targets: to_remove.append(child) for elem in to_remove: parent.remove(elem) deltext_tag = f"{{{self.namespaces['w']}}}delText" t_tag = f"{{{self.namespaces['w']}}}t" for parent in root.iter(): to_process = [] for child in parent: if child.tag == del_tag and child in targets: to_process.append((child, list(parent).index(child))) for del_elem, del_index in reversed(to_process): for elem in del_elem.iter(): if elem.tag == deltext_tag: elem.tag = t_tag for child in reversed(list(del_elem)): parent.insert(del_index, child) parent.remove(del_elem) def _extract_text_content(self, root): p_tag = f"{{{self.namespaces['w']}}}p" t_tag = f"{{{self.namespaces['w']}}}t" paragraphs = [] for p_elem in root.findall(f".//{p_tag}"): text_parts = [] for t_elem in p_elem.findall(f".//{t_tag}"): text_parts.append(self._rendered_text(t_elem)) paragraph_text = "".join(text_parts) if paragraph_text: paragraphs.append(paragraph_text) return "\n".join(paragraphs) if __name__ == "__main__": raise RuntimeError("This module should not be run directly.")