mirror of
https://github.com/anthropics/skills.git
synced 2026-08-02 21:15:27 +08:00
fa0fa64bdc
Add support for template formats (.dotx, .potx, .xltx) across validation and the helper scripts. Consolidate the shared office helpers into a single module, replacing the pack/unpack pipeline with explicit zip/unzip steps. Extraction now rejects symlink and path-traversal archive entries. Move run merging to a standalone docx script, since it only ever applied to Word documents. Fix the redlining validator so it compares against the original even when a document has no tracked changes, which is when an untracked edit would otherwise go unreported. Provision a LibreOffice user profile per invocation so conversions work in sandboxed environments. Trim the skill docs to the guidance that earns its place.
112 lines
3.3 KiB
Python
112 lines
3.3 KiB
Python
import os
|
|
import posixpath
|
|
import re
|
|
import stat
|
|
import tempfile
|
|
import urllib.parse
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
OOXML_FAMILY = {
|
|
".docx": "docx",
|
|
".dotx": "docx",
|
|
".pptx": "pptx",
|
|
".potx": "pptx",
|
|
".xlsx": "xlsx",
|
|
".xltx": "xlsx",
|
|
}
|
|
|
|
_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*:")
|
|
|
|
SLIDE_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/slide"
|
|
|
|
|
|
def opc_target(target: str, source_part: str, target_mode: str = "") -> str | None:
|
|
if not target:
|
|
return None
|
|
if target_mode.lower() == "external":
|
|
return None
|
|
if _SCHEME_RE.match(target):
|
|
return None
|
|
|
|
target = urllib.parse.unquote(target)
|
|
|
|
if "\\" in target:
|
|
raise ValueError(f"relationship target is not a POSIX part name: {target!r}")
|
|
|
|
if target.startswith("/"):
|
|
joined = target.lstrip("/")
|
|
else:
|
|
joined = posixpath.join(posixpath.dirname(source_part), target)
|
|
|
|
parts: list[str] = []
|
|
for segment in posixpath.normpath(joined).split("/"):
|
|
if segment in ("", "."):
|
|
continue
|
|
if segment == "..":
|
|
if not parts:
|
|
raise ValueError(f"relationship target escapes the package: {target!r}")
|
|
parts.pop()
|
|
else:
|
|
parts.append(segment)
|
|
|
|
if not parts:
|
|
raise ValueError(f"relationship target resolves to nothing: {target!r}")
|
|
return "/".join(parts)
|
|
|
|
|
|
def rels_source_part(rels_file: Path, unpacked_dir: Path) -> str:
|
|
owner_dir = rels_file.parent.parent.relative_to(unpacked_dir)
|
|
return posixpath.join(owner_dir.as_posix(), rels_file.name[: -len(".rels")]).lstrip("./")
|
|
|
|
|
|
def part_text(data: bytes) -> str:
|
|
return data.decode("utf-8", "surrogateescape")
|
|
|
|
|
|
XML_SPACE = " \t\r\n"
|
|
|
|
|
|
def rendered_text(text: str, preserve: bool) -> str:
|
|
return text if preserve else text.strip(XML_SPACE)
|
|
|
|
|
|
def safe_extract(zf: zipfile.ZipFile, dest: Path) -> None:
|
|
dest = dest.resolve()
|
|
for m in zf.infolist():
|
|
if stat.S_ISLNK(m.external_attr >> 16):
|
|
raise ValueError(f"symlink archive entry not allowed: {m.filename!r}")
|
|
target = (dest / m.filename).resolve()
|
|
if not target.is_relative_to(dest):
|
|
raise ValueError(f"unsafe archive entry: {m.filename!r}")
|
|
zf.extract(m, dest)
|
|
|
|
|
|
def rezip(src_dir: Path, out_path: Path) -> None:
|
|
files = sorted(p for p in src_dir.rglob("*") if p.is_file())
|
|
ct = src_dir / "[Content_Types].xml"
|
|
fd, tmp_name = tempfile.mkstemp(
|
|
prefix=out_path.name + ".", suffix=".tmp", dir=out_path.parent
|
|
)
|
|
tmp_out = Path(tmp_name)
|
|
try:
|
|
with os.fdopen(fd, "wb") as fh:
|
|
with zipfile.ZipFile(fh, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
if ct.exists():
|
|
zf.write(ct, ct.relative_to(src_dir), compress_type=zipfile.ZIP_STORED)
|
|
for f in files:
|
|
if f == ct:
|
|
continue
|
|
zf.write(f, f.relative_to(src_dir))
|
|
if out_path.exists():
|
|
mode = out_path.stat().st_mode & 0o777
|
|
else:
|
|
umask = os.umask(0)
|
|
os.umask(umask)
|
|
mode = 0o666 & ~umask
|
|
os.chmod(tmp_out, mode)
|
|
os.replace(tmp_out, out_path)
|
|
finally:
|
|
if tmp_out.exists():
|
|
tmp_out.unlink()
|