|
|
@@ -16,15 +16,31 @@ import json
|
|
|
import posixpath
|
|
|
import sys
|
|
|
import zipfile
|
|
|
+import zlib
|
|
|
from pathlib import Path
|
|
|
from urllib.parse import unquote, urlsplit
|
|
|
from xml.etree import ElementTree as ET
|
|
|
|
|
|
-W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
|
|
|
-A = "{http://schemas.openxmlformats.org/drawingml/2006/main}"
|
|
|
-P = "{http://schemas.openxmlformats.org/presentationml/2006/main}"
|
|
|
-S = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
|
|
|
-R = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}"
|
|
|
+W_NAMESPACES = (
|
|
|
+ "http://schemas.openxmlformats.org/wordprocessingml/2006/main",
|
|
|
+ "http://purl.oclc.org/ooxml/wordprocessingml/main",
|
|
|
+)
|
|
|
+A_NAMESPACES = (
|
|
|
+ "http://schemas.openxmlformats.org/drawingml/2006/main",
|
|
|
+ "http://purl.oclc.org/ooxml/drawingml/main",
|
|
|
+)
|
|
|
+P_NAMESPACES = (
|
|
|
+ "http://schemas.openxmlformats.org/presentationml/2006/main",
|
|
|
+ "http://purl.oclc.org/ooxml/presentationml/main",
|
|
|
+)
|
|
|
+S_NAMESPACES = (
|
|
|
+ "http://schemas.openxmlformats.org/spreadsheetml/2006/main",
|
|
|
+ "http://purl.oclc.org/ooxml/spreadsheetml/main",
|
|
|
+)
|
|
|
+R_NAMESPACES = (
|
|
|
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
|
|
|
+ "http://purl.oclc.org/ooxml/officeDocument/relationships",
|
|
|
+)
|
|
|
MAIN_PARTS = {
|
|
|
".docx": ("word/document.xml", "wordprocessingml.document.main+xml"),
|
|
|
".pptx": ("ppt/presentation.xml", "presentationml.presentation.main+xml"),
|
|
|
@@ -32,6 +48,34 @@ MAIN_PARTS = {
|
|
|
}
|
|
|
|
|
|
|
|
|
+def namespace(root: ET.Element, supported: tuple[str, ...], part: str) -> str:
|
|
|
+ """Return the main XML namespace after checking its OOXML variant."""
|
|
|
+ uri = root.tag[1:].split("}", 1)[0] if root.tag.startswith("{") else ""
|
|
|
+ if uri not in supported:
|
|
|
+ raise ValueError(f"{part} uses unsupported XML namespace: {uri or '(none)'}")
|
|
|
+ return "{" + uri + "}"
|
|
|
+
|
|
|
+
|
|
|
+def relationship_id(node: ET.Element, part: str) -> str:
|
|
|
+ """Read an office-document relationship id from Transitional or Strict OOXML."""
|
|
|
+ for uri in R_NAMESPACES:
|
|
|
+ value = node.get("{" + uri + "}id")
|
|
|
+ if value is not None:
|
|
|
+ return value
|
|
|
+ raise ValueError(f"{part} has a reference without a relationship id")
|
|
|
+
|
|
|
+
|
|
|
+def relationship_types(kind: str) -> set[str]:
|
|
|
+ """Return the Transitional and Strict relationship type names for one role."""
|
|
|
+ return {f"{uri}/{kind}" for uri in R_NAMESPACES}
|
|
|
+
|
|
|
+
|
|
|
+def iter_namespaces(root: ET.Element, namespaces: tuple[str, ...], local_name: str):
|
|
|
+ """Iterate matching elements across Transitional and Strict namespaces."""
|
|
|
+ for uri in namespaces:
|
|
|
+ yield from root.iter("{" + uri + "}" + local_name)
|
|
|
+
|
|
|
+
|
|
|
def relationship_target(part: str, target: str) -> str:
|
|
|
"""Resolve a package relationship without fetching external resources."""
|
|
|
path = unquote(urlsplit(target).path)
|
|
|
@@ -62,78 +106,85 @@ def related_xml(part: str, reference: str, links: dict[str, str], xml: dict[str,
|
|
|
def inspect_docx(xml: dict[str, ET.Element]) -> tuple[dict, str]:
|
|
|
part = "word/document.xml"
|
|
|
root = xml[part]
|
|
|
- body = root.find(f"{W}body")
|
|
|
+ w = namespace(root, W_NAMESPACES, part)
|
|
|
+ body = root.find(f"{w}body")
|
|
|
if body is None:
|
|
|
raise ValueError("word/document.xml has no document body")
|
|
|
tables = []
|
|
|
- for table in body.iter(f"{W}tbl"):
|
|
|
- grid = table.findall(f"{W}tblGrid/{W}gridCol")
|
|
|
- rows = table.findall(f"{W}tr")
|
|
|
+ for table in body.iter(f"{w}tbl"):
|
|
|
+ grid = table.findall(f"{w}tblGrid/{w}gridCol")
|
|
|
+ rows = table.findall(f"{w}tr")
|
|
|
# Merged cells span logical grid columns; counting physical cells loses them.
|
|
|
columns = len(grid) if grid else max((sum(
|
|
|
- int(cell.find(f"{W}tcPr/{W}gridSpan").get(f"{W}val", "1"))
|
|
|
- if cell.find(f"{W}tcPr/{W}gridSpan") is not None else 1
|
|
|
- for cell in row.findall(f"{W}tc")
|
|
|
+ int(cell.find(f"{w}tcPr/{w}gridSpan").get(f"{w}val", "1"))
|
|
|
+ if cell.find(f"{w}tcPr/{w}gridSpan") is not None else 1
|
|
|
+ for cell in row.findall(f"{w}tc")
|
|
|
) for row in rows), default=0)
|
|
|
tables.append({"rows": len(rows), "columns": columns})
|
|
|
sections = []
|
|
|
- for section in body.iter(f"{W}sectPr"):
|
|
|
- size = section.find(f"{W}pgSz")
|
|
|
- margins = section.find(f"{W}pgMar")
|
|
|
+ for section in body.iter(f"{w}sectPr"):
|
|
|
+ size = section.find(f"{w}pgSz")
|
|
|
+ margins = section.find(f"{w}pgMar")
|
|
|
sections.append({
|
|
|
- "page_twips": {} if size is None else {key.removeprefix(W): value for key, value in size.attrib.items()},
|
|
|
- "margins_twips": {} if margins is None else {key.removeprefix(W): value for key, value in margins.attrib.items()},
|
|
|
+ "page_twips": {} if size is None else {key.removeprefix(w): value for key, value in size.attrib.items()},
|
|
|
+ "margins_twips": {} if margins is None else {key.removeprefix(w): value for key, value in margins.attrib.items()},
|
|
|
})
|
|
|
text_parts = [body]
|
|
|
for kind in ("header", "footer"):
|
|
|
- links = relationships(part, xml, {f"{R[1:-1]}/{kind}"})
|
|
|
- for section in body.iter(f"{W}sectPr"):
|
|
|
- for reference in section.findall(f"{W}{kind}Reference"):
|
|
|
- text_parts.append(related_xml(part, reference.attrib[f"{R}id"], links, xml))
|
|
|
+ links = relationships(part, xml, relationship_types(kind))
|
|
|
+ for section in body.iter(f"{w}sectPr"):
|
|
|
+ for reference in section.findall(f"{w}{kind}Reference"):
|
|
|
+ text_parts.append(related_xml(part, relationship_id(reference, part), links, xml))
|
|
|
for kind in ("footnote", "endnote"):
|
|
|
- references = {node.attrib[f"{W}id"] for node in body.iter(f"{W}{kind}Reference")}
|
|
|
+ references = {node.attrib[f"{w}id"] for node in body.iter(f"{w}{kind}Reference")}
|
|
|
if not references:
|
|
|
continue
|
|
|
- links = relationships(part, xml, {f"{R[1:-1]}/{kind}s"})
|
|
|
+ links = relationships(part, xml, relationship_types(f"{kind}s"))
|
|
|
for reference in links:
|
|
|
tree = related_xml(part, reference, links, xml)
|
|
|
- text_parts.extend(note for note in tree.findall(f"{W}{kind}") if note.get(f"{W}id") in references)
|
|
|
- text = "\n".join("".join(node.text or "" for node in paragraph.iter(f"{W}t")) for tree in text_parts
|
|
|
- for paragraph in tree.iter(f"{W}p"))
|
|
|
- return {"paragraphs": len(list(body.iter(f"{W}p"))), "tables": tables, "sections": sections}, text
|
|
|
+ text_parts.extend(note for note in tree.findall(f"{w}{kind}") if note.get(f"{w}id") in references)
|
|
|
+ text = "\n".join("".join(node.text or "" for node in paragraph.iter(f"{w}t")) for tree in text_parts
|
|
|
+ for paragraph in tree.iter(f"{w}p"))
|
|
|
+ return {"paragraphs": len(list(body.iter(f"{w}p"))), "tables": tables, "sections": sections}, text
|
|
|
|
|
|
|
|
|
def inspect_pptx(xml: dict[str, ET.Element]) -> tuple[dict, str]:
|
|
|
part = "ppt/presentation.xml"
|
|
|
links = relationships(part, xml)
|
|
|
- slides = xml[part].findall(f"{P}sldIdLst/{P}sldId")
|
|
|
+ root = xml[part]
|
|
|
+ p = namespace(root, P_NAMESPACES, part)
|
|
|
+ slides = root.findall(f"{p}sldIdLst/{p}sldId")
|
|
|
texts = []
|
|
|
for slide in slides:
|
|
|
- reference = slide.attrib[f"{R}id"]
|
|
|
+ reference = relationship_id(slide, part)
|
|
|
tree = related_xml(part, reference, links, xml)
|
|
|
- texts.append("\n".join("".join(node.text or "" for node in paragraph.iter(f"{A}t"))
|
|
|
- for paragraph in tree.iter(f"{A}p")))
|
|
|
+ texts.append("\n".join("".join(node.text or "" for node in iter_namespaces(paragraph, A_NAMESPACES, "t"))
|
|
|
+ for paragraph in iter_namespaces(tree, A_NAMESPACES, "p")))
|
|
|
return {"slides": len(slides)}, "\n".join(texts)
|
|
|
|
|
|
|
|
|
def inspect_xlsx(xml: dict[str, ET.Element]) -> tuple[dict, str]:
|
|
|
part = "xl/workbook.xml"
|
|
|
links = relationships(part, xml)
|
|
|
+ root = xml[part]
|
|
|
+ s = namespace(root, S_NAMESPACES, part)
|
|
|
sheets = []
|
|
|
texts = []
|
|
|
shared = xml.get("xl/sharedStrings.xml")
|
|
|
+ shared_s = s if shared is None else namespace(shared, S_NAMESPACES, "xl/sharedStrings.xml")
|
|
|
shared_strings = [] if shared is None else [
|
|
|
- "".join(node.text or "" for node in item.iter(f"{S}t")) for item in shared.iter(f"{S}si")
|
|
|
+ "".join(node.text or "" for node in item.iter(f"{shared_s}t")) for item in shared.iter(f"{shared_s}si")
|
|
|
]
|
|
|
- for sheet in xml[part].findall(f"{S}sheets/{S}sheet"):
|
|
|
- reference = sheet.attrib[f"{R}id"]
|
|
|
+ for sheet in root.findall(f"{s}sheets/{s}sheet"):
|
|
|
+ reference = relationship_id(sheet, part)
|
|
|
tree = related_xml(part, reference, links, xml)
|
|
|
- cells = list(tree.iter(f"{S}c"))
|
|
|
- formulas = sum(cell.find(f"{S}f") is not None for cell in cells)
|
|
|
+ sheet_s = namespace(tree, S_NAMESPACES, links[reference])
|
|
|
+ cells = list(tree.iter(f"{sheet_s}c"))
|
|
|
+ formulas = sum(cell.find(f"{sheet_s}f") is not None for cell in cells)
|
|
|
sheets.append({"name": sheet.attrib["name"], "cells": len(cells), "formulas": formulas})
|
|
|
for cell in cells:
|
|
|
if cell.get("t") == "s":
|
|
|
- value = cell.findtext(f"{S}v", "")
|
|
|
+ value = cell.findtext(f"{sheet_s}v", "")
|
|
|
location = f"{links[reference]} cell {cell.get('r', '(no reference)')}"
|
|
|
try:
|
|
|
index = int(value)
|
|
|
@@ -142,8 +193,8 @@ def inspect_xlsx(xml: dict[str, ET.Element]) -> tuple[dict, str]:
|
|
|
if not 0 <= index < len(shared_strings):
|
|
|
raise ValueError(f"{location}: shared string index out of range: {index}")
|
|
|
texts.append(shared_strings[index])
|
|
|
- texts.extend("".join(node.text or "" for node in cell.iter(f"{S}t")) for cell in cells)
|
|
|
- texts.extend(cell.findtext(f"{S}v", "") for cell in cells if cell.get("t") == "str")
|
|
|
+ texts.extend("".join(node.text or "" for node in cell.iter(f"{sheet_s}t")) for cell in cells)
|
|
|
+ texts.extend(cell.findtext(f"{sheet_s}v", "") for cell in cells if cell.get("t") == "str")
|
|
|
texts.append(sheet.attrib["name"])
|
|
|
return {"sheets": sheets, "formulas_evaluated": False}, "\n".join(texts)
|
|
|
|
|
|
@@ -206,7 +257,7 @@ def main() -> int:
|
|
|
if args.count is not None:
|
|
|
actual = summary.get("slides", len(summary.get("sheets", [])))
|
|
|
checks.append({"id": "count", "status": "pass" if actual == args.count else "fail", "expected": args.count, "actual": actual})
|
|
|
- except (OSError, ValueError, KeyError, RuntimeError, ET.ParseError, zipfile.BadZipFile) as error:
|
|
|
+ except (OSError, ValueError, KeyError, RuntimeError, ET.ParseError, zipfile.BadZipFile, zlib.error) as error:
|
|
|
checks.append({"id": "package", "status": "fail", "detail": str(error)})
|
|
|
failed = any(check["status"] == "fail" for check in checks)
|
|
|
report = {"format": args.input.suffix.lower()[1:], "verdict": "fail" if failed else "pass", "checks": checks, "summary": summary}
|