docx-integrity 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,40 @@
1
+ """
2
+ docx-integrity - structural integrity checks for .docx files.
3
+
4
+ Two questions, both needed:
5
+
6
+ check(path) is this .docx self-consistent?
7
+ compare(source, edited) what did the edit lose relative to the source?
8
+ check_pptx(path) does this deck's text fit, and do shapes collide?
9
+
10
+ None of them needs a model, a renderer, or the network.
11
+
12
+ >>> from docx_integrity import check, compare
13
+ >>> for f in check("edited.docx"):
14
+ ... print(f)
15
+ >>> for f in compare("original.docx", "edited.docx"):
16
+ ... print(f)
17
+ """
18
+ from .fidelity import TRACKED, compare
19
+ from .finding import ERROR, INFO, WARN, Finding, Severity, summarize, worst
20
+ from .inspector import Inspector, check, check_many
21
+ from .pptx_checks import check_pptx
22
+
23
+ __version__ = "0.1.1"
24
+
25
+ __all__ = [
26
+ "check",
27
+ "check_many",
28
+ "check_pptx",
29
+ "compare",
30
+ "Inspector",
31
+ "Finding",
32
+ "Severity",
33
+ "ERROR",
34
+ "WARN",
35
+ "INFO",
36
+ "summarize",
37
+ "worst",
38
+ "TRACKED",
39
+ "__version__",
40
+ ]
docx_integrity/cli.py ADDED
@@ -0,0 +1,179 @@
1
+ """Command line interface.
2
+
3
+ Exit codes are the contract with CI:
4
+
5
+ 0 nothing at or above the --fail-on threshold
6
+ 1 findings at or above the threshold
7
+ 2 usage error, or a file that could not be read at all
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import glob
13
+ import json
14
+ import sys
15
+ from pathlib import Path
16
+
17
+ from . import __version__
18
+ from .fidelity import compare
19
+ from .finding import Finding, Severity, summarize, worst
20
+ from .inspector import check
21
+ from .pptx_checks import check_pptx
22
+
23
+ EXIT_OK, EXIT_FINDINGS, EXIT_USAGE = 0, 1, 2
24
+
25
+
26
+ _GLOB_CHARS = "*?["
27
+
28
+
29
+ def _expand(patterns: list[str]) -> tuple[list[Path], list[str]]:
30
+ """Expand globs ourselves so behaviour matches on every shell and OS.
31
+
32
+ A named path that does not exist is kept, so it gets reported as a finding
33
+ about that file rather than silently skipped. A *glob* that matches nothing
34
+ is different: there is no file to report on, and "file not found: *.docx"
35
+ would be a nonsense message. Those come back as `empty` for the caller to
36
+ treat as a usage error.
37
+ """
38
+ found: list[Path] = []
39
+ empty: list[str] = []
40
+ for pat in patterns:
41
+ p = Path(pat)
42
+ if p.exists():
43
+ found.append(p)
44
+ continue
45
+ if any(c in pat for c in _GLOB_CHARS):
46
+ hits = sorted(glob.glob(pat, recursive=True))
47
+ if hits:
48
+ found.extend(Path(h) for h in hits)
49
+ else:
50
+ empty.append(pat)
51
+ else:
52
+ found.append(p)
53
+ return found, empty
54
+
55
+
56
+ def _run_one(path: Path, source: Path | None) -> list[Finding]:
57
+ if path.suffix.lower() in (".pptx", ".potx", ".ppsx"):
58
+ if source is not None:
59
+ return check_pptx(path) + [
60
+ Finding("FID000", Severity.INFO,
61
+ "--against is not implemented for .pptx yet; only layout "
62
+ "checks were run")
63
+ ]
64
+ return check_pptx(path)
65
+ findings = check(path)
66
+ unreadable = any(f.code in ("PKG000", "PKG002") for f in findings)
67
+ if source is not None and not unreadable:
68
+ try:
69
+ findings = findings + compare(source, path)
70
+ except Exception as e:
71
+ findings = findings + [
72
+ Finding("FID000", Severity.WARN,
73
+ f"could not compare against {source}: {e}")
74
+ ]
75
+ return findings
76
+
77
+
78
+ def _print_human(path: Path, findings: list[Finding], threshold: Severity,
79
+ quiet: bool, out) -> None:
80
+ shown = [f for f in findings if f.severity >= threshold] if quiet else findings
81
+ counts = summarize(findings)
82
+ head = (f"{path}: "
83
+ f"{counts['error']} error(s), {counts['warn']} warning(s), "
84
+ f"{counts['info']} info")
85
+ if not shown and not findings:
86
+ print(f"{head} - clean", file=out)
87
+ return
88
+ print(head, file=out)
89
+ for f in shown:
90
+ print(" " + str(f).replace("\n", "\n "), file=out)
91
+
92
+
93
+ def build_parser() -> argparse.ArgumentParser:
94
+ p = argparse.ArgumentParser(
95
+ prog="docx-integrity",
96
+ description="Structural integrity checks for .docx files. "
97
+ "Answers 'will Word open this, and did the edit lose "
98
+ "anything', which schema validation and rendering do not.",
99
+ epilog="exit codes: 0 clean, 1 findings at or above --fail-on, 2 usage error",
100
+ )
101
+ p.add_argument("--version", action="version",
102
+ version=f"docx-integrity {__version__}")
103
+ sub = p.add_subparsers(dest="command", required=True)
104
+
105
+ c = sub.add_parser("check", help="inspect one or more .docx files")
106
+ c.add_argument("files", nargs="+",
107
+ help="paths or globs, e.g. 'out/**/*.docx'. "
108
+ ".pptx files get the layout checks instead")
109
+ c.add_argument("--against", metavar="SOURCE", type=Path, default=None,
110
+ help="also report what was lost relative to SOURCE")
111
+ c.add_argument("--fail-on", default="error", metavar="SEVERITY",
112
+ help="minimum severity that makes the run fail: "
113
+ "error (default), warn, info")
114
+ c.add_argument("--json", action="store_true",
115
+ help="machine-readable output on stdout")
116
+ c.add_argument("--quiet", "-q", action="store_true",
117
+ help="print only findings at or above --fail-on")
118
+ return p
119
+
120
+
121
+ def main(argv: list[str] | None = None) -> int:
122
+ args = build_parser().parse_args(argv)
123
+
124
+ try:
125
+ threshold = Severity.parse(args.fail_on)
126
+ except ValueError as e:
127
+ print(f"docx-integrity: {e}", file=sys.stderr)
128
+ return EXIT_USAGE
129
+
130
+ if args.against is not None and not args.against.exists():
131
+ print(f"docx-integrity: --against file not found: {args.against}",
132
+ file=sys.stderr)
133
+ return EXIT_USAGE
134
+
135
+ paths, unmatched = _expand(args.files)
136
+ if unmatched and not paths:
137
+ print("docx-integrity: no files matched: " + ", ".join(unmatched),
138
+ file=sys.stderr)
139
+ return EXIT_USAGE
140
+ for pat in unmatched:
141
+ print(f"docx-integrity: warning: no files matched {pat}", file=sys.stderr)
142
+ if not paths:
143
+ print("docx-integrity: nothing to check", file=sys.stderr)
144
+ return EXIT_USAGE
145
+
146
+ results: dict[Path, list[Finding]] = {}
147
+ for path in paths:
148
+ results[path] = _run_one(path, args.against)
149
+
150
+ if args.json:
151
+ payload = {
152
+ "version": __version__,
153
+ "fail_on": threshold.value,
154
+ "files": [
155
+ {
156
+ "path": str(p),
157
+ "summary": summarize(f),
158
+ "worst": (w.value if (w := worst(f)) else None),
159
+ "findings": [x.as_dict() for x in f],
160
+ }
161
+ for p, f in results.items()
162
+ ],
163
+ }
164
+ json.dump(payload, sys.stdout, indent=2, ensure_ascii=False)
165
+ sys.stdout.write("\n")
166
+ else:
167
+ for i, (p, f) in enumerate(results.items()):
168
+ if i:
169
+ print()
170
+ _print_human(p, f, threshold, args.quiet, sys.stdout)
171
+
172
+ failed = any(
173
+ f.severity >= threshold for findings in results.values() for f in findings
174
+ )
175
+ return EXIT_FINDINGS if failed else EXIT_OK
176
+
177
+
178
+ if __name__ == "__main__":
179
+ sys.exit(main())
@@ -0,0 +1,101 @@
1
+ """
2
+ Fidelity check against the source document.
3
+
4
+ The inspector answers "is this file self-consistent?". That is not enough: a
5
+ document stripped of every style, footnote and revision is perfectly
6
+ self-consistent. A second question is needed - "what was lost relative to the
7
+ original?".
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import zipfile
12
+ from pathlib import Path
13
+
14
+ from lxml import etree
15
+
16
+ from .finding import ERROR, INFO, WARN, Finding
17
+
18
+ W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
19
+
20
+ #: (tag, human label, severity when some are lost)
21
+ #:
22
+ #: The severity rule: losing something that makes content or an audit trail
23
+ #: INVISIBLE is an error, because nothing downstream will report it. Losing
24
+ #: something that only changes how the document looks is a warning. Losing all
25
+ #: of any construct is always an error.
26
+ TRACKED: tuple[tuple[str, str, object], ...] = (
27
+ ("commentReference", "comment anchors", ERROR),
28
+ ("footnoteReference", "footnote references", ERROR),
29
+ ("ins", "tracked insertions", ERROR),
30
+ ("del", "tracked deletions", ERROR),
31
+ ("sdt", "content controls", ERROR),
32
+ ("drawing", "images and charts", ERROR),
33
+ ("tbl", "tables", ERROR),
34
+ ("hyperlink", "hyperlinks", WARN),
35
+ ("pStyle", "paragraph style references", WARN),
36
+ ("rStyle", "character style references", WARN),
37
+ ("numPr", "numbered list items", WARN),
38
+ ("tblHeader", "table header rows", WARN),
39
+ )
40
+
41
+ #: below this fraction of the source's text length, report FID003
42
+ TEXT_LOSS_THRESHOLD = 0.95
43
+
44
+
45
+ def _document(path: str | Path):
46
+ with zipfile.ZipFile(path) as z:
47
+ return etree.fromstring(z.read("word/document.xml"))
48
+
49
+
50
+ def _counts(path: str | Path) -> dict[str, int]:
51
+ doc = _document(path)
52
+ return {tag: len(list(doc.iter(W + tag))) for tag, _, _ in TRACKED}
53
+
54
+
55
+ def _text(path: str | Path) -> str:
56
+ doc = _document(path)
57
+ return "".join(t.text or "" for t in doc.iter(W + "t"))
58
+
59
+
60
+ def compare(source: str | Path, edited: str | Path) -> list[Finding]:
61
+ """What did `edited` lose relative to `source`?
62
+
63
+ Raises the same exceptions as opening a zip - callers that may be handed a
64
+ corrupt file should run `check()` first, which reports rather than raises.
65
+ """
66
+ before, after = _counts(source), _counts(edited)
67
+ out: list[Finding] = []
68
+
69
+ for tag, label, sev in TRACKED:
70
+ a, b = before[tag], after[tag]
71
+ if not a:
72
+ continue
73
+ if b < a:
74
+ lost = a - b
75
+ out.append(Finding(
76
+ "FID001",
77
+ ERROR if b == 0 else sev,
78
+ f"{label}: {a} -> {b} "
79
+ f'({"all lost" if b == 0 else f"{lost} lost"})',
80
+ extra={"tag": tag, "before": a, "after": b},
81
+ ))
82
+ elif b > a:
83
+ # A higher count is not itself a defect: the agent may legitimately
84
+ # have added an item, or wrapped its edit in w:ins. Real duplication
85
+ # is caught by colliding ids (REV001), not by a counter.
86
+ out.append(Finding(
87
+ "FID002", INFO,
88
+ f"{label}: {a} -> {b} - added during editing "
89
+ "(only a defect if ids collide, see REV001)",
90
+ extra={"tag": tag, "before": a, "after": b},
91
+ ))
92
+
93
+ ta, tb = _text(source), _text(edited)
94
+ if ta and len(tb) < len(ta) * TEXT_LOSS_THRESHOLD:
95
+ out.append(Finding(
96
+ "FID003", ERROR,
97
+ f"text volume fell from {len(ta)} to {len(tb)} characters "
98
+ f"({round(100 * (1 - len(tb) / len(ta)))}% of content lost)",
99
+ extra={"before": len(ta), "after": len(tb)},
100
+ ))
101
+ return out
@@ -0,0 +1,89 @@
1
+ """The single result type shared by the inspector and the fidelity check."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import dataclass, field, asdict
5
+ from enum import Enum
6
+ from typing import Any
7
+
8
+
9
+ class Severity(str, Enum):
10
+ """Ordered so comparisons work: ERROR > WARN > INFO."""
11
+
12
+ ERROR = "error"
13
+ WARN = "warn"
14
+ INFO = "info"
15
+
16
+ @property
17
+ def rank(self) -> int:
18
+ return {"error": 3, "warn": 2, "info": 1}[self.value]
19
+
20
+ def __ge__(self, other: "Severity") -> bool: # type: ignore[override]
21
+ return self.rank >= other.rank
22
+
23
+ def __gt__(self, other: "Severity") -> bool: # type: ignore[override]
24
+ return self.rank > other.rank
25
+
26
+ def __le__(self, other: "Severity") -> bool: # type: ignore[override]
27
+ return self.rank <= other.rank
28
+
29
+ def __lt__(self, other: "Severity") -> bool: # type: ignore[override]
30
+ return self.rank < other.rank
31
+
32
+ @classmethod
33
+ def parse(cls, s: str) -> "Severity":
34
+ try:
35
+ return cls(s.strip().lower())
36
+ except ValueError:
37
+ raise ValueError(
38
+ f"unknown severity {s!r}; expected one of: "
39
+ + ", ".join(m.value for m in cls)
40
+ ) from None
41
+
42
+
43
+ # Convenience aliases, so check code reads cleanly.
44
+ ERROR = Severity.ERROR
45
+ WARN = Severity.WARN
46
+ INFO = Severity.INFO
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class Finding:
51
+ """One problem found in one document.
52
+
53
+ code stable identifier, e.g. CMT005 - safe to grep, safe to suppress
54
+ severity ERROR / WARN / INFO
55
+ message human-readable, says what breaks rather than what rule fired
56
+ where XPath to the offending node, or a part name, when known
57
+ part the package part the finding belongs to, when known
58
+ """
59
+
60
+ code: str
61
+ severity: Severity
62
+ message: str
63
+ where: str = ""
64
+ part: str = ""
65
+ extra: dict[str, Any] = field(default_factory=dict, compare=False)
66
+
67
+ def as_dict(self) -> dict[str, Any]:
68
+ d = asdict(self)
69
+ d["severity"] = self.severity.value
70
+ if not d["extra"]:
71
+ d.pop("extra")
72
+ return {k: v for k, v in d.items() if v != ""}
73
+
74
+ def __str__(self) -> str:
75
+ head = f"[{self.severity.value.upper():5}] {self.code} {self.message}"
76
+ return f"{head}\n -> {self.where}" if self.where else head
77
+
78
+
79
+ def summarize(findings: list[Finding]) -> dict[str, int]:
80
+ """Counts by severity, always with all three keys present."""
81
+ out = {m.value: 0 for m in Severity}
82
+ for f in findings:
83
+ out[f.severity.value] += 1
84
+ return out
85
+
86
+
87
+ def worst(findings: list[Finding]) -> Severity | None:
88
+ """Highest severity present, or None for an empty list."""
89
+ return max((f.severity for f in findings), key=lambda s: s.rank, default=None)