docx-integrity 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docx_integrity/__init__.py +40 -0
- docx_integrity/cli.py +179 -0
- docx_integrity/fidelity.py +101 -0
- docx_integrity/finding.py +89 -0
- docx_integrity/fonts.py +573 -0
- docx_integrity/inspector.py +423 -0
- docx_integrity/pptx_checks.py +253 -0
- docx_integrity/pptx_layout.py +665 -0
- docx_integrity-0.1.1.dist-info/METADATA +597 -0
- docx_integrity-0.1.1.dist-info/RECORD +13 -0
- docx_integrity-0.1.1.dist-info/WHEEL +4 -0
- docx_integrity-0.1.1.dist-info/entry_points.txt +2 -0
- docx_integrity-0.1.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""
|
|
2
|
+
docx-integrity - structural integrity checks for .docx files.
|
|
3
|
+
|
|
4
|
+
Two questions, both needed:
|
|
5
|
+
|
|
6
|
+
check(path) is this .docx self-consistent?
|
|
7
|
+
compare(source, edited) what did the edit lose relative to the source?
|
|
8
|
+
check_pptx(path) does this deck's text fit, and do shapes collide?
|
|
9
|
+
|
|
10
|
+
None of them needs a model, a renderer, or the network.
|
|
11
|
+
|
|
12
|
+
>>> from docx_integrity import check, compare
|
|
13
|
+
>>> for f in check("edited.docx"):
|
|
14
|
+
... print(f)
|
|
15
|
+
>>> for f in compare("original.docx", "edited.docx"):
|
|
16
|
+
... print(f)
|
|
17
|
+
"""
|
|
18
|
+
from .fidelity import TRACKED, compare
|
|
19
|
+
from .finding import ERROR, INFO, WARN, Finding, Severity, summarize, worst
|
|
20
|
+
from .inspector import Inspector, check, check_many
|
|
21
|
+
from .pptx_checks import check_pptx
|
|
22
|
+
|
|
23
|
+
__version__ = "0.1.1"
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"check",
|
|
27
|
+
"check_many",
|
|
28
|
+
"check_pptx",
|
|
29
|
+
"compare",
|
|
30
|
+
"Inspector",
|
|
31
|
+
"Finding",
|
|
32
|
+
"Severity",
|
|
33
|
+
"ERROR",
|
|
34
|
+
"WARN",
|
|
35
|
+
"INFO",
|
|
36
|
+
"summarize",
|
|
37
|
+
"worst",
|
|
38
|
+
"TRACKED",
|
|
39
|
+
"__version__",
|
|
40
|
+
]
|
docx_integrity/cli.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
"""Command line interface.
|
|
2
|
+
|
|
3
|
+
Exit codes are the contract with CI:
|
|
4
|
+
|
|
5
|
+
0 nothing at or above the --fail-on threshold
|
|
6
|
+
1 findings at or above the threshold
|
|
7
|
+
2 usage error, or a file that could not be read at all
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import glob
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from . import __version__
|
|
18
|
+
from .fidelity import compare
|
|
19
|
+
from .finding import Finding, Severity, summarize, worst
|
|
20
|
+
from .inspector import check
|
|
21
|
+
from .pptx_checks import check_pptx
|
|
22
|
+
|
|
23
|
+
EXIT_OK, EXIT_FINDINGS, EXIT_USAGE = 0, 1, 2
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
_GLOB_CHARS = "*?["
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _expand(patterns: list[str]) -> tuple[list[Path], list[str]]:
|
|
30
|
+
"""Expand globs ourselves so behaviour matches on every shell and OS.
|
|
31
|
+
|
|
32
|
+
A named path that does not exist is kept, so it gets reported as a finding
|
|
33
|
+
about that file rather than silently skipped. A *glob* that matches nothing
|
|
34
|
+
is different: there is no file to report on, and "file not found: *.docx"
|
|
35
|
+
would be a nonsense message. Those come back as `empty` for the caller to
|
|
36
|
+
treat as a usage error.
|
|
37
|
+
"""
|
|
38
|
+
found: list[Path] = []
|
|
39
|
+
empty: list[str] = []
|
|
40
|
+
for pat in patterns:
|
|
41
|
+
p = Path(pat)
|
|
42
|
+
if p.exists():
|
|
43
|
+
found.append(p)
|
|
44
|
+
continue
|
|
45
|
+
if any(c in pat for c in _GLOB_CHARS):
|
|
46
|
+
hits = sorted(glob.glob(pat, recursive=True))
|
|
47
|
+
if hits:
|
|
48
|
+
found.extend(Path(h) for h in hits)
|
|
49
|
+
else:
|
|
50
|
+
empty.append(pat)
|
|
51
|
+
else:
|
|
52
|
+
found.append(p)
|
|
53
|
+
return found, empty
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _run_one(path: Path, source: Path | None) -> list[Finding]:
|
|
57
|
+
if path.suffix.lower() in (".pptx", ".potx", ".ppsx"):
|
|
58
|
+
if source is not None:
|
|
59
|
+
return check_pptx(path) + [
|
|
60
|
+
Finding("FID000", Severity.INFO,
|
|
61
|
+
"--against is not implemented for .pptx yet; only layout "
|
|
62
|
+
"checks were run")
|
|
63
|
+
]
|
|
64
|
+
return check_pptx(path)
|
|
65
|
+
findings = check(path)
|
|
66
|
+
unreadable = any(f.code in ("PKG000", "PKG002") for f in findings)
|
|
67
|
+
if source is not None and not unreadable:
|
|
68
|
+
try:
|
|
69
|
+
findings = findings + compare(source, path)
|
|
70
|
+
except Exception as e:
|
|
71
|
+
findings = findings + [
|
|
72
|
+
Finding("FID000", Severity.WARN,
|
|
73
|
+
f"could not compare against {source}: {e}")
|
|
74
|
+
]
|
|
75
|
+
return findings
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _print_human(path: Path, findings: list[Finding], threshold: Severity,
|
|
79
|
+
quiet: bool, out) -> None:
|
|
80
|
+
shown = [f for f in findings if f.severity >= threshold] if quiet else findings
|
|
81
|
+
counts = summarize(findings)
|
|
82
|
+
head = (f"{path}: "
|
|
83
|
+
f"{counts['error']} error(s), {counts['warn']} warning(s), "
|
|
84
|
+
f"{counts['info']} info")
|
|
85
|
+
if not shown and not findings:
|
|
86
|
+
print(f"{head} - clean", file=out)
|
|
87
|
+
return
|
|
88
|
+
print(head, file=out)
|
|
89
|
+
for f in shown:
|
|
90
|
+
print(" " + str(f).replace("\n", "\n "), file=out)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
94
|
+
p = argparse.ArgumentParser(
|
|
95
|
+
prog="docx-integrity",
|
|
96
|
+
description="Structural integrity checks for .docx files. "
|
|
97
|
+
"Answers 'will Word open this, and did the edit lose "
|
|
98
|
+
"anything', which schema validation and rendering do not.",
|
|
99
|
+
epilog="exit codes: 0 clean, 1 findings at or above --fail-on, 2 usage error",
|
|
100
|
+
)
|
|
101
|
+
p.add_argument("--version", action="version",
|
|
102
|
+
version=f"docx-integrity {__version__}")
|
|
103
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
104
|
+
|
|
105
|
+
c = sub.add_parser("check", help="inspect one or more .docx files")
|
|
106
|
+
c.add_argument("files", nargs="+",
|
|
107
|
+
help="paths or globs, e.g. 'out/**/*.docx'. "
|
|
108
|
+
".pptx files get the layout checks instead")
|
|
109
|
+
c.add_argument("--against", metavar="SOURCE", type=Path, default=None,
|
|
110
|
+
help="also report what was lost relative to SOURCE")
|
|
111
|
+
c.add_argument("--fail-on", default="error", metavar="SEVERITY",
|
|
112
|
+
help="minimum severity that makes the run fail: "
|
|
113
|
+
"error (default), warn, info")
|
|
114
|
+
c.add_argument("--json", action="store_true",
|
|
115
|
+
help="machine-readable output on stdout")
|
|
116
|
+
c.add_argument("--quiet", "-q", action="store_true",
|
|
117
|
+
help="print only findings at or above --fail-on")
|
|
118
|
+
return p
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def main(argv: list[str] | None = None) -> int:
|
|
122
|
+
args = build_parser().parse_args(argv)
|
|
123
|
+
|
|
124
|
+
try:
|
|
125
|
+
threshold = Severity.parse(args.fail_on)
|
|
126
|
+
except ValueError as e:
|
|
127
|
+
print(f"docx-integrity: {e}", file=sys.stderr)
|
|
128
|
+
return EXIT_USAGE
|
|
129
|
+
|
|
130
|
+
if args.against is not None and not args.against.exists():
|
|
131
|
+
print(f"docx-integrity: --against file not found: {args.against}",
|
|
132
|
+
file=sys.stderr)
|
|
133
|
+
return EXIT_USAGE
|
|
134
|
+
|
|
135
|
+
paths, unmatched = _expand(args.files)
|
|
136
|
+
if unmatched and not paths:
|
|
137
|
+
print("docx-integrity: no files matched: " + ", ".join(unmatched),
|
|
138
|
+
file=sys.stderr)
|
|
139
|
+
return EXIT_USAGE
|
|
140
|
+
for pat in unmatched:
|
|
141
|
+
print(f"docx-integrity: warning: no files matched {pat}", file=sys.stderr)
|
|
142
|
+
if not paths:
|
|
143
|
+
print("docx-integrity: nothing to check", file=sys.stderr)
|
|
144
|
+
return EXIT_USAGE
|
|
145
|
+
|
|
146
|
+
results: dict[Path, list[Finding]] = {}
|
|
147
|
+
for path in paths:
|
|
148
|
+
results[path] = _run_one(path, args.against)
|
|
149
|
+
|
|
150
|
+
if args.json:
|
|
151
|
+
payload = {
|
|
152
|
+
"version": __version__,
|
|
153
|
+
"fail_on": threshold.value,
|
|
154
|
+
"files": [
|
|
155
|
+
{
|
|
156
|
+
"path": str(p),
|
|
157
|
+
"summary": summarize(f),
|
|
158
|
+
"worst": (w.value if (w := worst(f)) else None),
|
|
159
|
+
"findings": [x.as_dict() for x in f],
|
|
160
|
+
}
|
|
161
|
+
for p, f in results.items()
|
|
162
|
+
],
|
|
163
|
+
}
|
|
164
|
+
json.dump(payload, sys.stdout, indent=2, ensure_ascii=False)
|
|
165
|
+
sys.stdout.write("\n")
|
|
166
|
+
else:
|
|
167
|
+
for i, (p, f) in enumerate(results.items()):
|
|
168
|
+
if i:
|
|
169
|
+
print()
|
|
170
|
+
_print_human(p, f, threshold, args.quiet, sys.stdout)
|
|
171
|
+
|
|
172
|
+
failed = any(
|
|
173
|
+
f.severity >= threshold for findings in results.values() for f in findings
|
|
174
|
+
)
|
|
175
|
+
return EXIT_FINDINGS if failed else EXIT_OK
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
if __name__ == "__main__":
|
|
179
|
+
sys.exit(main())
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Fidelity check against the source document.
|
|
3
|
+
|
|
4
|
+
The inspector answers "is this file self-consistent?". That is not enough: a
|
|
5
|
+
document stripped of every style, footnote and revision is perfectly
|
|
6
|
+
self-consistent. A second question is needed - "what was lost relative to the
|
|
7
|
+
original?".
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import zipfile
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from lxml import etree
|
|
15
|
+
|
|
16
|
+
from .finding import ERROR, INFO, WARN, Finding
|
|
17
|
+
|
|
18
|
+
W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
|
|
19
|
+
|
|
20
|
+
#: (tag, human label, severity when some are lost)
|
|
21
|
+
#:
|
|
22
|
+
#: The severity rule: losing something that makes content or an audit trail
|
|
23
|
+
#: INVISIBLE is an error, because nothing downstream will report it. Losing
|
|
24
|
+
#: something that only changes how the document looks is a warning. Losing all
|
|
25
|
+
#: of any construct is always an error.
|
|
26
|
+
TRACKED: tuple[tuple[str, str, object], ...] = (
|
|
27
|
+
("commentReference", "comment anchors", ERROR),
|
|
28
|
+
("footnoteReference", "footnote references", ERROR),
|
|
29
|
+
("ins", "tracked insertions", ERROR),
|
|
30
|
+
("del", "tracked deletions", ERROR),
|
|
31
|
+
("sdt", "content controls", ERROR),
|
|
32
|
+
("drawing", "images and charts", ERROR),
|
|
33
|
+
("tbl", "tables", ERROR),
|
|
34
|
+
("hyperlink", "hyperlinks", WARN),
|
|
35
|
+
("pStyle", "paragraph style references", WARN),
|
|
36
|
+
("rStyle", "character style references", WARN),
|
|
37
|
+
("numPr", "numbered list items", WARN),
|
|
38
|
+
("tblHeader", "table header rows", WARN),
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
#: below this fraction of the source's text length, report FID003
|
|
42
|
+
TEXT_LOSS_THRESHOLD = 0.95
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _document(path: str | Path):
|
|
46
|
+
with zipfile.ZipFile(path) as z:
|
|
47
|
+
return etree.fromstring(z.read("word/document.xml"))
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _counts(path: str | Path) -> dict[str, int]:
|
|
51
|
+
doc = _document(path)
|
|
52
|
+
return {tag: len(list(doc.iter(W + tag))) for tag, _, _ in TRACKED}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _text(path: str | Path) -> str:
|
|
56
|
+
doc = _document(path)
|
|
57
|
+
return "".join(t.text or "" for t in doc.iter(W + "t"))
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def compare(source: str | Path, edited: str | Path) -> list[Finding]:
|
|
61
|
+
"""What did `edited` lose relative to `source`?
|
|
62
|
+
|
|
63
|
+
Raises the same exceptions as opening a zip - callers that may be handed a
|
|
64
|
+
corrupt file should run `check()` first, which reports rather than raises.
|
|
65
|
+
"""
|
|
66
|
+
before, after = _counts(source), _counts(edited)
|
|
67
|
+
out: list[Finding] = []
|
|
68
|
+
|
|
69
|
+
for tag, label, sev in TRACKED:
|
|
70
|
+
a, b = before[tag], after[tag]
|
|
71
|
+
if not a:
|
|
72
|
+
continue
|
|
73
|
+
if b < a:
|
|
74
|
+
lost = a - b
|
|
75
|
+
out.append(Finding(
|
|
76
|
+
"FID001",
|
|
77
|
+
ERROR if b == 0 else sev,
|
|
78
|
+
f"{label}: {a} -> {b} "
|
|
79
|
+
f'({"all lost" if b == 0 else f"{lost} lost"})',
|
|
80
|
+
extra={"tag": tag, "before": a, "after": b},
|
|
81
|
+
))
|
|
82
|
+
elif b > a:
|
|
83
|
+
# A higher count is not itself a defect: the agent may legitimately
|
|
84
|
+
# have added an item, or wrapped its edit in w:ins. Real duplication
|
|
85
|
+
# is caught by colliding ids (REV001), not by a counter.
|
|
86
|
+
out.append(Finding(
|
|
87
|
+
"FID002", INFO,
|
|
88
|
+
f"{label}: {a} -> {b} - added during editing "
|
|
89
|
+
"(only a defect if ids collide, see REV001)",
|
|
90
|
+
extra={"tag": tag, "before": a, "after": b},
|
|
91
|
+
))
|
|
92
|
+
|
|
93
|
+
ta, tb = _text(source), _text(edited)
|
|
94
|
+
if ta and len(tb) < len(ta) * TEXT_LOSS_THRESHOLD:
|
|
95
|
+
out.append(Finding(
|
|
96
|
+
"FID003", ERROR,
|
|
97
|
+
f"text volume fell from {len(ta)} to {len(tb)} characters "
|
|
98
|
+
f"({round(100 * (1 - len(tb) / len(ta)))}% of content lost)",
|
|
99
|
+
extra={"before": len(ta), "after": len(tb)},
|
|
100
|
+
))
|
|
101
|
+
return out
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""The single result type shared by the inspector and the fidelity check."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass, field, asdict
|
|
5
|
+
from enum import Enum
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Severity(str, Enum):
|
|
10
|
+
"""Ordered so comparisons work: ERROR > WARN > INFO."""
|
|
11
|
+
|
|
12
|
+
ERROR = "error"
|
|
13
|
+
WARN = "warn"
|
|
14
|
+
INFO = "info"
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def rank(self) -> int:
|
|
18
|
+
return {"error": 3, "warn": 2, "info": 1}[self.value]
|
|
19
|
+
|
|
20
|
+
def __ge__(self, other: "Severity") -> bool: # type: ignore[override]
|
|
21
|
+
return self.rank >= other.rank
|
|
22
|
+
|
|
23
|
+
def __gt__(self, other: "Severity") -> bool: # type: ignore[override]
|
|
24
|
+
return self.rank > other.rank
|
|
25
|
+
|
|
26
|
+
def __le__(self, other: "Severity") -> bool: # type: ignore[override]
|
|
27
|
+
return self.rank <= other.rank
|
|
28
|
+
|
|
29
|
+
def __lt__(self, other: "Severity") -> bool: # type: ignore[override]
|
|
30
|
+
return self.rank < other.rank
|
|
31
|
+
|
|
32
|
+
@classmethod
|
|
33
|
+
def parse(cls, s: str) -> "Severity":
|
|
34
|
+
try:
|
|
35
|
+
return cls(s.strip().lower())
|
|
36
|
+
except ValueError:
|
|
37
|
+
raise ValueError(
|
|
38
|
+
f"unknown severity {s!r}; expected one of: "
|
|
39
|
+
+ ", ".join(m.value for m in cls)
|
|
40
|
+
) from None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# Convenience aliases, so check code reads cleanly.
|
|
44
|
+
ERROR = Severity.ERROR
|
|
45
|
+
WARN = Severity.WARN
|
|
46
|
+
INFO = Severity.INFO
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class Finding:
|
|
51
|
+
"""One problem found in one document.
|
|
52
|
+
|
|
53
|
+
code stable identifier, e.g. CMT005 - safe to grep, safe to suppress
|
|
54
|
+
severity ERROR / WARN / INFO
|
|
55
|
+
message human-readable, says what breaks rather than what rule fired
|
|
56
|
+
where XPath to the offending node, or a part name, when known
|
|
57
|
+
part the package part the finding belongs to, when known
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
code: str
|
|
61
|
+
severity: Severity
|
|
62
|
+
message: str
|
|
63
|
+
where: str = ""
|
|
64
|
+
part: str = ""
|
|
65
|
+
extra: dict[str, Any] = field(default_factory=dict, compare=False)
|
|
66
|
+
|
|
67
|
+
def as_dict(self) -> dict[str, Any]:
|
|
68
|
+
d = asdict(self)
|
|
69
|
+
d["severity"] = self.severity.value
|
|
70
|
+
if not d["extra"]:
|
|
71
|
+
d.pop("extra")
|
|
72
|
+
return {k: v for k, v in d.items() if v != ""}
|
|
73
|
+
|
|
74
|
+
def __str__(self) -> str:
|
|
75
|
+
head = f"[{self.severity.value.upper():5}] {self.code} {self.message}"
|
|
76
|
+
return f"{head}\n -> {self.where}" if self.where else head
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def summarize(findings: list[Finding]) -> dict[str, int]:
|
|
80
|
+
"""Counts by severity, always with all three keys present."""
|
|
81
|
+
out = {m.value: 0 for m in Severity}
|
|
82
|
+
for f in findings:
|
|
83
|
+
out[f.severity.value] += 1
|
|
84
|
+
return out
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def worst(findings: list[Finding]) -> Severity | None:
|
|
88
|
+
"""Highest severity present, or None for an empty list."""
|
|
89
|
+
return max((f.severity for f in findings), key=lambda s: s.rank, default=None)
|