pdfua 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pdfua/__init__.py ADDED
@@ -0,0 +1,46 @@
1
+ """pdfua — check the machine-checkable subset of PDF/UA-1 without a JVM.
2
+
3
+ Quick start::
4
+
5
+ from pdfua import validate
6
+
7
+ report = validate("document.pdf")
8
+ print(report.exit_code)
9
+ for finding in report.findings:
10
+ print(finding.rule_id, finding.severity.name, finding.message)
11
+
12
+ This library checks a subset of PDF/UA-1 (ISO 14289-1) and the WCAG 2.1
13
+ criteria that are decidable from the PDF alone. It does not certify
14
+ conformance. See :func:`pdfua.catalog.coverage` for what is implemented.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from .catalog import Coverage, coverage, describe_rules
20
+ from .document import PdfDocument
21
+ from .errors import PdfuaError, RuleSelectionError, UnreadableDocumentError
22
+ from .model import Confidence, Finding, Location, Report, Severity
23
+ from .rules import default_registry
24
+ from .validator import Validator, ValidatorOptions, validate
25
+
26
+ __version__ = "0.1.1"
27
+
28
+ __all__ = [
29
+ "Confidence",
30
+ "Coverage",
31
+ "Finding",
32
+ "Location",
33
+ "PdfDocument",
34
+ "PdfuaError",
35
+ "Report",
36
+ "RuleSelectionError",
37
+ "Severity",
38
+ "UnreadableDocumentError",
39
+ "Validator",
40
+ "ValidatorOptions",
41
+ "__version__",
42
+ "coverage",
43
+ "default_registry",
44
+ "describe_rules",
45
+ "validate",
46
+ ]
pdfua/__main__.py ADDED
@@ -0,0 +1,10 @@
1
+ """Support ``python -m pdfua`` by delegating to the console entry point."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+ from .cli import main
8
+
9
+ if __name__ == "__main__": # pragma: no cover
10
+ sys.exit(main())
pdfua/catalog.py ADDED
@@ -0,0 +1,185 @@
1
+ """The PDF/UA-1 rule catalogue, and coverage accounting over it.
2
+
3
+ This module is named ``catalog`` rather than ``coverage`` deliberately: the
4
+ package exports a ``coverage()`` function as its public API, and a module of the
5
+ same name would shadow it, so ``pdfua.coverage`` would sometimes be a function
6
+ and sometimes a module depending on import order. ``pdfua.catalog.coverage()``
7
+ and ``from pdfua import coverage`` both work and never collide.
8
+
9
+ Coverage accounting: how many PDF/UA-1 rules this tool actually checks.
10
+
11
+ The point of this module is honesty, and honesty here is subtler than it looks.
12
+
13
+ A validator that reports "PASS" is making an implicit claim about everything it
14
+ did not check. The obvious way to quantify that is per ISO clause: "we cover
15
+ clauses 5, 6.2, 7.1, 7.2, …". That measure is *misleading*, and this module
16
+ rejects it. Clause 7.2 contains **41** distinct rules; implementing one rule that
17
+ mentions clause 7.2 does not make the other forty checked. Reporting "7.2: 41/41"
18
+ because of a single `/Lang` check would be exactly the false comfort this project
19
+ exists to avoid.
20
+
21
+ Coverage is therefore counted at **rule granularity**: each implemented rule
22
+ declares, via :attr:`Rule.covers`, the exact set of PDF/UA-1 rule identifiers
23
+ (``clause``-``test``) it decides. The union of those sets is the numerator; the
24
+ 106 identifiers in the bundled catalogue are the denominator. A rule that
25
+ declares no identifier — because its check does not correspond to a single
26
+ profile rule — contributes nothing, and says so.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import json
32
+ from collections.abc import Mapping, Sequence
33
+ from dataclasses import dataclass
34
+ from functools import lru_cache
35
+ from importlib import resources
36
+
37
+ from .rules import RuleRegistry, default_registry
38
+
39
+ #: Number of machine-checkable rules in the PDF/UA-1 (ISO 14289-1) profile
40
+ #: published by veraPDF. This is the denominator used throughout the README and
41
+ #: the CLI. It is asserted in tests against the bundled identifier list, so it
42
+ #: cannot silently drift.
43
+ PDFUA1_TOTAL_RULES = 106
44
+
45
+
46
+ @dataclass(frozen=True, slots=True)
47
+ class RuleInfo:
48
+ """Public description of one implemented rule."""
49
+
50
+ id: str
51
+ title: str
52
+ severity: str
53
+ confidence: str
54
+ pdfua_clause: str | None
55
+ wcag: tuple[str, ...]
56
+ covers: tuple[str, ...]
57
+
58
+
59
+ @dataclass(frozen=True, slots=True)
60
+ class Coverage:
61
+ """How much of PDF/UA-1 this build checks, counted at rule granularity."""
62
+
63
+ covered: int
64
+ total: int
65
+ implemented_rules: int
66
+ partial: bool = False
67
+
68
+ @property
69
+ def fraction(self) -> float:
70
+ return self.covered / self.total if self.total else 0.0
71
+
72
+ def summary(self) -> str:
73
+ return (
74
+ f"{self.covered} of {self.total} PDF/UA-1 rules "
75
+ f"({self.fraction:.0%}) across {self.implemented_rules} implemented checks"
76
+ )
77
+
78
+
79
+ @lru_cache(maxsize=1)
80
+ def pdfua1_rule_identifiers() -> tuple[Mapping[str, str], ...]:
81
+ """Return the clause/test/object identifiers of every PDF/UA-1 rule.
82
+
83
+ The data file contains identifiers only — clause number, test number, and
84
+ the name of the validation object. It deliberately does not reproduce the
85
+ descriptive text of any validation profile, which belongs to its authors.
86
+ """
87
+ text = resources.files("pdfua.data").joinpath("pdfua1_rules.json").read_text("utf-8")
88
+ payload = json.loads(text)
89
+ rules = payload["rules"]
90
+ if not isinstance(rules, list):
91
+ raise ValueError("bundled rule catalogue is malformed")
92
+ return tuple(rules)
93
+
94
+
95
+ def _identifier(entry: Mapping[str, str]) -> str:
96
+ return f"{entry['clause']}-{entry['test']}"
97
+
98
+
99
+ @lru_cache(maxsize=1)
100
+ def _all_identifiers() -> frozenset[str]:
101
+ return frozenset(_identifier(e) for e in pdfua1_rule_identifiers())
102
+
103
+
104
+ def coverage(registry: RuleRegistry | None = None) -> Coverage:
105
+ """Return the rule-granularity coverage of ``registry``.
106
+
107
+ Unknown identifiers declared by a rule are counted in the numerator only if
108
+ they exist in the bundled catalogue; a typo in ``covers`` therefore lowers
109
+ the reported coverage rather than inflating it. Tests assert that every
110
+ declared identifier is real.
111
+ """
112
+ reg = registry if registry is not None else default_registry()
113
+ known = _all_identifiers()
114
+ declared: set[str] = set()
115
+ for rule in reg:
116
+ declared.update(i for i in rule.covers if i in known)
117
+ return Coverage(
118
+ covered=len(declared),
119
+ total=PDFUA1_TOTAL_RULES,
120
+ implemented_rules=len(reg),
121
+ )
122
+
123
+
124
+ def describe_rules(registry: RuleRegistry | None = None) -> tuple[RuleInfo, ...]:
125
+ """Return a public description of every rule in ``registry``."""
126
+ reg = registry if registry is not None else default_registry()
127
+ return tuple(
128
+ RuleInfo(
129
+ id=r.id,
130
+ title=r.title,
131
+ severity=r.severity.name.lower(),
132
+ confidence=r.confidence.value,
133
+ pdfua_clause=r.pdfua_clause,
134
+ wcag=r.wcag,
135
+ covers=r.covers,
136
+ )
137
+ for r in reg
138
+ )
139
+
140
+
141
+ def unchecked_rules(registry: RuleRegistry | None = None) -> tuple[Mapping[str, str], ...]:
142
+ """Return the PDF/UA-1 rules with no implemented check.
143
+
144
+ This is the honest "what is missing" list, at rule granularity. It is the
145
+ list the README's "What this does NOT check" section is built from.
146
+ """
147
+ reg = registry if registry is not None else default_registry()
148
+ covered = {i for rule in reg for i in rule.covers}
149
+ return tuple(e for e in pdfua1_rule_identifiers() if _identifier(e) not in covered)
150
+
151
+
152
+ def clause_summary(
153
+ registry: RuleRegistry | None = None,
154
+ ) -> tuple[tuple[str, int, int], ...]:
155
+ """Return ``(clause, covered, total)`` per clause, at rule granularity.
156
+
157
+ ``total`` is the number of PDF/UA-1 rules in that clause and ``covered`` is
158
+ how many of them this tool decides. Both numbers are real; neither is
159
+ inferred from the presence of a single check.
160
+ """
161
+ reg = registry if registry is not None else default_registry()
162
+ covered_ids = {i for rule in reg for i in rule.covers}
163
+
164
+ counts: dict[str, list[int]] = {}
165
+ for entry in pdfua1_rule_identifiers():
166
+ clause = str(entry["clause"])
167
+ bucket = counts.setdefault(clause, [0, 0])
168
+ bucket[1] += 1
169
+ if _identifier(entry) in covered_ids:
170
+ bucket[0] += 1
171
+
172
+ def sort_key(item: tuple[str, list[int]]) -> tuple[int, ...]:
173
+ return tuple(int(p) for p in item[0].split(".") if p.isdigit())
174
+
175
+ return tuple(
176
+ (clause, checked, total)
177
+ for clause, (checked, total) in sorted(counts.items(), key=sort_key)
178
+ )
179
+
180
+
181
+ def uncovered_clause_rules(
182
+ registry: RuleRegistry | None = None,
183
+ ) -> Sequence[Mapping[str, str]]:
184
+ """Alias for :func:`unchecked_rules`, kept for readability at call sites."""
185
+ return unchecked_rules(registry)
pdfua/cli.py ADDED
@@ -0,0 +1,278 @@
1
+ """Command-line interface: ``pdfua check <file>`` and ``pdfua rules``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import contextlib
7
+ import io
8
+ import os
9
+ import sys
10
+ from collections.abc import Sequence
11
+
12
+ from . import __version__
13
+ from .catalog import (
14
+ PDFUA1_TOTAL_RULES,
15
+ clause_summary,
16
+ coverage,
17
+ describe_rules,
18
+ pdfua1_rule_identifiers,
19
+ unchecked_rules,
20
+ )
21
+ from .errors import PdfuaError
22
+ from .model import Severity, filter_findings
23
+ from .reporters import FORMATTERS, format_report
24
+ from .rules import RuleRegistry, default_registry
25
+ from .validator import Validator, ValidatorOptions
26
+
27
+ EXIT_OK = 0
28
+ EXIT_FINDINGS = 1
29
+ EXIT_USAGE = 3
30
+ EXIT_UNREADABLE = 4
31
+
32
+ _EPILOG = """\
33
+ exit codes:
34
+ 0 no findings
35
+ 1 findings at or above --min-severity (default: any)
36
+ 2 findings at ERROR severity (see below)
37
+ 3 usage error
38
+ 4 the file could not be read as a PDF
39
+
40
+ When --min-severity is the default, the exit code is ordered by severity:
41
+ 0 clean, 1 info, 2 warning, 3 error. So `pdfua check f.pdf || alert` alerts on
42
+ anything, and `pdfua check f.pdf; case $? in 3) alert;; esac` alerts only on
43
+ errors. Gate on SARIF or JSON if you need the finding detail.
44
+ """
45
+
46
+
47
+ def build_parser() -> argparse.ArgumentParser:
48
+ """Construct the argument parser for the ``pdfua`` CLI."""
49
+ parser = argparse.ArgumentParser(
50
+ prog="pdfua",
51
+ description=(
52
+ "Check the machine-checkable subset of PDF/UA-1 and the related "
53
+ "WCAG 2.1 criteria. No JVM required."
54
+ ),
55
+ epilog=_EPILOG,
56
+ formatter_class=argparse.RawDescriptionHelpFormatter,
57
+ )
58
+ parser.add_argument("--version", action="version", version=f"pdfua {__version__}")
59
+ sub = parser.add_subparsers(dest="command", required=True)
60
+
61
+ check = sub.add_parser("check", help="validate one or more PDF files")
62
+ check.add_argument("files", nargs="+", metavar="FILE", help="PDF file(s) to check")
63
+ check.add_argument(
64
+ "--format",
65
+ "-f",
66
+ choices=sorted(FORMATTERS),
67
+ default="text",
68
+ help="output format (default: text)",
69
+ )
70
+ check.add_argument(
71
+ "--rules",
72
+ metavar="ID[,ID...]",
73
+ default=None,
74
+ help="only run these rules; a prefix selects a family (e.g. UA-01)",
75
+ )
76
+ check.add_argument(
77
+ "--min-severity",
78
+ choices=[s.name.lower() for s in Severity],
79
+ default=None,
80
+ help="suppress findings below this severity",
81
+ )
82
+ check.add_argument(
83
+ "--certain-only",
84
+ action="store_true",
85
+ help="suppress heuristic findings (those needing human judgement)",
86
+ )
87
+ check.add_argument(
88
+ "--fail-fast",
89
+ action="store_true",
90
+ help="stop after the first rule that reports a finding",
91
+ )
92
+ check.add_argument(
93
+ "--quiet",
94
+ "-q",
95
+ action="store_true",
96
+ help="with multiple files, print only files that have findings",
97
+ )
98
+
99
+ rules_cmd = sub.add_parser("rules", help="list implemented and unimplemented rules")
100
+ rules_cmd.add_argument(
101
+ "--format",
102
+ "-f",
103
+ choices=("text", "json"),
104
+ default="text",
105
+ help="output format (default: text)",
106
+ )
107
+ return parser
108
+
109
+
110
+ def _select_registry(spec: str | None) -> RuleRegistry:
111
+ registry = default_registry()
112
+ if spec is None:
113
+ return registry
114
+ patterns = [p.strip() for p in spec.split(",") if p.strip()]
115
+ try:
116
+ return registry.select(patterns)
117
+ except ValueError as exc:
118
+ raise PdfuaError(str(exc)) from exc
119
+
120
+
121
+ def _cmd_check(args: argparse.Namespace) -> int:
122
+ try:
123
+ registry = _select_registry(args.rules)
124
+ except PdfuaError as exc:
125
+ print(f"pdfua: {exc}", file=sys.stderr)
126
+ return EXIT_USAGE
127
+
128
+ min_severity = Severity.from_name(args.min_severity) if args.min_severity else None
129
+ options = ValidatorOptions(registry=registry, fail_fast=args.fail_fast)
130
+ validator = Validator(options)
131
+ coverage_info = coverage(registry)
132
+
133
+ exit_code = EXIT_OK
134
+ for path in args.files:
135
+ try:
136
+ report = validator.validate(path)
137
+ except PdfuaError as exc:
138
+ print(f"pdfua: {exc}", file=sys.stderr)
139
+ exit_code = max(exit_code, EXIT_UNREADABLE)
140
+ continue
141
+
142
+ kept = filter_findings(
143
+ report.findings,
144
+ min_severity=min_severity,
145
+ min_confidence=None,
146
+ only_rules=None,
147
+ )
148
+ if args.certain_only:
149
+ from .model import Confidence
150
+
151
+ kept = filter_findings(kept, min_confidence=Confidence.CERTAIN)
152
+
153
+ if args.quiet and not kept:
154
+ continue
155
+
156
+ # Rebuild the report so the formatter sees only the kept findings.
157
+ from .model import Report, RuleOutcome
158
+
159
+ filtered = Report(
160
+ path=report.path,
161
+ outcomes=tuple(
162
+ RuleOutcome(
163
+ rule_id=o.rule_id,
164
+ findings=tuple(f for f in o.findings if f in kept),
165
+ skipped_reason=o.skipped_reason,
166
+ duration_ms=o.duration_ms,
167
+ )
168
+ for o in report.outcomes
169
+ ),
170
+ pdfua_version=report.pdfua_version,
171
+ duration_ms=report.duration_ms,
172
+ )
173
+ print(format_report(filtered, args.format, coverage_info=coverage_info))
174
+
175
+ worst = filtered.worst_severity()
176
+ if worst is not None:
177
+ code = int(worst) + 1 if min_severity is None else EXIT_FINDINGS
178
+ exit_code = max(exit_code, code)
179
+
180
+ return exit_code
181
+
182
+
183
+ def _cmd_rules(args: argparse.Namespace) -> int:
184
+ implemented = describe_rules()
185
+ cov = coverage()
186
+ if args.format == "json":
187
+ import json
188
+
189
+ payload = {
190
+ "pdfua1_total_rules": PDFUA1_TOTAL_RULES,
191
+ "implemented": [
192
+ {
193
+ "id": r.id,
194
+ "title": r.title,
195
+ "severity": r.severity,
196
+ "confidence": r.confidence,
197
+ "pdfua_clause": r.pdfua_clause,
198
+ "wcag": list(r.wcag),
199
+ "covers": list(r.covers),
200
+ }
201
+ for r in implemented
202
+ ],
203
+ "covered_rule_identifiers": sorted({i for r in describe_rules() for i in r.covers}),
204
+ "clause_summary": [
205
+ {"clause": c, "covered": n, "total": t} for c, n, t in clause_summary()
206
+ ],
207
+ "unchecked_rules": [
208
+ {"clause": e["clause"], "test": e["test"], "object": e["object"]}
209
+ for e in unchecked_rules()
210
+ ],
211
+ }
212
+ print(json.dumps(payload, indent=2))
213
+ return EXIT_OK
214
+
215
+ print(f"pdfua {__version__}")
216
+ print()
217
+ print(f"Implemented: {cov.summary()}")
218
+ print(" (denominator = machine-checkable rules in the PDF/UA-1 profile,")
219
+ print(f" {len(pdfua1_rule_identifiers())} rule identifiers bundled)")
220
+ print()
221
+ print(f"{'RULE ID':<12} {'SEV':<8} {'CONF':<10} {'PDF/UA-1':<20} {'WCAG':<14} TITLE")
222
+ for r in implemented:
223
+ wcag = ",".join(r.wcag) if r.wcag else "-"
224
+ covers = ",".join(r.covers) if r.covers else "(no profile rule)"
225
+ print(f"{r.id:<12} {r.severity:<8} {r.confidence:<10} {covers:<20} {wcag:<14} {r.title}")
226
+ print()
227
+ print("Coverage by ISO 14289-1 clause, counted at rule granularity.")
228
+ print("Clause 7.21 covers font embedding, CIDSet and glyph widths — the rules")
229
+ print("that catch professionally-remediated files:")
230
+ print()
231
+ print(f" {'CLAUSE':<10} {'COVERED/TOTAL':<15} BAR")
232
+ for clause, checked, total in clause_summary():
233
+ bar = "#" * round(12 * checked / total)
234
+ flag = " <-- no coverage" if checked == 0 else ""
235
+ print(f" {clause:<10} {f'{checked}/{total}':<15} {bar:<12}{flag}")
236
+ print()
237
+ print("Everything not listed above is unimplemented. In particular this tool")
238
+ print("does NOT check font embedding, CIDSet, glyph widths, colour spaces,")
239
+ print("reading order, or WTPDF structure. See the README section")
240
+ print("'What this does NOT check'.")
241
+ return EXIT_OK
242
+
243
+
244
+ def main(argv: Sequence[str] | None = None) -> int:
245
+ """Entry point. Returns a process exit code."""
246
+ parser = build_parser()
247
+ args = parser.parse_args(argv)
248
+ try:
249
+ if args.command == "check":
250
+ return _cmd_check(args)
251
+ if args.command == "rules":
252
+ return _cmd_rules(args)
253
+ except PdfuaError as exc:
254
+ print(f"pdfua: {exc}", file=sys.stderr)
255
+ return EXIT_USAGE
256
+ except KeyboardInterrupt:
257
+ return 130
258
+ except BrokenPipeError:
259
+ # `pdfua check *.pdf | head` is a normal thing to do. Python turns the
260
+ # closed pipe into an exception on the next write, which would print a
261
+ # traceback for a command that did exactly what the user asked. Redirect
262
+ # stdout to devnull so the interpreter's own flush at shutdown does not
263
+ # raise a second time.
264
+ # The documented recipe: redirect the underlying file descriptor to
265
+ # devnull so the interpreter's final flush does not raise a second time.
266
+ # A new Python file object is avoided deliberately, because it would be
267
+ # collected unclosed and emit a ResourceWarning.
268
+ with contextlib.suppress(OSError, ValueError, io.UnsupportedOperation):
269
+ devnull = os.open(os.devnull, os.O_WRONLY)
270
+ os.dup2(devnull, sys.stdout.fileno())
271
+ os.close(devnull)
272
+ return EXIT_OK
273
+ parser.print_help()
274
+ return EXIT_USAGE
275
+
276
+
277
+ if __name__ == "__main__": # pragma: no cover
278
+ sys.exit(main())
pdfua/data/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """Bundled data: PDF/UA-1 rule identifiers (no third-party descriptive text)."""