before-you-send 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,22 @@
1
+ """Reads a PDF and reports what is still inside it that you may not mean to send."""
2
+
3
+ # Read from the installed package rather than repeated here. A hand-written copy
4
+ # drifts the moment a release is cut, and then the tool misreports itself. That
5
+ # happened once already on a sibling project: 0.2.0 went out announcing itself as
6
+ # 0.1.0, and no test caught it because the string agreed with itself everywhere it
7
+ # appeared in the source tree. Only installing the built artefact and asking it
8
+ # found the lie.
9
+ try:
10
+ from importlib.metadata import PackageNotFoundError
11
+ from importlib.metadata import version as _installed_version
12
+
13
+ try:
14
+ __version__ = _installed_version("before-you-send")
15
+ except PackageNotFoundError: # running from a source tree, not installed
16
+ __version__ = "unknown (not installed)"
17
+ except ImportError: # pragma: no cover - Python 3.7 and earlier
18
+ __version__ = "unknown"
19
+
20
+ from before_you_send.findings import Blindspot, Finding, Level, Report, Unchecked
21
+
22
+ __all__ = ["Blindspot", "Finding", "Level", "Report", "Unchecked", "__version__"]
@@ -0,0 +1,60 @@
1
+ """Reading a page's annotations without letting one bad entry hide the rest.
2
+
3
+ An /Annots array can contain a reference to an object that is not there, or an entry
4
+ that is not a dictionary at all. Wrapping the whole loop in one try means the first
5
+ broken entry ends the loop, every annotation after it is never looked at, and the
6
+ report says nothing was found. That is the failure mode this module exists to stop:
7
+ the difference between "there were none" and "I could not read them" has to survive
8
+ all the way to the output.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass, field
14
+
15
+
16
+ @dataclass
17
+ class AnnotationScan:
18
+ """The annotations that could be read, and a count of those that could not."""
19
+
20
+ items: list = field(default_factory=list) # (page number, annotation dictionary)
21
+ unreadable: int = 0
22
+ pages_affected: set = field(default_factory=set)
23
+
24
+ def note_gap(self, report, topic: str) -> None:
25
+ """Record unreadable entries so they cannot be mistaken for absent ones."""
26
+ if not self.unreadable:
27
+ return
28
+ pages = ", ".join(str(p) for p in sorted(self.pages_affected))
29
+ report.note_unchecked(
30
+ topic,
31
+ f"{self.unreadable} annotation(s) on page(s) {pages} could not be read, so "
32
+ "anything they contain was not examined",
33
+ )
34
+
35
+
36
+ def scan(pages) -> AnnotationScan:
37
+ """Every readable annotation across the document, with the failures counted."""
38
+ found = AnnotationScan()
39
+ for number, page in enumerate(pages, start=1):
40
+ try:
41
+ annots = page.get("/Annots")
42
+ if annots is None:
43
+ continue
44
+ annots = annots.get_object()
45
+ entries = list(annots)
46
+ except Exception:
47
+ found.unreadable += 1
48
+ found.pages_affected.add(number)
49
+ continue
50
+
51
+ for entry in entries:
52
+ try:
53
+ annot = entry.get_object()
54
+ if not hasattr(annot, "get"):
55
+ raise TypeError("annotation is not a dictionary")
56
+ found.items.append((number, annot))
57
+ except Exception:
58
+ found.unreadable += 1
59
+ found.pages_affected.add(number)
60
+ return found
@@ -0,0 +1,60 @@
1
+ """The checks, in the order a reader should meet them.
2
+
3
+ Two kinds. Page checks look at what was painted on one page and need the drawing
4
+ instructions. Document checks look at the file as a whole and need only the object
5
+ graph. Keeping them apart means a page that cannot be interpreted costs the report
6
+ its page checks and nothing else.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from before_you_send.checks.history import earlier_versions_retained
12
+ from before_you_send.checks.metadata import (
13
+ annotation_authors,
14
+ build_path_in_metadata,
15
+ descriptive_metadata,
16
+ document_author,
17
+ xmp_metadata,
18
+ )
19
+ from before_you_send.checks.payload import active_content, embedded_files, form_field_values
20
+ from before_you_send.checks.protection import (
21
+ encryption_without_a_password,
22
+ hidden_layers,
23
+ unapplied_redaction_marks,
24
+ )
25
+ from before_you_send.checks.visibility import (
26
+ covered_text,
27
+ image_over_text,
28
+ invisible_text,
29
+ text_clipped_away,
30
+ text_matching_background,
31
+ text_outside_page,
32
+ text_too_small_to_read,
33
+ )
34
+
35
+ PAGE_CHECKS = (
36
+ covered_text,
37
+ invisible_text,
38
+ text_matching_background,
39
+ text_clipped_away,
40
+ text_too_small_to_read,
41
+ text_outside_page,
42
+ image_over_text,
43
+ )
44
+
45
+ DOCUMENT_CHECKS = (
46
+ unapplied_redaction_marks,
47
+ embedded_files,
48
+ active_content,
49
+ form_field_values,
50
+ earlier_versions_retained,
51
+ hidden_layers,
52
+ annotation_authors,
53
+ document_author,
54
+ xmp_metadata,
55
+ build_path_in_metadata,
56
+ descriptive_metadata,
57
+ encryption_without_a_password,
58
+ )
59
+
60
+ __all__ = ["PAGE_CHECKS", "DOCUMENT_CHECKS"]
@@ -0,0 +1,157 @@
1
+ """What the file remembers about its own past.
2
+
3
+ A PDF can be edited without rewriting it. The original bytes stay exactly where they
4
+ are and a new section is appended that points past them. Nothing is deleted. A page
5
+ "removed" and saved this way is still in the file, in full, and comes back out with
6
+ ordinary tools.
7
+
8
+ This has to be read from the cross-reference chain rather than from the bytes.
9
+ Counting how many times ``%%EOF`` appears in a file is not a count of its versions:
10
+ an attached document contains its own, and ``/Prev`` is also a key on bookmarks and
11
+ page trees. The question "does this file's cross-reference table point at an older
12
+ one" has an exact answer, and it is the only one worth reporting.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import re
18
+
19
+ from before_you_send.findings import Finding, Level
20
+
21
+ # The /Prev entry of a cross-reference section, whether it sits in a classic trailer
22
+ # dictionary or in the dictionary of a cross-reference stream.
23
+ _PREV = re.compile(rb"/Prev\s+(\d+)")
24
+
25
+ MAX_CHAIN = 64
26
+
27
+
28
+ def _revision_count(doc) -> int:
29
+ """How many cross-reference sections this file chains together.
30
+
31
+ Each hop is one earlier version of the document. The walk starts from the offset
32
+ the file itself points at, so bytes that merely look like a trailer are ignored.
33
+ """
34
+ try:
35
+ last = doc.raw.rsplit(b"startxref", 1)[1]
36
+ offset = int(last.split()[0])
37
+ except Exception:
38
+ return 1
39
+
40
+ seen = set()
41
+ revisions = 1
42
+ while 0 < offset < len(doc.raw) and offset not in seen and revisions < MAX_CHAIN:
43
+ seen.add(offset)
44
+ # A cross-reference section ends at its own %%EOF. Reading past that would
45
+ # run into the next revision and match its /Prev instead of this one's.
46
+ window = doc.raw[offset : offset + 4096]
47
+ end = window.find(b"%%EOF")
48
+ if end != -1:
49
+ window = window[:end]
50
+ match = _PREV.search(window)
51
+ if not match:
52
+ break
53
+ revisions += 1
54
+ try:
55
+ offset = int(match.group(1))
56
+ except Exception:
57
+ break
58
+ return revisions
59
+
60
+
61
+ def _is_linearized(doc) -> bool:
62
+ """Whether this file is laid out for fast web viewing.
63
+
64
+ A linearized file has a second cross-reference section by construction, so one
65
+ hop is expected and says nothing about the document's history.
66
+ """
67
+ return b"/Linearized" in doc.raw[:2048]
68
+
69
+
70
+ def _has_signature(doc) -> bool:
71
+ """True when the document carries a signature, which requires an update to add."""
72
+ try:
73
+ root = doc.reader.trailer["/Root"].get_object()
74
+ form = root.get("/AcroForm")
75
+ if form is None:
76
+ return False
77
+ fields = form.get_object().get("/Fields")
78
+ if fields is None:
79
+ return False
80
+ for field in fields.get_object():
81
+ field = field.get_object()
82
+ if str(field.get("/FT", "")) == "/Sig" and field.get("/V") is not None:
83
+ return True
84
+ except Exception:
85
+ return False
86
+ return False
87
+
88
+
89
+ def earlier_versions_retained(report, doc) -> None:
90
+ """Earlier revisions of the document still present in the file.
91
+
92
+ The count comes from walking the file's own bytes, never from the parsed
93
+ trailer. A parsed trailer only carries ``/Prev`` when the file uses a classic
94
+ cross-reference table; a file written with a cross-reference stream — which is
95
+ what Word, Acrobat, InDesign, Chrome and every linearized government PDF
96
+ produce — keeps ``/Prev`` inside the stream dictionary, where the parser does
97
+ not surface it. Gating on the parsed value meant this check returned on its
98
+ first line for most modern documents and never reached the byte walk written
99
+ for exactly this question. Measured on 887 real published PDFs, that gate hid
100
+ every retained revision in 256 of them.
101
+ """
102
+ revisions = _revision_count(doc)
103
+ if revisions < 2:
104
+ return
105
+
106
+ earlier = revisions - 1
107
+ signed = _has_signature(doc)
108
+ linearized = _is_linearized(doc)
109
+
110
+ # One extra section with an ordinary explanation is not a finding worth alarming
111
+ # anybody about, but it is still worth saying, because what was in that section
112
+ # is readable either way.
113
+ explained = (signed or linearized) and earlier == 1
114
+ reason = "it carries a signature" if signed else "it is laid out for fast web viewing"
115
+
116
+ if explained:
117
+ report.add(
118
+ Finding(
119
+ check="earlier_versions_retained",
120
+ level=Level.LOW,
121
+ page="document",
122
+ location="file structure",
123
+ summary=(
124
+ "The file has one earlier cross-reference section, and "
125
+ f"{reason}, which is the ordinary reason for that."
126
+ ),
127
+ detail=(
128
+ "Signing a PDF, and laying one out for fast web viewing, both append "
129
+ "to the file rather than rewriting it. One extra section is expected "
130
+ "here. Whatever was in it is still readable, so if the document was "
131
+ "edited before that step, the earlier content is in the file."
132
+ ),
133
+ )
134
+ )
135
+ return
136
+
137
+ report.add(
138
+ Finding(
139
+ check="earlier_versions_retained",
140
+ level=Level.MEDIUM if signed or linearized else Level.HIGH,
141
+ page="document",
142
+ location="file structure",
143
+ summary=(
144
+ f"The file contains {earlier} earlier version(s) of itself, kept in "
145
+ "full alongside the current one."
146
+ ),
147
+ detail=(
148
+ "PDFs are commonly edited by appending, which leaves every previous "
149
+ "version intact inside the same file. Text deleted from a page, a page "
150
+ "removed, or a value changed in an earlier draft can still be read out "
151
+ "of the older section. Saving the document again as a new file, rather "
152
+ "than saving over it, is what discards the history."
153
+ + (f" Note that {reason}, which accounts for one of these." if signed or
154
+ linearized else "")
155
+ ),
156
+ )
157
+ )
@@ -0,0 +1,237 @@
1
+ """Who made this, on what machine, and what they called it.
2
+
3
+ None of this is hidden. All of it is two clicks away in any viewer, which is exactly
4
+ why nobody looks at it before sending. A document's properties routinely carry the
5
+ author's account name, the internal filename it was saved under, and the folder it
6
+ was built in.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+
13
+ from before_you_send.annotations import scan
14
+ from before_you_send.findings import Finding, Level
15
+
16
+ DESCRIPTIVE_FIELDS = ("/Title", "/Subject", "/Keywords")
17
+
18
+ # A value that looks like somewhere on somebody's disk rather than a description.
19
+ PATH_LIKE = re.compile(
20
+ r"""(
21
+ [A-Za-z]:\\ # C:\ ...
22
+ | \\\\[^\\]+\\ # \\server\share
23
+ | /(?:Users|home)/[^/\s]+ # /Users/name or /home/name
24
+ | /Volumes/[^/\s]+ # a mounted disk
25
+ )""",
26
+ re.VERBOSE,
27
+ )
28
+
29
+ # A web address with a host in it. Its path is published, not private: a link like
30
+ # "https://www.gov.uk/home/guidance" is not a build path, and calling one "a
31
+ # filesystem path naming a user account" is the kind of wrong that costs the reader's
32
+ # trust in everything else the report says. Addresses are removed before the search.
33
+ #
34
+ # The host has to be non-empty for this to apply, which is exactly what keeps
35
+ # "file:///Users/someone/draft.pdf" — a real leak, and a common one out of a word
36
+ # processor — matching as the path it is.
37
+ WEB_ADDRESS = re.compile(r"\b[a-z][a-z0-9+.\-]*://[^\s/]+\S*", re.IGNORECASE)
38
+
39
+
40
+ def _without_web_addresses(value: str) -> str:
41
+ return WEB_ADDRESS.sub(" ", value)
42
+
43
+
44
+ def _info(doc) -> dict:
45
+ try:
46
+ meta = doc.reader.metadata
47
+ if meta is None:
48
+ return {}
49
+ return {str(k): str(v) for k, v in meta.items() if v is not None}
50
+ except Exception:
51
+ return {}
52
+
53
+
54
+ def document_author(report, doc) -> None:
55
+ """A named author recorded in the document properties."""
56
+ info = _info(doc)
57
+ author = (info.get("/Author") or "").strip()
58
+ if not author:
59
+ return
60
+ report.add(
61
+ Finding(
62
+ check="document_author",
63
+ level=Level.MEDIUM,
64
+ page="document",
65
+ location="document properties",
66
+ summary=f"The document names an author ({len(author)} characters).",
67
+ detail=(
68
+ "This is usually the account name of whoever created the file, and it "
69
+ "travels with the document. It is often not the person the recipient "
70
+ "is expecting, particularly when a document was drafted by one team "
71
+ "and sent by another."
72
+ ),
73
+ sample=author,
74
+ )
75
+ )
76
+
77
+
78
+ def descriptive_metadata(report, doc) -> None:
79
+ """Title, subject or keywords left on the document."""
80
+ info = _info(doc)
81
+ present = [f for f in DESCRIPTIVE_FIELDS if (info.get(f) or "").strip()]
82
+ if not present:
83
+ return
84
+ labels = ", ".join(f.lstrip("/").lower() for f in present)
85
+ report.add(
86
+ Finding(
87
+ check="descriptive_metadata",
88
+ level=Level.LOW,
89
+ page="document",
90
+ location="document properties",
91
+ summary=f"The document carries {labels}.",
92
+ detail=(
93
+ "Exporting from a word processor usually copies the original filename "
94
+ "into the title. That filename is frequently more candid than the "
95
+ "document, because nobody expects it to be read."
96
+ ),
97
+ sample=" | ".join(f"{f.lstrip('/')}: {info[f]}" for f in present),
98
+ )
99
+ )
100
+
101
+
102
+ def build_path_in_metadata(report, doc) -> None:
103
+ """A filesystem path left behind in the document properties."""
104
+ info = _info(doc)
105
+ hits = {
106
+ key: value
107
+ for key, value in info.items()
108
+ if PATH_LIKE.search(_without_web_addresses(value))
109
+ }
110
+ if not hits:
111
+ return
112
+ report.add(
113
+ Finding(
114
+ check="build_path_in_metadata",
115
+ level=Level.MEDIUM,
116
+ page="document",
117
+ location="document properties",
118
+ summary=(
119
+ f"{len(hits)} document property value(s) contain a filesystem path."
120
+ ),
121
+ detail=(
122
+ "A path names a user account, a machine, and often a project or client "
123
+ "folder. It is one of the few things in a document that describes the "
124
+ "sender's organisation rather than the document's subject."
125
+ ),
126
+ sample=" | ".join(f"{k.lstrip('/')}: {v}" for k, v in hits.items()),
127
+ )
128
+ )
129
+
130
+
131
+ def xmp_metadata(report, doc) -> None:
132
+ """A second, separate metadata record that can disagree with the first."""
133
+ try:
134
+ xmp = doc.reader.xmp_metadata
135
+ except Exception:
136
+ xmp = None
137
+ if xmp is None:
138
+ return
139
+
140
+ info = _info(doc)
141
+ info_author = (info.get("/Author") or "").strip()
142
+
143
+ xmp_authors: list = []
144
+ try:
145
+ for value in xmp.dc_creator or []:
146
+ value = str(value).strip()
147
+ if value:
148
+ xmp_authors.append(value)
149
+ except Exception:
150
+ pass
151
+
152
+ disagrees = bool(xmp_authors) and bool(info_author) and info_author not in xmp_authors
153
+
154
+ if disagrees:
155
+ report.add(
156
+ Finding(
157
+ check="xmp_metadata",
158
+ level=Level.MEDIUM,
159
+ page="document",
160
+ location="XMP metadata",
161
+ summary=(
162
+ "The document carries two separate author records and they do not "
163
+ "match."
164
+ ),
165
+ detail=(
166
+ "A PDF stores properties in two places. Editing tools often update "
167
+ "one and leave the other, so the second record can name whoever "
168
+ "held the document earlier. Clearing the visible author does not "
169
+ "clear this one."
170
+ ),
171
+ sample=f"properties: {info_author} | XMP: {', '.join(xmp_authors)}",
172
+ )
173
+ )
174
+ return
175
+
176
+ if xmp_authors:
177
+ report.add(
178
+ Finding(
179
+ check="xmp_metadata",
180
+ level=Level.LOW,
181
+ page="document",
182
+ location="XMP metadata",
183
+ summary="A second metadata record also names an author.",
184
+ detail=(
185
+ "Clearing the author in a viewer's properties dialog does not "
186
+ "always clear this copy, so it is worth checking separately."
187
+ ),
188
+ sample=", ".join(xmp_authors),
189
+ )
190
+ )
191
+
192
+
193
+ def annotation_authors(report, doc) -> None:
194
+ """Comments and markup, and the names attached to them."""
195
+ found = scan(doc.pages)
196
+ found.note_gap(report, "annotation_authors")
197
+
198
+ named: list = []
199
+ with_text = 0
200
+ for number, annot in found.items:
201
+ try:
202
+ subtype = str(annot.get("/Subtype", ""))
203
+ if subtype in ("/Link", "/Widget", "/Popup"):
204
+ continue
205
+ who = annot.get("/T")
206
+ what = annot.get("/Contents")
207
+ if who is not None and str(who).strip():
208
+ named.append((number, str(who).strip()))
209
+ if what is not None and str(what).strip():
210
+ with_text += 1
211
+ except Exception:
212
+ continue
213
+
214
+ if not named and not with_text:
215
+ return
216
+
217
+ people = sorted({who for _, who in named})
218
+ report.add(
219
+ Finding(
220
+ check="annotation_authors",
221
+ level=Level.MEDIUM,
222
+ page="document",
223
+ location="annotations",
224
+ summary=(
225
+ f"{len(named)} annotation(s) name an author"
226
+ + (f" and {with_text} carry comment text" if with_text else "")
227
+ + f", across {len({p for p, _ in named}) or doc.page_count} page(s)."
228
+ ),
229
+ detail=(
230
+ "Comments, highlights and sticky notes stay in the file and carry the "
231
+ "name of whoever made them. Many viewers hide them until the reader "
232
+ "turns comments on, so a document can look finished and still contain "
233
+ "the discussion that produced it."
234
+ ),
235
+ sample=" | ".join(people),
236
+ )
237
+ )