before-you-send 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- before_you_send/__init__.py +22 -0
- before_you_send/annotations.py +60 -0
- before_you_send/checks/__init__.py +60 -0
- before_you_send/checks/history.py +157 -0
- before_you_send/checks/metadata.py +237 -0
- before_you_send/checks/payload.py +218 -0
- before_you_send/checks/protection.py +106 -0
- before_you_send/checks/visibility.py +463 -0
- before_you_send/cli.py +108 -0
- before_you_send/composite.py +165 -0
- before_you_send/content.py +983 -0
- before_you_send/document.py +129 -0
- before_you_send/findings.py +237 -0
- before_you_send/fontmetrics.py +257 -0
- before_you_send/report.py +191 -0
- before_you_send/run.py +112 -0
- before_you_send-0.2.0.dist-info/METADATA +382 -0
- before_you_send-0.2.0.dist-info/RECORD +21 -0
- before_you_send-0.2.0.dist-info/WHEEL +4 -0
- before_you_send-0.2.0.dist-info/entry_points.txt +2 -0
- before_you_send-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Reads a PDF and reports what is still inside it that you may not mean to send."""
|
|
2
|
+
|
|
3
|
+
# Read from the installed package rather than repeated here. A hand-written copy
|
|
4
|
+
# drifts the moment a release is cut, and then the tool misreports itself. That
|
|
5
|
+
# happened once already on a sibling project: 0.2.0 went out announcing itself as
|
|
6
|
+
# 0.1.0, and no test caught it because the string agreed with itself everywhere it
|
|
7
|
+
# appeared in the source tree. Only installing the built artefact and asking it
|
|
8
|
+
# found the lie.
|
|
9
|
+
try:
|
|
10
|
+
from importlib.metadata import PackageNotFoundError
|
|
11
|
+
from importlib.metadata import version as _installed_version
|
|
12
|
+
|
|
13
|
+
try:
|
|
14
|
+
__version__ = _installed_version("before-you-send")
|
|
15
|
+
except PackageNotFoundError: # running from a source tree, not installed
|
|
16
|
+
__version__ = "unknown (not installed)"
|
|
17
|
+
except ImportError: # pragma: no cover - Python 3.7 and earlier
|
|
18
|
+
__version__ = "unknown"
|
|
19
|
+
|
|
20
|
+
from before_you_send.findings import Blindspot, Finding, Level, Report, Unchecked
|
|
21
|
+
|
|
22
|
+
__all__ = ["Blindspot", "Finding", "Level", "Report", "Unchecked", "__version__"]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Reading a page's annotations without letting one bad entry hide the rest.
|
|
2
|
+
|
|
3
|
+
An /Annots array can contain a reference to an object that is not there, or an entry
|
|
4
|
+
that is not a dictionary at all. Wrapping the whole loop in one try means the first
|
|
5
|
+
broken entry ends the loop, every annotation after it is never looked at, and the
|
|
6
|
+
report says nothing was found. That is the failure mode this module exists to stop:
|
|
7
|
+
the difference between "there were none" and "I could not read them" has to survive
|
|
8
|
+
all the way to the output.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class AnnotationScan:
|
|
18
|
+
"""The annotations that could be read, and a count of those that could not."""
|
|
19
|
+
|
|
20
|
+
items: list = field(default_factory=list) # (page number, annotation dictionary)
|
|
21
|
+
unreadable: int = 0
|
|
22
|
+
pages_affected: set = field(default_factory=set)
|
|
23
|
+
|
|
24
|
+
def note_gap(self, report, topic: str) -> None:
|
|
25
|
+
"""Record unreadable entries so they cannot be mistaken for absent ones."""
|
|
26
|
+
if not self.unreadable:
|
|
27
|
+
return
|
|
28
|
+
pages = ", ".join(str(p) for p in sorted(self.pages_affected))
|
|
29
|
+
report.note_unchecked(
|
|
30
|
+
topic,
|
|
31
|
+
f"{self.unreadable} annotation(s) on page(s) {pages} could not be read, so "
|
|
32
|
+
"anything they contain was not examined",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def scan(pages) -> AnnotationScan:
|
|
37
|
+
"""Every readable annotation across the document, with the failures counted."""
|
|
38
|
+
found = AnnotationScan()
|
|
39
|
+
for number, page in enumerate(pages, start=1):
|
|
40
|
+
try:
|
|
41
|
+
annots = page.get("/Annots")
|
|
42
|
+
if annots is None:
|
|
43
|
+
continue
|
|
44
|
+
annots = annots.get_object()
|
|
45
|
+
entries = list(annots)
|
|
46
|
+
except Exception:
|
|
47
|
+
found.unreadable += 1
|
|
48
|
+
found.pages_affected.add(number)
|
|
49
|
+
continue
|
|
50
|
+
|
|
51
|
+
for entry in entries:
|
|
52
|
+
try:
|
|
53
|
+
annot = entry.get_object()
|
|
54
|
+
if not hasattr(annot, "get"):
|
|
55
|
+
raise TypeError("annotation is not a dictionary")
|
|
56
|
+
found.items.append((number, annot))
|
|
57
|
+
except Exception:
|
|
58
|
+
found.unreadable += 1
|
|
59
|
+
found.pages_affected.add(number)
|
|
60
|
+
return found
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""The checks, in the order a reader should meet them.
|
|
2
|
+
|
|
3
|
+
Two kinds. Page checks look at what was painted on one page and need the drawing
|
|
4
|
+
instructions. Document checks look at the file as a whole and need only the object
|
|
5
|
+
graph. Keeping them apart means a page that cannot be interpreted costs the report
|
|
6
|
+
its page checks and nothing else.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from before_you_send.checks.history import earlier_versions_retained
|
|
12
|
+
from before_you_send.checks.metadata import (
|
|
13
|
+
annotation_authors,
|
|
14
|
+
build_path_in_metadata,
|
|
15
|
+
descriptive_metadata,
|
|
16
|
+
document_author,
|
|
17
|
+
xmp_metadata,
|
|
18
|
+
)
|
|
19
|
+
from before_you_send.checks.payload import active_content, embedded_files, form_field_values
|
|
20
|
+
from before_you_send.checks.protection import (
|
|
21
|
+
encryption_without_a_password,
|
|
22
|
+
hidden_layers,
|
|
23
|
+
unapplied_redaction_marks,
|
|
24
|
+
)
|
|
25
|
+
from before_you_send.checks.visibility import (
|
|
26
|
+
covered_text,
|
|
27
|
+
image_over_text,
|
|
28
|
+
invisible_text,
|
|
29
|
+
text_clipped_away,
|
|
30
|
+
text_matching_background,
|
|
31
|
+
text_outside_page,
|
|
32
|
+
text_too_small_to_read,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
PAGE_CHECKS = (
|
|
36
|
+
covered_text,
|
|
37
|
+
invisible_text,
|
|
38
|
+
text_matching_background,
|
|
39
|
+
text_clipped_away,
|
|
40
|
+
text_too_small_to_read,
|
|
41
|
+
text_outside_page,
|
|
42
|
+
image_over_text,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
DOCUMENT_CHECKS = (
|
|
46
|
+
unapplied_redaction_marks,
|
|
47
|
+
embedded_files,
|
|
48
|
+
active_content,
|
|
49
|
+
form_field_values,
|
|
50
|
+
earlier_versions_retained,
|
|
51
|
+
hidden_layers,
|
|
52
|
+
annotation_authors,
|
|
53
|
+
document_author,
|
|
54
|
+
xmp_metadata,
|
|
55
|
+
build_path_in_metadata,
|
|
56
|
+
descriptive_metadata,
|
|
57
|
+
encryption_without_a_password,
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
__all__ = ["PAGE_CHECKS", "DOCUMENT_CHECKS"]
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""What the file remembers about its own past.
|
|
2
|
+
|
|
3
|
+
A PDF can be edited without rewriting it. The original bytes stay exactly where they
|
|
4
|
+
are and a new section is appended that points past them. Nothing is deleted. A page
|
|
5
|
+
"removed" and saved this way is still in the file, in full, and comes back out with
|
|
6
|
+
ordinary tools.
|
|
7
|
+
|
|
8
|
+
This has to be read from the cross-reference chain rather than from the bytes.
|
|
9
|
+
Counting how many times ``%%EOF`` appears in a file is not a count of its versions:
|
|
10
|
+
an attached document contains its own, and ``/Prev`` is also a key on bookmarks and
|
|
11
|
+
page trees. The question "does this file's cross-reference table point at an older
|
|
12
|
+
one" has an exact answer, and it is the only one worth reporting.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
from before_you_send.findings import Finding, Level
|
|
20
|
+
|
|
21
|
+
# The /Prev entry of a cross-reference section, whether it sits in a classic trailer
|
|
22
|
+
# dictionary or in the dictionary of a cross-reference stream.
|
|
23
|
+
_PREV = re.compile(rb"/Prev\s+(\d+)")
|
|
24
|
+
|
|
25
|
+
MAX_CHAIN = 64
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _revision_count(doc) -> int:
|
|
29
|
+
"""How many cross-reference sections this file chains together.
|
|
30
|
+
|
|
31
|
+
Each hop is one earlier version of the document. The walk starts from the offset
|
|
32
|
+
the file itself points at, so bytes that merely look like a trailer are ignored.
|
|
33
|
+
"""
|
|
34
|
+
try:
|
|
35
|
+
last = doc.raw.rsplit(b"startxref", 1)[1]
|
|
36
|
+
offset = int(last.split()[0])
|
|
37
|
+
except Exception:
|
|
38
|
+
return 1
|
|
39
|
+
|
|
40
|
+
seen = set()
|
|
41
|
+
revisions = 1
|
|
42
|
+
while 0 < offset < len(doc.raw) and offset not in seen and revisions < MAX_CHAIN:
|
|
43
|
+
seen.add(offset)
|
|
44
|
+
# A cross-reference section ends at its own %%EOF. Reading past that would
|
|
45
|
+
# run into the next revision and match its /Prev instead of this one's.
|
|
46
|
+
window = doc.raw[offset : offset + 4096]
|
|
47
|
+
end = window.find(b"%%EOF")
|
|
48
|
+
if end != -1:
|
|
49
|
+
window = window[:end]
|
|
50
|
+
match = _PREV.search(window)
|
|
51
|
+
if not match:
|
|
52
|
+
break
|
|
53
|
+
revisions += 1
|
|
54
|
+
try:
|
|
55
|
+
offset = int(match.group(1))
|
|
56
|
+
except Exception:
|
|
57
|
+
break
|
|
58
|
+
return revisions
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _is_linearized(doc) -> bool:
|
|
62
|
+
"""Whether this file is laid out for fast web viewing.
|
|
63
|
+
|
|
64
|
+
A linearized file has a second cross-reference section by construction, so one
|
|
65
|
+
hop is expected and says nothing about the document's history.
|
|
66
|
+
"""
|
|
67
|
+
return b"/Linearized" in doc.raw[:2048]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _has_signature(doc) -> bool:
|
|
71
|
+
"""True when the document carries a signature, which requires an update to add."""
|
|
72
|
+
try:
|
|
73
|
+
root = doc.reader.trailer["/Root"].get_object()
|
|
74
|
+
form = root.get("/AcroForm")
|
|
75
|
+
if form is None:
|
|
76
|
+
return False
|
|
77
|
+
fields = form.get_object().get("/Fields")
|
|
78
|
+
if fields is None:
|
|
79
|
+
return False
|
|
80
|
+
for field in fields.get_object():
|
|
81
|
+
field = field.get_object()
|
|
82
|
+
if str(field.get("/FT", "")) == "/Sig" and field.get("/V") is not None:
|
|
83
|
+
return True
|
|
84
|
+
except Exception:
|
|
85
|
+
return False
|
|
86
|
+
return False
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def earlier_versions_retained(report, doc) -> None:
|
|
90
|
+
"""Earlier revisions of the document still present in the file.
|
|
91
|
+
|
|
92
|
+
The count comes from walking the file's own bytes, never from the parsed
|
|
93
|
+
trailer. A parsed trailer only carries ``/Prev`` when the file uses a classic
|
|
94
|
+
cross-reference table; a file written with a cross-reference stream — which is
|
|
95
|
+
what Word, Acrobat, InDesign, Chrome and every linearized government PDF
|
|
96
|
+
produce — keeps ``/Prev`` inside the stream dictionary, where the parser does
|
|
97
|
+
not surface it. Gating on the parsed value meant this check returned on its
|
|
98
|
+
first line for most modern documents and never reached the byte walk written
|
|
99
|
+
for exactly this question. Measured on 887 real published PDFs, that gate hid
|
|
100
|
+
every retained revision in 256 of them.
|
|
101
|
+
"""
|
|
102
|
+
revisions = _revision_count(doc)
|
|
103
|
+
if revisions < 2:
|
|
104
|
+
return
|
|
105
|
+
|
|
106
|
+
earlier = revisions - 1
|
|
107
|
+
signed = _has_signature(doc)
|
|
108
|
+
linearized = _is_linearized(doc)
|
|
109
|
+
|
|
110
|
+
# One extra section with an ordinary explanation is not a finding worth alarming
|
|
111
|
+
# anybody about, but it is still worth saying, because what was in that section
|
|
112
|
+
# is readable either way.
|
|
113
|
+
explained = (signed or linearized) and earlier == 1
|
|
114
|
+
reason = "it carries a signature" if signed else "it is laid out for fast web viewing"
|
|
115
|
+
|
|
116
|
+
if explained:
|
|
117
|
+
report.add(
|
|
118
|
+
Finding(
|
|
119
|
+
check="earlier_versions_retained",
|
|
120
|
+
level=Level.LOW,
|
|
121
|
+
page="document",
|
|
122
|
+
location="file structure",
|
|
123
|
+
summary=(
|
|
124
|
+
"The file has one earlier cross-reference section, and "
|
|
125
|
+
f"{reason}, which is the ordinary reason for that."
|
|
126
|
+
),
|
|
127
|
+
detail=(
|
|
128
|
+
"Signing a PDF, and laying one out for fast web viewing, both append "
|
|
129
|
+
"to the file rather than rewriting it. One extra section is expected "
|
|
130
|
+
"here. Whatever was in it is still readable, so if the document was "
|
|
131
|
+
"edited before that step, the earlier content is in the file."
|
|
132
|
+
),
|
|
133
|
+
)
|
|
134
|
+
)
|
|
135
|
+
return
|
|
136
|
+
|
|
137
|
+
report.add(
|
|
138
|
+
Finding(
|
|
139
|
+
check="earlier_versions_retained",
|
|
140
|
+
level=Level.MEDIUM if signed or linearized else Level.HIGH,
|
|
141
|
+
page="document",
|
|
142
|
+
location="file structure",
|
|
143
|
+
summary=(
|
|
144
|
+
f"The file contains {earlier} earlier version(s) of itself, kept in "
|
|
145
|
+
"full alongside the current one."
|
|
146
|
+
),
|
|
147
|
+
detail=(
|
|
148
|
+
"PDFs are commonly edited by appending, which leaves every previous "
|
|
149
|
+
"version intact inside the same file. Text deleted from a page, a page "
|
|
150
|
+
"removed, or a value changed in an earlier draft can still be read out "
|
|
151
|
+
"of the older section. Saving the document again as a new file, rather "
|
|
152
|
+
"than saving over it, is what discards the history."
|
|
153
|
+
+ (f" Note that {reason}, which accounts for one of these." if signed or
|
|
154
|
+
linearized else "")
|
|
155
|
+
),
|
|
156
|
+
)
|
|
157
|
+
)
|
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
"""Who made this, on what machine, and what they called it.
|
|
2
|
+
|
|
3
|
+
None of this is hidden. All of it is two clicks away in any viewer, which is exactly
|
|
4
|
+
why nobody looks at it before sending. A document's properties routinely carry the
|
|
5
|
+
author's account name, the internal filename it was saved under, and the folder it
|
|
6
|
+
was built in.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
from before_you_send.annotations import scan
|
|
14
|
+
from before_you_send.findings import Finding, Level
|
|
15
|
+
|
|
16
|
+
DESCRIPTIVE_FIELDS = ("/Title", "/Subject", "/Keywords")
|
|
17
|
+
|
|
18
|
+
# A value that looks like somewhere on somebody's disk rather than a description.
|
|
19
|
+
PATH_LIKE = re.compile(
|
|
20
|
+
r"""(
|
|
21
|
+
[A-Za-z]:\\ # C:\ ...
|
|
22
|
+
| \\\\[^\\]+\\ # \\server\share
|
|
23
|
+
| /(?:Users|home)/[^/\s]+ # /Users/name or /home/name
|
|
24
|
+
| /Volumes/[^/\s]+ # a mounted disk
|
|
25
|
+
)""",
|
|
26
|
+
re.VERBOSE,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
# A web address with a host in it. Its path is published, not private: a link like
|
|
30
|
+
# "https://www.gov.uk/home/guidance" is not a build path, and calling one "a
|
|
31
|
+
# filesystem path naming a user account" is the kind of wrong that costs the reader's
|
|
32
|
+
# trust in everything else the report says. Addresses are removed before the search.
|
|
33
|
+
#
|
|
34
|
+
# The host has to be non-empty for this to apply, which is exactly what keeps
|
|
35
|
+
# "file:///Users/someone/draft.pdf" — a real leak, and a common one out of a word
|
|
36
|
+
# processor — matching as the path it is.
|
|
37
|
+
WEB_ADDRESS = re.compile(r"\b[a-z][a-z0-9+.\-]*://[^\s/]+\S*", re.IGNORECASE)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _without_web_addresses(value: str) -> str:
|
|
41
|
+
return WEB_ADDRESS.sub(" ", value)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _info(doc) -> dict:
|
|
45
|
+
try:
|
|
46
|
+
meta = doc.reader.metadata
|
|
47
|
+
if meta is None:
|
|
48
|
+
return {}
|
|
49
|
+
return {str(k): str(v) for k, v in meta.items() if v is not None}
|
|
50
|
+
except Exception:
|
|
51
|
+
return {}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def document_author(report, doc) -> None:
|
|
55
|
+
"""A named author recorded in the document properties."""
|
|
56
|
+
info = _info(doc)
|
|
57
|
+
author = (info.get("/Author") or "").strip()
|
|
58
|
+
if not author:
|
|
59
|
+
return
|
|
60
|
+
report.add(
|
|
61
|
+
Finding(
|
|
62
|
+
check="document_author",
|
|
63
|
+
level=Level.MEDIUM,
|
|
64
|
+
page="document",
|
|
65
|
+
location="document properties",
|
|
66
|
+
summary=f"The document names an author ({len(author)} characters).",
|
|
67
|
+
detail=(
|
|
68
|
+
"This is usually the account name of whoever created the file, and it "
|
|
69
|
+
"travels with the document. It is often not the person the recipient "
|
|
70
|
+
"is expecting, particularly when a document was drafted by one team "
|
|
71
|
+
"and sent by another."
|
|
72
|
+
),
|
|
73
|
+
sample=author,
|
|
74
|
+
)
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def descriptive_metadata(report, doc) -> None:
|
|
79
|
+
"""Title, subject or keywords left on the document."""
|
|
80
|
+
info = _info(doc)
|
|
81
|
+
present = [f for f in DESCRIPTIVE_FIELDS if (info.get(f) or "").strip()]
|
|
82
|
+
if not present:
|
|
83
|
+
return
|
|
84
|
+
labels = ", ".join(f.lstrip("/").lower() for f in present)
|
|
85
|
+
report.add(
|
|
86
|
+
Finding(
|
|
87
|
+
check="descriptive_metadata",
|
|
88
|
+
level=Level.LOW,
|
|
89
|
+
page="document",
|
|
90
|
+
location="document properties",
|
|
91
|
+
summary=f"The document carries {labels}.",
|
|
92
|
+
detail=(
|
|
93
|
+
"Exporting from a word processor usually copies the original filename "
|
|
94
|
+
"into the title. That filename is frequently more candid than the "
|
|
95
|
+
"document, because nobody expects it to be read."
|
|
96
|
+
),
|
|
97
|
+
sample=" | ".join(f"{f.lstrip('/')}: {info[f]}" for f in present),
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def build_path_in_metadata(report, doc) -> None:
|
|
103
|
+
"""A filesystem path left behind in the document properties."""
|
|
104
|
+
info = _info(doc)
|
|
105
|
+
hits = {
|
|
106
|
+
key: value
|
|
107
|
+
for key, value in info.items()
|
|
108
|
+
if PATH_LIKE.search(_without_web_addresses(value))
|
|
109
|
+
}
|
|
110
|
+
if not hits:
|
|
111
|
+
return
|
|
112
|
+
report.add(
|
|
113
|
+
Finding(
|
|
114
|
+
check="build_path_in_metadata",
|
|
115
|
+
level=Level.MEDIUM,
|
|
116
|
+
page="document",
|
|
117
|
+
location="document properties",
|
|
118
|
+
summary=(
|
|
119
|
+
f"{len(hits)} document property value(s) contain a filesystem path."
|
|
120
|
+
),
|
|
121
|
+
detail=(
|
|
122
|
+
"A path names a user account, a machine, and often a project or client "
|
|
123
|
+
"folder. It is one of the few things in a document that describes the "
|
|
124
|
+
"sender's organisation rather than the document's subject."
|
|
125
|
+
),
|
|
126
|
+
sample=" | ".join(f"{k.lstrip('/')}: {v}" for k, v in hits.items()),
|
|
127
|
+
)
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def xmp_metadata(report, doc) -> None:
|
|
132
|
+
"""A second, separate metadata record that can disagree with the first."""
|
|
133
|
+
try:
|
|
134
|
+
xmp = doc.reader.xmp_metadata
|
|
135
|
+
except Exception:
|
|
136
|
+
xmp = None
|
|
137
|
+
if xmp is None:
|
|
138
|
+
return
|
|
139
|
+
|
|
140
|
+
info = _info(doc)
|
|
141
|
+
info_author = (info.get("/Author") or "").strip()
|
|
142
|
+
|
|
143
|
+
xmp_authors: list = []
|
|
144
|
+
try:
|
|
145
|
+
for value in xmp.dc_creator or []:
|
|
146
|
+
value = str(value).strip()
|
|
147
|
+
if value:
|
|
148
|
+
xmp_authors.append(value)
|
|
149
|
+
except Exception:
|
|
150
|
+
pass
|
|
151
|
+
|
|
152
|
+
disagrees = bool(xmp_authors) and bool(info_author) and info_author not in xmp_authors
|
|
153
|
+
|
|
154
|
+
if disagrees:
|
|
155
|
+
report.add(
|
|
156
|
+
Finding(
|
|
157
|
+
check="xmp_metadata",
|
|
158
|
+
level=Level.MEDIUM,
|
|
159
|
+
page="document",
|
|
160
|
+
location="XMP metadata",
|
|
161
|
+
summary=(
|
|
162
|
+
"The document carries two separate author records and they do not "
|
|
163
|
+
"match."
|
|
164
|
+
),
|
|
165
|
+
detail=(
|
|
166
|
+
"A PDF stores properties in two places. Editing tools often update "
|
|
167
|
+
"one and leave the other, so the second record can name whoever "
|
|
168
|
+
"held the document earlier. Clearing the visible author does not "
|
|
169
|
+
"clear this one."
|
|
170
|
+
),
|
|
171
|
+
sample=f"properties: {info_author} | XMP: {', '.join(xmp_authors)}",
|
|
172
|
+
)
|
|
173
|
+
)
|
|
174
|
+
return
|
|
175
|
+
|
|
176
|
+
if xmp_authors:
|
|
177
|
+
report.add(
|
|
178
|
+
Finding(
|
|
179
|
+
check="xmp_metadata",
|
|
180
|
+
level=Level.LOW,
|
|
181
|
+
page="document",
|
|
182
|
+
location="XMP metadata",
|
|
183
|
+
summary="A second metadata record also names an author.",
|
|
184
|
+
detail=(
|
|
185
|
+
"Clearing the author in a viewer's properties dialog does not "
|
|
186
|
+
"always clear this copy, so it is worth checking separately."
|
|
187
|
+
),
|
|
188
|
+
sample=", ".join(xmp_authors),
|
|
189
|
+
)
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def annotation_authors(report, doc) -> None:
|
|
194
|
+
"""Comments and markup, and the names attached to them."""
|
|
195
|
+
found = scan(doc.pages)
|
|
196
|
+
found.note_gap(report, "annotation_authors")
|
|
197
|
+
|
|
198
|
+
named: list = []
|
|
199
|
+
with_text = 0
|
|
200
|
+
for number, annot in found.items:
|
|
201
|
+
try:
|
|
202
|
+
subtype = str(annot.get("/Subtype", ""))
|
|
203
|
+
if subtype in ("/Link", "/Widget", "/Popup"):
|
|
204
|
+
continue
|
|
205
|
+
who = annot.get("/T")
|
|
206
|
+
what = annot.get("/Contents")
|
|
207
|
+
if who is not None and str(who).strip():
|
|
208
|
+
named.append((number, str(who).strip()))
|
|
209
|
+
if what is not None and str(what).strip():
|
|
210
|
+
with_text += 1
|
|
211
|
+
except Exception:
|
|
212
|
+
continue
|
|
213
|
+
|
|
214
|
+
if not named and not with_text:
|
|
215
|
+
return
|
|
216
|
+
|
|
217
|
+
people = sorted({who for _, who in named})
|
|
218
|
+
report.add(
|
|
219
|
+
Finding(
|
|
220
|
+
check="annotation_authors",
|
|
221
|
+
level=Level.MEDIUM,
|
|
222
|
+
page="document",
|
|
223
|
+
location="annotations",
|
|
224
|
+
summary=(
|
|
225
|
+
f"{len(named)} annotation(s) name an author"
|
|
226
|
+
+ (f" and {with_text} carry comment text" if with_text else "")
|
|
227
|
+
+ f", across {len({p for p, _ in named}) or doc.page_count} page(s)."
|
|
228
|
+
),
|
|
229
|
+
detail=(
|
|
230
|
+
"Comments, highlights and sticky notes stay in the file and carry the "
|
|
231
|
+
"name of whoever made them. Many viewers hide them until the reader "
|
|
232
|
+
"turns comments on, so a document can look finished and still contain "
|
|
233
|
+
"the discussion that produced it."
|
|
234
|
+
),
|
|
235
|
+
sample=" | ".join(people),
|
|
236
|
+
)
|
|
237
|
+
)
|