codendium 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codendium-1.0.0.dist-info/METADATA +332 -0
- codendium-1.0.0.dist-info/RECORD +45 -0
- codendium-1.0.0.dist-info/WHEEL +5 -0
- codendium-1.0.0.dist-info/entry_points.txt +6 -0
- codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
- codendium-1.0.0.dist-info/top_level.txt +1 -0
- copyright_deposit/__init__.py +15 -0
- copyright_deposit/__main__.py +34 -0
- copyright_deposit/assets/__init__.py +5 -0
- copyright_deposit/assets/fonts/README.md +31 -0
- copyright_deposit/assets/logo.svg +26 -0
- copyright_deposit/cli.py +314 -0
- copyright_deposit/config.py +269 -0
- copyright_deposit/core/__init__.py +1 -0
- copyright_deposit/core/deposit.py +117 -0
- copyright_deposit/core/discovery.py +260 -0
- copyright_deposit/core/encoding.py +136 -0
- copyright_deposit/core/languages.py +190 -0
- copyright_deposit/core/layout.py +349 -0
- copyright_deposit/core/lineranges.py +219 -0
- copyright_deposit/core/manifest.py +282 -0
- copyright_deposit/core/metrics.py +279 -0
- copyright_deposit/core/ordering.py +349 -0
- copyright_deposit/core/pipeline.py +386 -0
- copyright_deposit/core/redaction.py +162 -0
- copyright_deposit/core/render.py +242 -0
- copyright_deposit/core/scanning/__init__.py +61 -0
- copyright_deposit/core/scanning/secrets.py +181 -0
- copyright_deposit/core/scanning/thirdparty.py +190 -0
- copyright_deposit/core/strip/__init__.py +337 -0
- copyright_deposit/core/strip/cfamily_strip.py +235 -0
- copyright_deposit/core/strip/pygments_strip.py +85 -0
- copyright_deposit/core/strip/python_strip.py +131 -0
- copyright_deposit/gui/__init__.py +1 -0
- copyright_deposit/gui/app.py +34 -0
- copyright_deposit/gui/branding.py +83 -0
- copyright_deposit/gui/history.py +192 -0
- copyright_deposit/gui/main_window.py +617 -0
- copyright_deposit/gui/panels/__init__.py +1 -0
- copyright_deposit/gui/panels/estimate.py +166 -0
- copyright_deposit/gui/panels/files.py +635 -0
- copyright_deposit/gui/panels/identification.py +193 -0
- copyright_deposit/gui/panels/options.py +445 -0
- copyright_deposit/gui/panels/preflight.py +260 -0
- copyright_deposit/gui/workers.py +96 -0
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""PDF rendering.
|
|
2
|
+
|
|
3
|
+
Two properties are load-bearing:
|
|
4
|
+
|
|
5
|
+
* **Deterministic.** The canvas runs in reportlab's invariant mode with a
|
|
6
|
+
fixed document date, so the same settings and the same sources produce a
|
|
7
|
+
byte-identical PDF. A re-run therefore proves what was filed.
|
|
8
|
+
* **Redaction is real.** Blocked-out text is never written to the content
|
|
9
|
+
stream at all - only a filled rectangle is drawn - so it cannot be
|
|
10
|
+
recovered with copy/paste or a text extractor.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from .. import DISPLAY_NAME, __version__
|
|
19
|
+
from ..config import HeaderInfo, LayoutOptions
|
|
20
|
+
from .deposit import MODE_HEAD_TAIL, DepositSelection
|
|
21
|
+
from .layout import (
|
|
22
|
+
KIND_ELISION,
|
|
23
|
+
KIND_HEADER,
|
|
24
|
+
KIND_NOTICE,
|
|
25
|
+
LayoutResult,
|
|
26
|
+
RenderLine,
|
|
27
|
+
)
|
|
28
|
+
from .metrics import PageGeometry
|
|
29
|
+
|
|
30
|
+
GUTTER_GRAY = 0.45
|
|
31
|
+
HEADER_GRAY = 0.35
|
|
32
|
+
ELISION_GRAY = 0.40
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class RenderReport:
|
|
37
|
+
output_path: str
|
|
38
|
+
page_count: int
|
|
39
|
+
font: str
|
|
40
|
+
embedded: bool
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _canvas(path: str, geometry: PageGeometry, header: HeaderInfo, fingerprint: str):
|
|
44
|
+
from reportlab.pdfgen import canvas as rl_canvas
|
|
45
|
+
|
|
46
|
+
c = rl_canvas.Canvas(
|
|
47
|
+
path,
|
|
48
|
+
pagesize=(geometry.width, geometry.height),
|
|
49
|
+
invariant=1, # stable document id and creation date
|
|
50
|
+
pageCompression=1,
|
|
51
|
+
)
|
|
52
|
+
c.setTitle(header.title_line())
|
|
53
|
+
c.setAuthor(header.copyright_owner or "")
|
|
54
|
+
c.setSubject(header.deposit_label or "Source code deposit")
|
|
55
|
+
c.setCreator(f"{DISPLAY_NAME} {__version__}")
|
|
56
|
+
c.setKeywords(f"settings-fingerprint:{fingerprint[:16]}")
|
|
57
|
+
return c
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _draw_text_with_redactions(c, geometry: PageGeometry, line: RenderLine, x0: float, y: float) -> None:
|
|
61
|
+
font = geometry.font
|
|
62
|
+
if not line.redactions:
|
|
63
|
+
c.drawString(x0, y, line.text)
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
char_w = font.char_width
|
|
67
|
+
bar_bottom = y - font.size * 0.22
|
|
68
|
+
bar_height = font.size * 1.02
|
|
69
|
+
|
|
70
|
+
position = 0
|
|
71
|
+
for start, end in line.redactions:
|
|
72
|
+
start = max(0, min(start, len(line.text)))
|
|
73
|
+
end = max(start, min(end, len(line.text)))
|
|
74
|
+
if start > position:
|
|
75
|
+
c.drawString(x0 + position * char_w, y, line.text[position:start])
|
|
76
|
+
c.saveState()
|
|
77
|
+
c.setFillGray(0.0)
|
|
78
|
+
c.rect(x0 + start * char_w, bar_bottom, (end - start) * char_w, bar_height, fill=1, stroke=0)
|
|
79
|
+
c.restoreState()
|
|
80
|
+
position = end
|
|
81
|
+
if position < len(line.text):
|
|
82
|
+
c.drawString(x0 + position * char_w, y, line.text[position:])
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _draw_page(
|
|
86
|
+
c,
|
|
87
|
+
geometry: PageGeometry,
|
|
88
|
+
lines: list[RenderLine],
|
|
89
|
+
*,
|
|
90
|
+
printed_number: int,
|
|
91
|
+
total_label: int,
|
|
92
|
+
title: str,
|
|
93
|
+
sheet_label: str = "",
|
|
94
|
+
show_number: bool = True,
|
|
95
|
+
) -> None:
|
|
96
|
+
font = geometry.font
|
|
97
|
+
code_x = geometry.column_x(geometry.gutter_columns)
|
|
98
|
+
|
|
99
|
+
if geometry.running_header:
|
|
100
|
+
c.saveState()
|
|
101
|
+
c.setFont(font.regular, max(6.5, font.size * 0.72))
|
|
102
|
+
c.setFillGray(HEADER_GRAY)
|
|
103
|
+
header_y = geometry.height - geometry.margin - font.size * 0.9
|
|
104
|
+
c.drawString(geometry.margin, header_y, title[:70])
|
|
105
|
+
current_file = next((ln.file_path for ln in lines if ln.file_path), "")
|
|
106
|
+
if current_file:
|
|
107
|
+
c.drawRightString(geometry.width - geometry.margin, header_y, current_file[-70:])
|
|
108
|
+
c.restoreState()
|
|
109
|
+
|
|
110
|
+
for row, line in enumerate(lines):
|
|
111
|
+
y = geometry.baseline(row)
|
|
112
|
+
if not line.text and not line.redactions:
|
|
113
|
+
continue
|
|
114
|
+
|
|
115
|
+
if line.number is not None and geometry.gutter_columns:
|
|
116
|
+
c.saveState()
|
|
117
|
+
c.setFont(font.regular, font.size)
|
|
118
|
+
c.setFillGray(GUTTER_GRAY)
|
|
119
|
+
c.drawRightString(
|
|
120
|
+
geometry.column_x(geometry.gutter_columns - 1),
|
|
121
|
+
y,
|
|
122
|
+
str(line.number),
|
|
123
|
+
)
|
|
124
|
+
c.restoreState()
|
|
125
|
+
|
|
126
|
+
c.setFont(font.bold if line.bold else font.regular, font.size)
|
|
127
|
+
# Elision rules are not source text, so they are set in grey: a
|
|
128
|
+
# reader can tell at a glance that nothing was written there.
|
|
129
|
+
c.setFillGray(ELISION_GRAY if line.kind == KIND_ELISION else 0.0)
|
|
130
|
+
x0 = geometry.margin if line.kind in (KIND_HEADER, KIND_NOTICE) else code_x
|
|
131
|
+
_draw_text_with_redactions(c, geometry, line, x0, y)
|
|
132
|
+
|
|
133
|
+
if geometry.page_numbers:
|
|
134
|
+
c.saveState()
|
|
135
|
+
c.setFont(font.regular, max(6.5, font.size * 0.78))
|
|
136
|
+
c.setFillGray(HEADER_GRAY)
|
|
137
|
+
footer_y = geometry.margin * 0.55
|
|
138
|
+
if show_number:
|
|
139
|
+
c.drawCentredString(
|
|
140
|
+
geometry.width / 2.0,
|
|
141
|
+
footer_y,
|
|
142
|
+
f"Page {printed_number} of {total_label}",
|
|
143
|
+
)
|
|
144
|
+
if sheet_label:
|
|
145
|
+
c.drawRightString(geometry.width - geometry.margin, footer_y, sheet_label)
|
|
146
|
+
c.restoreState()
|
|
147
|
+
|
|
148
|
+
c.showPage()
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _separator_lines(text: str, geometry: PageGeometry) -> list[RenderLine]:
|
|
152
|
+
"""Center the omission notice on its own sheet."""
|
|
153
|
+
from .layout import wrap_source_line
|
|
154
|
+
|
|
155
|
+
width = min(geometry.code_columns, 72)
|
|
156
|
+
chunks = [body for _s, _p, body in wrap_source_line(text, width, "")]
|
|
157
|
+
top_pad = max(0, (geometry.lines_per_page - len(chunks)) // 2)
|
|
158
|
+
lines = [RenderLine() for _ in range(top_pad)]
|
|
159
|
+
lines.extend(RenderLine(text=chunk, kind=KIND_NOTICE, bold=True) for chunk in chunks)
|
|
160
|
+
return lines
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def render_pdf(
|
|
164
|
+
layout: LayoutResult,
|
|
165
|
+
output_path: str | Path,
|
|
166
|
+
header: HeaderInfo,
|
|
167
|
+
options: LayoutOptions,
|
|
168
|
+
*,
|
|
169
|
+
selection: DepositSelection | None = None,
|
|
170
|
+
fingerprint: str = "",
|
|
171
|
+
include_separator: bool = True,
|
|
172
|
+
) -> RenderReport:
|
|
173
|
+
"""Write `layout` to a PDF, optionally restricted to a deposit selection."""
|
|
174
|
+
geometry = layout.geometry
|
|
175
|
+
if geometry is None:
|
|
176
|
+
raise ValueError("layout has no geometry; build it with build_layout()")
|
|
177
|
+
|
|
178
|
+
output_path = str(output_path)
|
|
179
|
+
Path(output_path).parent.mkdir(parents=True, exist_ok=True)
|
|
180
|
+
|
|
181
|
+
pages = layout.pages
|
|
182
|
+
total_label = len(pages)
|
|
183
|
+
title = header.title_line()
|
|
184
|
+
|
|
185
|
+
if selection is None:
|
|
186
|
+
chosen = list(range(1, len(pages) + 1))
|
|
187
|
+
separator_after = None
|
|
188
|
+
sheets = len(chosen)
|
|
189
|
+
else:
|
|
190
|
+
chosen = list(selection.page_numbers)
|
|
191
|
+
total_label = selection.total_pages
|
|
192
|
+
sheets = len(chosen)
|
|
193
|
+
separator_after = None
|
|
194
|
+
if selection.mode == MODE_HEAD_TAIL and include_separator:
|
|
195
|
+
separator_after = min(len(chosen), max(0, selection_head_count(selection)))
|
|
196
|
+
|
|
197
|
+
c = _canvas(output_path, geometry, header, fingerprint)
|
|
198
|
+
|
|
199
|
+
emitted = 0
|
|
200
|
+
for position, page_number in enumerate(chosen):
|
|
201
|
+
page = pages[page_number - 1]
|
|
202
|
+
emitted += 1
|
|
203
|
+
_draw_page(
|
|
204
|
+
c,
|
|
205
|
+
geometry,
|
|
206
|
+
page.lines,
|
|
207
|
+
printed_number=page_number,
|
|
208
|
+
total_label=total_label,
|
|
209
|
+
title=title,
|
|
210
|
+
sheet_label=(
|
|
211
|
+
f"Deposit sheet {emitted} of {sheets}" if selection and selection.mode == MODE_HEAD_TAIL else ""
|
|
212
|
+
),
|
|
213
|
+
)
|
|
214
|
+
if separator_after is not None and position + 1 == separator_after:
|
|
215
|
+
notice = selection.separator_notice() if selection else ""
|
|
216
|
+
if notice:
|
|
217
|
+
_draw_page(
|
|
218
|
+
c,
|
|
219
|
+
geometry,
|
|
220
|
+
_separator_lines(notice, geometry),
|
|
221
|
+
printed_number=page_number,
|
|
222
|
+
total_label=total_label,
|
|
223
|
+
title=title,
|
|
224
|
+
sheet_label="omission notice",
|
|
225
|
+
show_number=False,
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
c.save()
|
|
229
|
+
return RenderReport(
|
|
230
|
+
output_path=output_path,
|
|
231
|
+
page_count=emitted,
|
|
232
|
+
font=geometry.font.regular,
|
|
233
|
+
embedded=geometry.font.embedded,
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def selection_head_count(selection: DepositSelection) -> int:
|
|
238
|
+
"""How many leading pages precede the omission gap."""
|
|
239
|
+
if selection.omitted is None:
|
|
240
|
+
return len(selection.page_numbers)
|
|
241
|
+
gap_start = selection.omitted[0]
|
|
242
|
+
return sum(1 for n in selection.page_numbers if n < gap_start)
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Pre-flight scanners.
|
|
2
|
+
|
|
3
|
+
A deposit becomes part of a public record that anyone may inspect, and a
|
|
4
|
+
registration may only claim material the applicant actually owns. These
|
|
5
|
+
scanners exist to catch both mistakes before the PDF is filed.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
SEVERITY_HIGH = "high"
|
|
13
|
+
SEVERITY_MEDIUM = "medium"
|
|
14
|
+
SEVERITY_LOW = "low"
|
|
15
|
+
|
|
16
|
+
_SEVERITY_ORDER = {SEVERITY_HIGH: 0, SEVERITY_MEDIUM: 1, SEVERITY_LOW: 2}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class Finding:
|
|
21
|
+
id: str
|
|
22
|
+
rel_path: str
|
|
23
|
+
line_number: int
|
|
24
|
+
rule: str
|
|
25
|
+
detail: str
|
|
26
|
+
severity: str = SEVERITY_MEDIUM
|
|
27
|
+
excerpt: str = ""
|
|
28
|
+
category: str = "secret"
|
|
29
|
+
|
|
30
|
+
def location(self) -> str:
|
|
31
|
+
return f"{self.rel_path}:{self.line_number}"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class ScanReport:
|
|
36
|
+
secrets: list[Finding] = field(default_factory=list)
|
|
37
|
+
third_party: list[Finding] = field(default_factory=list)
|
|
38
|
+
warnings: list[str] = field(default_factory=list)
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def all_findings(self) -> list[Finding]:
|
|
42
|
+
return self.secrets + self.third_party
|
|
43
|
+
|
|
44
|
+
def blocking(self, ignored: set[str]) -> list[Finding]:
|
|
45
|
+
return [
|
|
46
|
+
f
|
|
47
|
+
for f in self.secrets
|
|
48
|
+
if f.severity == SEVERITY_HIGH and f.id not in ignored
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
def sorted_secrets(self) -> list[Finding]:
|
|
52
|
+
return sorted(
|
|
53
|
+
self.secrets,
|
|
54
|
+
key=lambda f: (_SEVERITY_ORDER.get(f.severity, 9), f.rel_path, f.line_number),
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
def sorted_third_party(self) -> list[Finding]:
|
|
58
|
+
return sorted(
|
|
59
|
+
self.third_party,
|
|
60
|
+
key=lambda f: (_SEVERITY_ORDER.get(f.severity, 9), f.rel_path, f.line_number),
|
|
61
|
+
)
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""Credential and PII detection.
|
|
2
|
+
|
|
3
|
+
Anything deposited can be inspected by the public, so a hardcoded key in
|
|
4
|
+
the source is not just a security bug - it is a disclosure. Findings are
|
|
5
|
+
reported with a masked excerpt, and each carries a stable id so the
|
|
6
|
+
operator's decision to ignore or redact it survives into the next run.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import math
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
from . import SEVERITY_HIGH, SEVERITY_LOW, SEVERITY_MEDIUM, Finding
|
|
16
|
+
|
|
17
|
+
# (rule name, pattern, severity, human explanation)
|
|
18
|
+
_RULES: tuple[tuple[str, re.Pattern[str], str, str], ...] = (
|
|
19
|
+
(
|
|
20
|
+
"aws-access-key",
|
|
21
|
+
re.compile(r"\b((?:AKIA|ASIA|ABIA|ACCA)[0-9A-Z]{16})\b"),
|
|
22
|
+
SEVERITY_HIGH,
|
|
23
|
+
"AWS access key id",
|
|
24
|
+
),
|
|
25
|
+
(
|
|
26
|
+
"private-key",
|
|
27
|
+
re.compile(r"-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----"),
|
|
28
|
+
SEVERITY_HIGH,
|
|
29
|
+
"embedded private key block",
|
|
30
|
+
),
|
|
31
|
+
(
|
|
32
|
+
"github-token",
|
|
33
|
+
re.compile(r"\b(gh[pousr]_[A-Za-z0-9]{16,})\b"),
|
|
34
|
+
SEVERITY_HIGH,
|
|
35
|
+
"GitHub token",
|
|
36
|
+
),
|
|
37
|
+
(
|
|
38
|
+
"slack-token",
|
|
39
|
+
re.compile(r"\b(xox[baprs]-[A-Za-z0-9-]{10,})\b"),
|
|
40
|
+
SEVERITY_HIGH,
|
|
41
|
+
"Slack token",
|
|
42
|
+
),
|
|
43
|
+
(
|
|
44
|
+
"google-api-key",
|
|
45
|
+
re.compile(r"\b(AIza[0-9A-Za-z_\-]{35})\b"),
|
|
46
|
+
SEVERITY_HIGH,
|
|
47
|
+
"Google API key",
|
|
48
|
+
),
|
|
49
|
+
(
|
|
50
|
+
"stripe-key",
|
|
51
|
+
re.compile(r"\b((?:sk|rk)_(?:live|test)_[0-9A-Za-z]{16,})\b"),
|
|
52
|
+
SEVERITY_HIGH,
|
|
53
|
+
"Stripe secret key",
|
|
54
|
+
),
|
|
55
|
+
(
|
|
56
|
+
"openai-key",
|
|
57
|
+
re.compile(r"\b(sk-[A-Za-z0-9_\-]{20,})\b"),
|
|
58
|
+
SEVERITY_HIGH,
|
|
59
|
+
"OpenAI-style API key",
|
|
60
|
+
),
|
|
61
|
+
(
|
|
62
|
+
"jwt",
|
|
63
|
+
re.compile(r"\beyJ[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}\b"),
|
|
64
|
+
SEVERITY_MEDIUM,
|
|
65
|
+
"JSON web token",
|
|
66
|
+
),
|
|
67
|
+
(
|
|
68
|
+
"connection-string",
|
|
69
|
+
re.compile(r"\b[a-z][a-z0-9+.\-]*://[^\s:/@]+:[^\s:/@]+@[^\s/]+", re.IGNORECASE),
|
|
70
|
+
SEVERITY_HIGH,
|
|
71
|
+
"URL containing credentials",
|
|
72
|
+
),
|
|
73
|
+
(
|
|
74
|
+
"email",
|
|
75
|
+
re.compile(r"\b[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}\b"),
|
|
76
|
+
SEVERITY_LOW,
|
|
77
|
+
"email address",
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
# name = "value" style assignments worth entropy-checking
|
|
82
|
+
# The prefix is optional so that a name which *is* the keyword - `api_key`,
|
|
83
|
+
# `token` - matches as readily as `service_api_key`.
|
|
84
|
+
_ASSIGNMENT = re.compile(
|
|
85
|
+
r"""(?ix)
|
|
86
|
+
\b(?P<name>[A-Za-z0-9_]*
|
|
87
|
+
(?:pass(?:wd|word)?|secret|token|api[_-]?key|apikey|auth|credential|private[_-]?key)
|
|
88
|
+
[A-Za-z0-9_]*)
|
|
89
|
+
\s*[:=]\s*
|
|
90
|
+
(?P<quote>["'])(?P<value>[^"']{8,})(?P=quote)
|
|
91
|
+
"""
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
# Values that are obviously not real credentials. The tails accept hyphens
|
|
95
|
+
# and dots so that "your-api-key-here" is recognised as a placeholder.
|
|
96
|
+
_PLACEHOLDER = re.compile(
|
|
97
|
+
r"(?i)^(?:\s*|x{3,}|\.{3,}|none|null|true|false|changeme|placeholder|redacted|"
|
|
98
|
+
r"(?:your|my|the|some|example|dummy|test|sample|fake|todo|insert|enter|put)"
|
|
99
|
+
r"[\w\-. ]*|"
|
|
100
|
+
r"\$\{.*\}|\{\{.*\}\}|<.*>|%\w+%|\*+)$"
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
_NON_SECRET_EXT = (".md", ".rst", ".txt")
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _shannon_entropy(value: str) -> float:
|
|
107
|
+
if not value:
|
|
108
|
+
return 0.0
|
|
109
|
+
counts: dict[str, int] = {}
|
|
110
|
+
for ch in value:
|
|
111
|
+
counts[ch] = counts.get(ch, 0) + 1
|
|
112
|
+
length = len(value)
|
|
113
|
+
return -sum((c / length) * math.log2(c / length) for c in counts.values())
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _mask(value: str) -> str:
|
|
117
|
+
if len(value) <= 8:
|
|
118
|
+
return value[0] + "*" * (len(value) - 1) if value else ""
|
|
119
|
+
return f"{value[:4]}{'*' * 8}{value[-2:]}"
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _finding_id(rel_path: str, line_number: int, rule: str, value: str) -> str:
|
|
123
|
+
digest = hashlib.sha256(f"{rel_path}|{rule}|{value}".encode("utf-8")).hexdigest()
|
|
124
|
+
return f"{rule}-{digest[:12]}"
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def scan_text(rel_path: str, text: str) -> list[Finding]:
|
|
128
|
+
"""Find credentials and PII in one file's original text."""
|
|
129
|
+
findings: list[Finding] = []
|
|
130
|
+
seen: set[tuple[int, str]] = set()
|
|
131
|
+
allow_low = not rel_path.lower().endswith(_NON_SECRET_EXT)
|
|
132
|
+
|
|
133
|
+
for number, line in enumerate(text.split("\n"), start=1):
|
|
134
|
+
if len(line) > 4000:
|
|
135
|
+
line = line[:4000]
|
|
136
|
+
|
|
137
|
+
for rule, pattern, severity, detail in _RULES:
|
|
138
|
+
if severity == SEVERITY_LOW and not allow_low:
|
|
139
|
+
continue
|
|
140
|
+
for match in pattern.finditer(line):
|
|
141
|
+
value = match.group(1) if match.groups() else match.group(0)
|
|
142
|
+
if _PLACEHOLDER.match(value):
|
|
143
|
+
continue
|
|
144
|
+
key = (number, rule)
|
|
145
|
+
if key in seen:
|
|
146
|
+
continue
|
|
147
|
+
seen.add(key)
|
|
148
|
+
findings.append(
|
|
149
|
+
Finding(
|
|
150
|
+
id=_finding_id(rel_path, number, rule, value),
|
|
151
|
+
rel_path=rel_path,
|
|
152
|
+
line_number=number,
|
|
153
|
+
rule=rule,
|
|
154
|
+
detail=detail,
|
|
155
|
+
severity=severity,
|
|
156
|
+
excerpt=_mask(value),
|
|
157
|
+
category="secret",
|
|
158
|
+
)
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
match = _ASSIGNMENT.search(line)
|
|
162
|
+
if match:
|
|
163
|
+
value = match.group("value")
|
|
164
|
+
name = match.group("name")
|
|
165
|
+
if not _PLACEHOLDER.match(value) and _shannon_entropy(value) >= 3.0:
|
|
166
|
+
key = (number, "hardcoded-credential")
|
|
167
|
+
if key not in seen:
|
|
168
|
+
seen.add(key)
|
|
169
|
+
findings.append(
|
|
170
|
+
Finding(
|
|
171
|
+
id=_finding_id(rel_path, number, "hardcoded-credential", value),
|
|
172
|
+
rel_path=rel_path,
|
|
173
|
+
line_number=number,
|
|
174
|
+
rule="hardcoded-credential",
|
|
175
|
+
detail=f"high-entropy value assigned to '{name}'",
|
|
176
|
+
severity=SEVERITY_HIGH,
|
|
177
|
+
excerpt=_mask(value),
|
|
178
|
+
category="secret",
|
|
179
|
+
)
|
|
180
|
+
)
|
|
181
|
+
return findings
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Detection of code the applicant may not own.
|
|
2
|
+
|
|
3
|
+
A registration covers the applicant's own authorship. Vendored libraries,
|
|
4
|
+
generated files and headers naming a different copyright holder all need a
|
|
5
|
+
decision - exclude them, or disclaim them in the application - before the
|
|
6
|
+
deposit is filed.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
from . import SEVERITY_HIGH, SEVERITY_LOW, SEVERITY_MEDIUM, Finding
|
|
15
|
+
|
|
16
|
+
HEADER_LINES = 60
|
|
17
|
+
|
|
18
|
+
# A real notice carries a (c)/copyright symbol or a year. Without that
|
|
19
|
+
# requirement the word "copyright" in ordinary prose - or in a variable
|
|
20
|
+
# name - produces a flood of false positives.
|
|
21
|
+
_COPYRIGHT = re.compile(
|
|
22
|
+
r"(?i)\bcopyright\b\s*(?P<sym>\(c\)|©|©)?\s*"
|
|
23
|
+
r"(?P<years>\d{4}(?:\s*[-,]\s*\d{4})*)?\s*"
|
|
24
|
+
r"(?:by\s+)?(?P<holder>[^\n<]{2,80})?"
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
# A plausible human or organisation name, not a fragment of source code.
|
|
28
|
+
_HOLDER_OK = re.compile(r"^[A-Za-z][A-Za-z0-9 .,&'’à-ÿ-]{1,60}$")
|
|
29
|
+
_SPDX = re.compile(r"SPDX-License-Identifier:\s*(?P<id>[A-Za-z0-9.\-+ ()]+)")
|
|
30
|
+
|
|
31
|
+
_LICENSE_PHRASES: tuple[tuple[str, str], ...] = (
|
|
32
|
+
("apache license", "Apache License"),
|
|
33
|
+
("mit license", "MIT License"),
|
|
34
|
+
("bsd license", "BSD License"),
|
|
35
|
+
("redistribution and use in source and binary forms", "BSD-style licence text"),
|
|
36
|
+
("gnu general public license", "GNU GPL"),
|
|
37
|
+
("gnu lesser general public license", "GNU LGPL"),
|
|
38
|
+
("mozilla public license", "Mozilla Public License"),
|
|
39
|
+
("permission is hereby granted, free of charge", "MIT-style permission notice"),
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
_GENERATED_PHRASES: tuple[str, ...] = (
|
|
43
|
+
"do not edit",
|
|
44
|
+
"do not modify",
|
|
45
|
+
"@generated",
|
|
46
|
+
"autogenerated",
|
|
47
|
+
"auto-generated",
|
|
48
|
+
"automatically generated",
|
|
49
|
+
"generated by",
|
|
50
|
+
"this file was generated",
|
|
51
|
+
"code generated by",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
_VENDOR_SEGMENTS = (
|
|
55
|
+
"vendor", "third_party", "thirdparty", "external", "extern", "deps",
|
|
56
|
+
"node_modules", "site-packages", "lib/external",
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
_CORP_SUFFIXES = re.compile(
|
|
60
|
+
r"(?i)\b(inc|inc\.|llc|ltd|ltd\.|limited|gmbh|corp|corporation|co|company|"
|
|
61
|
+
r"foundation|project|team|authors|contributors|and contributors|all rights reserved)\b"
|
|
62
|
+
)
|
|
63
|
+
_NON_WORD = re.compile(r"[^a-z0-9]+")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _normalise_holder(name: str) -> str:
|
|
67
|
+
name = name.strip().strip(".,;:-* \t")
|
|
68
|
+
name = _CORP_SUFFIXES.sub(" ", name.lower())
|
|
69
|
+
return _NON_WORD.sub(" ", name).strip()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _finding_id(rel_path: str, rule: str, detail: str) -> str:
|
|
73
|
+
digest = hashlib.sha256(f"{rel_path}|{rule}|{detail}".encode("utf-8")).hexdigest()
|
|
74
|
+
return f"{rule}-{digest[:12]}"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _same_owner(holder: str, owner: str) -> bool:
|
|
78
|
+
left, right = _normalise_holder(holder), _normalise_holder(owner)
|
|
79
|
+
if not left or not right:
|
|
80
|
+
return False
|
|
81
|
+
return left == right or left in right or right in left
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def scan_text(
|
|
85
|
+
rel_path: str,
|
|
86
|
+
comment_lines: list[str],
|
|
87
|
+
declared_owner: str = "",
|
|
88
|
+
) -> list[Finding]:
|
|
89
|
+
"""Inspect a file's header comments (and path) for third-party markers.
|
|
90
|
+
|
|
91
|
+
`comment_lines` is the file with all non-comment characters blanked out
|
|
92
|
+
(see ``strip.comments_only_lines``), so line numbers stay accurate while
|
|
93
|
+
the scanner never reads a character of actual code.
|
|
94
|
+
"""
|
|
95
|
+
findings: list[Finding] = []
|
|
96
|
+
lines = list(comment_lines)
|
|
97
|
+
head = "\n".join(lines[:HEADER_LINES])
|
|
98
|
+
lowered = head.lower()
|
|
99
|
+
|
|
100
|
+
lowered_path = rel_path.lower()
|
|
101
|
+
for segment in _VENDOR_SEGMENTS:
|
|
102
|
+
if f"/{segment}/" in f"/{lowered_path}":
|
|
103
|
+
findings.append(
|
|
104
|
+
Finding(
|
|
105
|
+
id=_finding_id(rel_path, "vendored-path", segment),
|
|
106
|
+
rel_path=rel_path,
|
|
107
|
+
line_number=1,
|
|
108
|
+
rule="vendored-path",
|
|
109
|
+
detail=f"path contains '{segment}', which usually holds third-party code",
|
|
110
|
+
severity=SEVERITY_MEDIUM,
|
|
111
|
+
category="third-party",
|
|
112
|
+
)
|
|
113
|
+
)
|
|
114
|
+
break
|
|
115
|
+
|
|
116
|
+
for number, line in enumerate(lines[:HEADER_LINES], start=1):
|
|
117
|
+
match = _COPYRIGHT.search(line)
|
|
118
|
+
if not match:
|
|
119
|
+
continue
|
|
120
|
+
if not (match.group("sym") or match.group("years")):
|
|
121
|
+
continue # the word alone is not a notice
|
|
122
|
+
holder = (match.group("holder") or "").strip(" .,;:*/-#")
|
|
123
|
+
holder = re.sub(r"(?i)\s*all rights reserved.*$", "", holder).strip(" .,;:*/-")
|
|
124
|
+
if not _HOLDER_OK.match(holder):
|
|
125
|
+
continue
|
|
126
|
+
if declared_owner and _same_owner(holder, declared_owner):
|
|
127
|
+
continue
|
|
128
|
+
findings.append(
|
|
129
|
+
Finding(
|
|
130
|
+
id=_finding_id(rel_path, "foreign-copyright", holder),
|
|
131
|
+
rel_path=rel_path,
|
|
132
|
+
line_number=number,
|
|
133
|
+
rule="foreign-copyright",
|
|
134
|
+
detail=(
|
|
135
|
+
f"header names '{holder}'"
|
|
136
|
+
+ (f", not '{declared_owner}'" if declared_owner else "")
|
|
137
|
+
),
|
|
138
|
+
severity=SEVERITY_HIGH if declared_owner else SEVERITY_MEDIUM,
|
|
139
|
+
excerpt=line.strip()[:120],
|
|
140
|
+
category="third-party",
|
|
141
|
+
)
|
|
142
|
+
)
|
|
143
|
+
break
|
|
144
|
+
|
|
145
|
+
spdx = _SPDX.search(head)
|
|
146
|
+
if spdx:
|
|
147
|
+
identifier = spdx.group("id").strip()
|
|
148
|
+
findings.append(
|
|
149
|
+
Finding(
|
|
150
|
+
id=_finding_id(rel_path, "spdx-license", identifier),
|
|
151
|
+
rel_path=rel_path,
|
|
152
|
+
line_number=1,
|
|
153
|
+
rule="spdx-license",
|
|
154
|
+
detail=f"declares licence '{identifier}'",
|
|
155
|
+
severity=SEVERITY_LOW,
|
|
156
|
+
category="third-party",
|
|
157
|
+
)
|
|
158
|
+
)
|
|
159
|
+
else:
|
|
160
|
+
for needle, label in _LICENSE_PHRASES:
|
|
161
|
+
if needle in lowered:
|
|
162
|
+
findings.append(
|
|
163
|
+
Finding(
|
|
164
|
+
id=_finding_id(rel_path, "license-text", label),
|
|
165
|
+
rel_path=rel_path,
|
|
166
|
+
line_number=1,
|
|
167
|
+
rule="license-text",
|
|
168
|
+
detail=f"header contains {label} text",
|
|
169
|
+
severity=SEVERITY_MEDIUM,
|
|
170
|
+
category="third-party",
|
|
171
|
+
)
|
|
172
|
+
)
|
|
173
|
+
break
|
|
174
|
+
|
|
175
|
+
for needle in _GENERATED_PHRASES:
|
|
176
|
+
if needle in lowered:
|
|
177
|
+
findings.append(
|
|
178
|
+
Finding(
|
|
179
|
+
id=_finding_id(rel_path, "generated-file", needle),
|
|
180
|
+
rel_path=rel_path,
|
|
181
|
+
line_number=1,
|
|
182
|
+
rule="generated-file",
|
|
183
|
+
detail=f"header says '{needle}'; generated code is usually not registrable authorship",
|
|
184
|
+
severity=SEVERITY_MEDIUM,
|
|
185
|
+
category="third-party",
|
|
186
|
+
)
|
|
187
|
+
)
|
|
188
|
+
break
|
|
189
|
+
|
|
190
|
+
return findings
|