codendium 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. codendium-1.0.0.dist-info/METADATA +332 -0
  2. codendium-1.0.0.dist-info/RECORD +45 -0
  3. codendium-1.0.0.dist-info/WHEEL +5 -0
  4. codendium-1.0.0.dist-info/entry_points.txt +6 -0
  5. codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
  6. codendium-1.0.0.dist-info/top_level.txt +1 -0
  7. copyright_deposit/__init__.py +15 -0
  8. copyright_deposit/__main__.py +34 -0
  9. copyright_deposit/assets/__init__.py +5 -0
  10. copyright_deposit/assets/fonts/README.md +31 -0
  11. copyright_deposit/assets/logo.svg +26 -0
  12. copyright_deposit/cli.py +314 -0
  13. copyright_deposit/config.py +269 -0
  14. copyright_deposit/core/__init__.py +1 -0
  15. copyright_deposit/core/deposit.py +117 -0
  16. copyright_deposit/core/discovery.py +260 -0
  17. copyright_deposit/core/encoding.py +136 -0
  18. copyright_deposit/core/languages.py +190 -0
  19. copyright_deposit/core/layout.py +349 -0
  20. copyright_deposit/core/lineranges.py +219 -0
  21. copyright_deposit/core/manifest.py +282 -0
  22. copyright_deposit/core/metrics.py +279 -0
  23. copyright_deposit/core/ordering.py +349 -0
  24. copyright_deposit/core/pipeline.py +386 -0
  25. copyright_deposit/core/redaction.py +162 -0
  26. copyright_deposit/core/render.py +242 -0
  27. copyright_deposit/core/scanning/__init__.py +61 -0
  28. copyright_deposit/core/scanning/secrets.py +181 -0
  29. copyright_deposit/core/scanning/thirdparty.py +190 -0
  30. copyright_deposit/core/strip/__init__.py +337 -0
  31. copyright_deposit/core/strip/cfamily_strip.py +235 -0
  32. copyright_deposit/core/strip/pygments_strip.py +85 -0
  33. copyright_deposit/core/strip/python_strip.py +131 -0
  34. copyright_deposit/gui/__init__.py +1 -0
  35. copyright_deposit/gui/app.py +34 -0
  36. copyright_deposit/gui/branding.py +83 -0
  37. copyright_deposit/gui/history.py +192 -0
  38. copyright_deposit/gui/main_window.py +617 -0
  39. copyright_deposit/gui/panels/__init__.py +1 -0
  40. copyright_deposit/gui/panels/estimate.py +166 -0
  41. copyright_deposit/gui/panels/files.py +635 -0
  42. copyright_deposit/gui/panels/identification.py +193 -0
  43. copyright_deposit/gui/panels/options.py +445 -0
  44. copyright_deposit/gui/panels/preflight.py +260 -0
  45. copyright_deposit/gui/workers.py +96 -0
@@ -0,0 +1,242 @@
1
+ """PDF rendering.
2
+
3
+ Two properties are load-bearing:
4
+
5
+ * **Deterministic.** The canvas runs in reportlab's invariant mode with a
6
+ fixed document date, so the same settings and the same sources produce a
7
+ byte-identical PDF. A re-run therefore proves what was filed.
8
+ * **Redaction is real.** Blocked-out text is never written to the content
9
+ stream at all - only a filled rectangle is drawn - so it cannot be
10
+ recovered with copy/paste or a text extractor.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass
16
+ from pathlib import Path
17
+
18
+ from .. import DISPLAY_NAME, __version__
19
+ from ..config import HeaderInfo, LayoutOptions
20
+ from .deposit import MODE_HEAD_TAIL, DepositSelection
21
+ from .layout import (
22
+ KIND_ELISION,
23
+ KIND_HEADER,
24
+ KIND_NOTICE,
25
+ LayoutResult,
26
+ RenderLine,
27
+ )
28
+ from .metrics import PageGeometry
29
+
30
+ GUTTER_GRAY = 0.45
31
+ HEADER_GRAY = 0.35
32
+ ELISION_GRAY = 0.40
33
+
34
+
35
+ @dataclass
36
+ class RenderReport:
37
+ output_path: str
38
+ page_count: int
39
+ font: str
40
+ embedded: bool
41
+
42
+
43
+ def _canvas(path: str, geometry: PageGeometry, header: HeaderInfo, fingerprint: str):
44
+ from reportlab.pdfgen import canvas as rl_canvas
45
+
46
+ c = rl_canvas.Canvas(
47
+ path,
48
+ pagesize=(geometry.width, geometry.height),
49
+ invariant=1, # stable document id and creation date
50
+ pageCompression=1,
51
+ )
52
+ c.setTitle(header.title_line())
53
+ c.setAuthor(header.copyright_owner or "")
54
+ c.setSubject(header.deposit_label or "Source code deposit")
55
+ c.setCreator(f"{DISPLAY_NAME} {__version__}")
56
+ c.setKeywords(f"settings-fingerprint:{fingerprint[:16]}")
57
+ return c
58
+
59
+
60
+ def _draw_text_with_redactions(c, geometry: PageGeometry, line: RenderLine, x0: float, y: float) -> None:
61
+ font = geometry.font
62
+ if not line.redactions:
63
+ c.drawString(x0, y, line.text)
64
+ return
65
+
66
+ char_w = font.char_width
67
+ bar_bottom = y - font.size * 0.22
68
+ bar_height = font.size * 1.02
69
+
70
+ position = 0
71
+ for start, end in line.redactions:
72
+ start = max(0, min(start, len(line.text)))
73
+ end = max(start, min(end, len(line.text)))
74
+ if start > position:
75
+ c.drawString(x0 + position * char_w, y, line.text[position:start])
76
+ c.saveState()
77
+ c.setFillGray(0.0)
78
+ c.rect(x0 + start * char_w, bar_bottom, (end - start) * char_w, bar_height, fill=1, stroke=0)
79
+ c.restoreState()
80
+ position = end
81
+ if position < len(line.text):
82
+ c.drawString(x0 + position * char_w, y, line.text[position:])
83
+
84
+
85
+ def _draw_page(
86
+ c,
87
+ geometry: PageGeometry,
88
+ lines: list[RenderLine],
89
+ *,
90
+ printed_number: int,
91
+ total_label: int,
92
+ title: str,
93
+ sheet_label: str = "",
94
+ show_number: bool = True,
95
+ ) -> None:
96
+ font = geometry.font
97
+ code_x = geometry.column_x(geometry.gutter_columns)
98
+
99
+ if geometry.running_header:
100
+ c.saveState()
101
+ c.setFont(font.regular, max(6.5, font.size * 0.72))
102
+ c.setFillGray(HEADER_GRAY)
103
+ header_y = geometry.height - geometry.margin - font.size * 0.9
104
+ c.drawString(geometry.margin, header_y, title[:70])
105
+ current_file = next((ln.file_path for ln in lines if ln.file_path), "")
106
+ if current_file:
107
+ c.drawRightString(geometry.width - geometry.margin, header_y, current_file[-70:])
108
+ c.restoreState()
109
+
110
+ for row, line in enumerate(lines):
111
+ y = geometry.baseline(row)
112
+ if not line.text and not line.redactions:
113
+ continue
114
+
115
+ if line.number is not None and geometry.gutter_columns:
116
+ c.saveState()
117
+ c.setFont(font.regular, font.size)
118
+ c.setFillGray(GUTTER_GRAY)
119
+ c.drawRightString(
120
+ geometry.column_x(geometry.gutter_columns - 1),
121
+ y,
122
+ str(line.number),
123
+ )
124
+ c.restoreState()
125
+
126
+ c.setFont(font.bold if line.bold else font.regular, font.size)
127
+ # Elision rules are not source text, so they are set in grey: a
128
+ # reader can tell at a glance that nothing was written there.
129
+ c.setFillGray(ELISION_GRAY if line.kind == KIND_ELISION else 0.0)
130
+ x0 = geometry.margin if line.kind in (KIND_HEADER, KIND_NOTICE) else code_x
131
+ _draw_text_with_redactions(c, geometry, line, x0, y)
132
+
133
+ if geometry.page_numbers:
134
+ c.saveState()
135
+ c.setFont(font.regular, max(6.5, font.size * 0.78))
136
+ c.setFillGray(HEADER_GRAY)
137
+ footer_y = geometry.margin * 0.55
138
+ if show_number:
139
+ c.drawCentredString(
140
+ geometry.width / 2.0,
141
+ footer_y,
142
+ f"Page {printed_number} of {total_label}",
143
+ )
144
+ if sheet_label:
145
+ c.drawRightString(geometry.width - geometry.margin, footer_y, sheet_label)
146
+ c.restoreState()
147
+
148
+ c.showPage()
149
+
150
+
151
+ def _separator_lines(text: str, geometry: PageGeometry) -> list[RenderLine]:
152
+ """Center the omission notice on its own sheet."""
153
+ from .layout import wrap_source_line
154
+
155
+ width = min(geometry.code_columns, 72)
156
+ chunks = [body for _s, _p, body in wrap_source_line(text, width, "")]
157
+ top_pad = max(0, (geometry.lines_per_page - len(chunks)) // 2)
158
+ lines = [RenderLine() for _ in range(top_pad)]
159
+ lines.extend(RenderLine(text=chunk, kind=KIND_NOTICE, bold=True) for chunk in chunks)
160
+ return lines
161
+
162
+
163
+ def render_pdf(
164
+ layout: LayoutResult,
165
+ output_path: str | Path,
166
+ header: HeaderInfo,
167
+ options: LayoutOptions,
168
+ *,
169
+ selection: DepositSelection | None = None,
170
+ fingerprint: str = "",
171
+ include_separator: bool = True,
172
+ ) -> RenderReport:
173
+ """Write `layout` to a PDF, optionally restricted to a deposit selection."""
174
+ geometry = layout.geometry
175
+ if geometry is None:
176
+ raise ValueError("layout has no geometry; build it with build_layout()")
177
+
178
+ output_path = str(output_path)
179
+ Path(output_path).parent.mkdir(parents=True, exist_ok=True)
180
+
181
+ pages = layout.pages
182
+ total_label = len(pages)
183
+ title = header.title_line()
184
+
185
+ if selection is None:
186
+ chosen = list(range(1, len(pages) + 1))
187
+ separator_after = None
188
+ sheets = len(chosen)
189
+ else:
190
+ chosen = list(selection.page_numbers)
191
+ total_label = selection.total_pages
192
+ sheets = len(chosen)
193
+ separator_after = None
194
+ if selection.mode == MODE_HEAD_TAIL and include_separator:
195
+ separator_after = min(len(chosen), max(0, selection_head_count(selection)))
196
+
197
+ c = _canvas(output_path, geometry, header, fingerprint)
198
+
199
+ emitted = 0
200
+ for position, page_number in enumerate(chosen):
201
+ page = pages[page_number - 1]
202
+ emitted += 1
203
+ _draw_page(
204
+ c,
205
+ geometry,
206
+ page.lines,
207
+ printed_number=page_number,
208
+ total_label=total_label,
209
+ title=title,
210
+ sheet_label=(
211
+ f"Deposit sheet {emitted} of {sheets}" if selection and selection.mode == MODE_HEAD_TAIL else ""
212
+ ),
213
+ )
214
+ if separator_after is not None and position + 1 == separator_after:
215
+ notice = selection.separator_notice() if selection else ""
216
+ if notice:
217
+ _draw_page(
218
+ c,
219
+ geometry,
220
+ _separator_lines(notice, geometry),
221
+ printed_number=page_number,
222
+ total_label=total_label,
223
+ title=title,
224
+ sheet_label="omission notice",
225
+ show_number=False,
226
+ )
227
+
228
+ c.save()
229
+ return RenderReport(
230
+ output_path=output_path,
231
+ page_count=emitted,
232
+ font=geometry.font.regular,
233
+ embedded=geometry.font.embedded,
234
+ )
235
+
236
+
237
+ def selection_head_count(selection: DepositSelection) -> int:
238
+ """How many leading pages precede the omission gap."""
239
+ if selection.omitted is None:
240
+ return len(selection.page_numbers)
241
+ gap_start = selection.omitted[0]
242
+ return sum(1 for n in selection.page_numbers if n < gap_start)
@@ -0,0 +1,61 @@
1
+ """Pre-flight scanners.
2
+
3
+ A deposit becomes part of a public record that anyone may inspect, and a
4
+ registration may only claim material the applicant actually owns. These
5
+ scanners exist to catch both mistakes before the PDF is filed.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+
12
+ SEVERITY_HIGH = "high"
13
+ SEVERITY_MEDIUM = "medium"
14
+ SEVERITY_LOW = "low"
15
+
16
+ _SEVERITY_ORDER = {SEVERITY_HIGH: 0, SEVERITY_MEDIUM: 1, SEVERITY_LOW: 2}
17
+
18
+
19
+ @dataclass
20
+ class Finding:
21
+ id: str
22
+ rel_path: str
23
+ line_number: int
24
+ rule: str
25
+ detail: str
26
+ severity: str = SEVERITY_MEDIUM
27
+ excerpt: str = ""
28
+ category: str = "secret"
29
+
30
+ def location(self) -> str:
31
+ return f"{self.rel_path}:{self.line_number}"
32
+
33
+
34
+ @dataclass
35
+ class ScanReport:
36
+ secrets: list[Finding] = field(default_factory=list)
37
+ third_party: list[Finding] = field(default_factory=list)
38
+ warnings: list[str] = field(default_factory=list)
39
+
40
+ @property
41
+ def all_findings(self) -> list[Finding]:
42
+ return self.secrets + self.third_party
43
+
44
+ def blocking(self, ignored: set[str]) -> list[Finding]:
45
+ return [
46
+ f
47
+ for f in self.secrets
48
+ if f.severity == SEVERITY_HIGH and f.id not in ignored
49
+ ]
50
+
51
+ def sorted_secrets(self) -> list[Finding]:
52
+ return sorted(
53
+ self.secrets,
54
+ key=lambda f: (_SEVERITY_ORDER.get(f.severity, 9), f.rel_path, f.line_number),
55
+ )
56
+
57
+ def sorted_third_party(self) -> list[Finding]:
58
+ return sorted(
59
+ self.third_party,
60
+ key=lambda f: (_SEVERITY_ORDER.get(f.severity, 9), f.rel_path, f.line_number),
61
+ )
@@ -0,0 +1,181 @@
1
+ """Credential and PII detection.
2
+
3
+ Anything deposited can be inspected by the public, so a hardcoded key in
4
+ the source is not just a security bug - it is a disclosure. Findings are
5
+ reported with a masked excerpt, and each carries a stable id so the
6
+ operator's decision to ignore or redact it survives into the next run.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import math
13
+ import re
14
+
15
+ from . import SEVERITY_HIGH, SEVERITY_LOW, SEVERITY_MEDIUM, Finding
16
+
17
+ # (rule name, pattern, severity, human explanation)
18
+ _RULES: tuple[tuple[str, re.Pattern[str], str, str], ...] = (
19
+ (
20
+ "aws-access-key",
21
+ re.compile(r"\b((?:AKIA|ASIA|ABIA|ACCA)[0-9A-Z]{16})\b"),
22
+ SEVERITY_HIGH,
23
+ "AWS access key id",
24
+ ),
25
+ (
26
+ "private-key",
27
+ re.compile(r"-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----"),
28
+ SEVERITY_HIGH,
29
+ "embedded private key block",
30
+ ),
31
+ (
32
+ "github-token",
33
+ re.compile(r"\b(gh[pousr]_[A-Za-z0-9]{16,})\b"),
34
+ SEVERITY_HIGH,
35
+ "GitHub token",
36
+ ),
37
+ (
38
+ "slack-token",
39
+ re.compile(r"\b(xox[baprs]-[A-Za-z0-9-]{10,})\b"),
40
+ SEVERITY_HIGH,
41
+ "Slack token",
42
+ ),
43
+ (
44
+ "google-api-key",
45
+ re.compile(r"\b(AIza[0-9A-Za-z_\-]{35})\b"),
46
+ SEVERITY_HIGH,
47
+ "Google API key",
48
+ ),
49
+ (
50
+ "stripe-key",
51
+ re.compile(r"\b((?:sk|rk)_(?:live|test)_[0-9A-Za-z]{16,})\b"),
52
+ SEVERITY_HIGH,
53
+ "Stripe secret key",
54
+ ),
55
+ (
56
+ "openai-key",
57
+ re.compile(r"\b(sk-[A-Za-z0-9_\-]{20,})\b"),
58
+ SEVERITY_HIGH,
59
+ "OpenAI-style API key",
60
+ ),
61
+ (
62
+ "jwt",
63
+ re.compile(r"\beyJ[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}\b"),
64
+ SEVERITY_MEDIUM,
65
+ "JSON web token",
66
+ ),
67
+ (
68
+ "connection-string",
69
+ re.compile(r"\b[a-z][a-z0-9+.\-]*://[^\s:/@]+:[^\s:/@]+@[^\s/]+", re.IGNORECASE),
70
+ SEVERITY_HIGH,
71
+ "URL containing credentials",
72
+ ),
73
+ (
74
+ "email",
75
+ re.compile(r"\b[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}\b"),
76
+ SEVERITY_LOW,
77
+ "email address",
78
+ ),
79
+ )
80
+
81
+ # name = "value" style assignments worth entropy-checking
82
+ # The prefix is optional so that a name which *is* the keyword - `api_key`,
83
+ # `token` - matches as readily as `service_api_key`.
84
+ _ASSIGNMENT = re.compile(
85
+ r"""(?ix)
86
+ \b(?P<name>[A-Za-z0-9_]*
87
+ (?:pass(?:wd|word)?|secret|token|api[_-]?key|apikey|auth|credential|private[_-]?key)
88
+ [A-Za-z0-9_]*)
89
+ \s*[:=]\s*
90
+ (?P<quote>["'])(?P<value>[^"']{8,})(?P=quote)
91
+ """
92
+ )
93
+
94
+ # Values that are obviously not real credentials. The tails accept hyphens
95
+ # and dots so that "your-api-key-here" is recognised as a placeholder.
96
+ _PLACEHOLDER = re.compile(
97
+ r"(?i)^(?:\s*|x{3,}|\.{3,}|none|null|true|false|changeme|placeholder|redacted|"
98
+ r"(?:your|my|the|some|example|dummy|test|sample|fake|todo|insert|enter|put)"
99
+ r"[\w\-. ]*|"
100
+ r"\$\{.*\}|\{\{.*\}\}|<.*>|%\w+%|\*+)$"
101
+ )
102
+
103
+ _NON_SECRET_EXT = (".md", ".rst", ".txt")
104
+
105
+
106
+ def _shannon_entropy(value: str) -> float:
107
+ if not value:
108
+ return 0.0
109
+ counts: dict[str, int] = {}
110
+ for ch in value:
111
+ counts[ch] = counts.get(ch, 0) + 1
112
+ length = len(value)
113
+ return -sum((c / length) * math.log2(c / length) for c in counts.values())
114
+
115
+
116
+ def _mask(value: str) -> str:
117
+ if len(value) <= 8:
118
+ return value[0] + "*" * (len(value) - 1) if value else ""
119
+ return f"{value[:4]}{'*' * 8}{value[-2:]}"
120
+
121
+
122
+ def _finding_id(rel_path: str, line_number: int, rule: str, value: str) -> str:
123
+ digest = hashlib.sha256(f"{rel_path}|{rule}|{value}".encode("utf-8")).hexdigest()
124
+ return f"{rule}-{digest[:12]}"
125
+
126
+
127
+ def scan_text(rel_path: str, text: str) -> list[Finding]:
128
+ """Find credentials and PII in one file's original text."""
129
+ findings: list[Finding] = []
130
+ seen: set[tuple[int, str]] = set()
131
+ allow_low = not rel_path.lower().endswith(_NON_SECRET_EXT)
132
+
133
+ for number, line in enumerate(text.split("\n"), start=1):
134
+ if len(line) > 4000:
135
+ line = line[:4000]
136
+
137
+ for rule, pattern, severity, detail in _RULES:
138
+ if severity == SEVERITY_LOW and not allow_low:
139
+ continue
140
+ for match in pattern.finditer(line):
141
+ value = match.group(1) if match.groups() else match.group(0)
142
+ if _PLACEHOLDER.match(value):
143
+ continue
144
+ key = (number, rule)
145
+ if key in seen:
146
+ continue
147
+ seen.add(key)
148
+ findings.append(
149
+ Finding(
150
+ id=_finding_id(rel_path, number, rule, value),
151
+ rel_path=rel_path,
152
+ line_number=number,
153
+ rule=rule,
154
+ detail=detail,
155
+ severity=severity,
156
+ excerpt=_mask(value),
157
+ category="secret",
158
+ )
159
+ )
160
+
161
+ match = _ASSIGNMENT.search(line)
162
+ if match:
163
+ value = match.group("value")
164
+ name = match.group("name")
165
+ if not _PLACEHOLDER.match(value) and _shannon_entropy(value) >= 3.0:
166
+ key = (number, "hardcoded-credential")
167
+ if key not in seen:
168
+ seen.add(key)
169
+ findings.append(
170
+ Finding(
171
+ id=_finding_id(rel_path, number, "hardcoded-credential", value),
172
+ rel_path=rel_path,
173
+ line_number=number,
174
+ rule="hardcoded-credential",
175
+ detail=f"high-entropy value assigned to '{name}'",
176
+ severity=SEVERITY_HIGH,
177
+ excerpt=_mask(value),
178
+ category="secret",
179
+ )
180
+ )
181
+ return findings
@@ -0,0 +1,190 @@
1
+ """Detection of code the applicant may not own.
2
+
3
+ A registration covers the applicant's own authorship. Vendored libraries,
4
+ generated files and headers naming a different copyright holder all need a
5
+ decision - exclude them, or disclaim them in the application - before the
6
+ deposit is filed.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import re
13
+
14
+ from . import SEVERITY_HIGH, SEVERITY_LOW, SEVERITY_MEDIUM, Finding
15
+
16
+ HEADER_LINES = 60
17
+
18
+ # A real notice carries a (c)/copyright symbol or a year. Without that
19
+ # requirement the word "copyright" in ordinary prose - or in a variable
20
+ # name - produces a flood of false positives.
21
+ _COPYRIGHT = re.compile(
22
+ r"(?i)\bcopyright\b\s*(?P<sym>\(c\)|©|&copy;)?\s*"
23
+ r"(?P<years>\d{4}(?:\s*[-,]\s*\d{4})*)?\s*"
24
+ r"(?:by\s+)?(?P<holder>[^\n<]{2,80})?"
25
+ )
26
+
27
+ # A plausible human or organisation name, not a fragment of source code.
28
+ _HOLDER_OK = re.compile(r"^[A-Za-z][A-Za-z0-9 .,&'’à-ÿ-]{1,60}$")
29
+ _SPDX = re.compile(r"SPDX-License-Identifier:\s*(?P<id>[A-Za-z0-9.\-+ ()]+)")
30
+
31
+ _LICENSE_PHRASES: tuple[tuple[str, str], ...] = (
32
+ ("apache license", "Apache License"),
33
+ ("mit license", "MIT License"),
34
+ ("bsd license", "BSD License"),
35
+ ("redistribution and use in source and binary forms", "BSD-style licence text"),
36
+ ("gnu general public license", "GNU GPL"),
37
+ ("gnu lesser general public license", "GNU LGPL"),
38
+ ("mozilla public license", "Mozilla Public License"),
39
+ ("permission is hereby granted, free of charge", "MIT-style permission notice"),
40
+ )
41
+
42
+ _GENERATED_PHRASES: tuple[str, ...] = (
43
+ "do not edit",
44
+ "do not modify",
45
+ "@generated",
46
+ "autogenerated",
47
+ "auto-generated",
48
+ "automatically generated",
49
+ "generated by",
50
+ "this file was generated",
51
+ "code generated by",
52
+ )
53
+
54
+ _VENDOR_SEGMENTS = (
55
+ "vendor", "third_party", "thirdparty", "external", "extern", "deps",
56
+ "node_modules", "site-packages", "lib/external",
57
+ )
58
+
59
+ _CORP_SUFFIXES = re.compile(
60
+ r"(?i)\b(inc|inc\.|llc|ltd|ltd\.|limited|gmbh|corp|corporation|co|company|"
61
+ r"foundation|project|team|authors|contributors|and contributors|all rights reserved)\b"
62
+ )
63
+ _NON_WORD = re.compile(r"[^a-z0-9]+")
64
+
65
+
66
+ def _normalise_holder(name: str) -> str:
67
+ name = name.strip().strip(".,;:-* \t")
68
+ name = _CORP_SUFFIXES.sub(" ", name.lower())
69
+ return _NON_WORD.sub(" ", name).strip()
70
+
71
+
72
+ def _finding_id(rel_path: str, rule: str, detail: str) -> str:
73
+ digest = hashlib.sha256(f"{rel_path}|{rule}|{detail}".encode("utf-8")).hexdigest()
74
+ return f"{rule}-{digest[:12]}"
75
+
76
+
77
+ def _same_owner(holder: str, owner: str) -> bool:
78
+ left, right = _normalise_holder(holder), _normalise_holder(owner)
79
+ if not left or not right:
80
+ return False
81
+ return left == right or left in right or right in left
82
+
83
+
84
+ def scan_text(
85
+ rel_path: str,
86
+ comment_lines: list[str],
87
+ declared_owner: str = "",
88
+ ) -> list[Finding]:
89
+ """Inspect a file's header comments (and path) for third-party markers.
90
+
91
+ `comment_lines` is the file with all non-comment characters blanked out
92
+ (see ``strip.comments_only_lines``), so line numbers stay accurate while
93
+ the scanner never reads a character of actual code.
94
+ """
95
+ findings: list[Finding] = []
96
+ lines = list(comment_lines)
97
+ head = "\n".join(lines[:HEADER_LINES])
98
+ lowered = head.lower()
99
+
100
+ lowered_path = rel_path.lower()
101
+ for segment in _VENDOR_SEGMENTS:
102
+ if f"/{segment}/" in f"/{lowered_path}":
103
+ findings.append(
104
+ Finding(
105
+ id=_finding_id(rel_path, "vendored-path", segment),
106
+ rel_path=rel_path,
107
+ line_number=1,
108
+ rule="vendored-path",
109
+ detail=f"path contains '{segment}', which usually holds third-party code",
110
+ severity=SEVERITY_MEDIUM,
111
+ category="third-party",
112
+ )
113
+ )
114
+ break
115
+
116
+ for number, line in enumerate(lines[:HEADER_LINES], start=1):
117
+ match = _COPYRIGHT.search(line)
118
+ if not match:
119
+ continue
120
+ if not (match.group("sym") or match.group("years")):
121
+ continue # the word alone is not a notice
122
+ holder = (match.group("holder") or "").strip(" .,;:*/-#")
123
+ holder = re.sub(r"(?i)\s*all rights reserved.*$", "", holder).strip(" .,;:*/-")
124
+ if not _HOLDER_OK.match(holder):
125
+ continue
126
+ if declared_owner and _same_owner(holder, declared_owner):
127
+ continue
128
+ findings.append(
129
+ Finding(
130
+ id=_finding_id(rel_path, "foreign-copyright", holder),
131
+ rel_path=rel_path,
132
+ line_number=number,
133
+ rule="foreign-copyright",
134
+ detail=(
135
+ f"header names '{holder}'"
136
+ + (f", not '{declared_owner}'" if declared_owner else "")
137
+ ),
138
+ severity=SEVERITY_HIGH if declared_owner else SEVERITY_MEDIUM,
139
+ excerpt=line.strip()[:120],
140
+ category="third-party",
141
+ )
142
+ )
143
+ break
144
+
145
+ spdx = _SPDX.search(head)
146
+ if spdx:
147
+ identifier = spdx.group("id").strip()
148
+ findings.append(
149
+ Finding(
150
+ id=_finding_id(rel_path, "spdx-license", identifier),
151
+ rel_path=rel_path,
152
+ line_number=1,
153
+ rule="spdx-license",
154
+ detail=f"declares licence '{identifier}'",
155
+ severity=SEVERITY_LOW,
156
+ category="third-party",
157
+ )
158
+ )
159
+ else:
160
+ for needle, label in _LICENSE_PHRASES:
161
+ if needle in lowered:
162
+ findings.append(
163
+ Finding(
164
+ id=_finding_id(rel_path, "license-text", label),
165
+ rel_path=rel_path,
166
+ line_number=1,
167
+ rule="license-text",
168
+ detail=f"header contains {label} text",
169
+ severity=SEVERITY_MEDIUM,
170
+ category="third-party",
171
+ )
172
+ )
173
+ break
174
+
175
+ for needle in _GENERATED_PHRASES:
176
+ if needle in lowered:
177
+ findings.append(
178
+ Finding(
179
+ id=_finding_id(rel_path, "generated-file", needle),
180
+ rel_path=rel_path,
181
+ line_number=1,
182
+ rule="generated-file",
183
+ detail=f"header says '{needle}'; generated code is usually not registrable authorship",
184
+ severity=SEVERITY_MEDIUM,
185
+ category="third-party",
186
+ )
187
+ )
188
+ break
189
+
190
+ return findings