codendium 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. codendium-1.0.0.dist-info/METADATA +332 -0
  2. codendium-1.0.0.dist-info/RECORD +45 -0
  3. codendium-1.0.0.dist-info/WHEEL +5 -0
  4. codendium-1.0.0.dist-info/entry_points.txt +6 -0
  5. codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
  6. codendium-1.0.0.dist-info/top_level.txt +1 -0
  7. copyright_deposit/__init__.py +15 -0
  8. copyright_deposit/__main__.py +34 -0
  9. copyright_deposit/assets/__init__.py +5 -0
  10. copyright_deposit/assets/fonts/README.md +31 -0
  11. copyright_deposit/assets/logo.svg +26 -0
  12. copyright_deposit/cli.py +314 -0
  13. copyright_deposit/config.py +269 -0
  14. copyright_deposit/core/__init__.py +1 -0
  15. copyright_deposit/core/deposit.py +117 -0
  16. copyright_deposit/core/discovery.py +260 -0
  17. copyright_deposit/core/encoding.py +136 -0
  18. copyright_deposit/core/languages.py +190 -0
  19. copyright_deposit/core/layout.py +349 -0
  20. copyright_deposit/core/lineranges.py +219 -0
  21. copyright_deposit/core/manifest.py +282 -0
  22. copyright_deposit/core/metrics.py +279 -0
  23. copyright_deposit/core/ordering.py +349 -0
  24. copyright_deposit/core/pipeline.py +386 -0
  25. copyright_deposit/core/redaction.py +162 -0
  26. copyright_deposit/core/render.py +242 -0
  27. copyright_deposit/core/scanning/__init__.py +61 -0
  28. copyright_deposit/core/scanning/secrets.py +181 -0
  29. copyright_deposit/core/scanning/thirdparty.py +190 -0
  30. copyright_deposit/core/strip/__init__.py +337 -0
  31. copyright_deposit/core/strip/cfamily_strip.py +235 -0
  32. copyright_deposit/core/strip/pygments_strip.py +85 -0
  33. copyright_deposit/core/strip/python_strip.py +131 -0
  34. copyright_deposit/gui/__init__.py +1 -0
  35. copyright_deposit/gui/app.py +34 -0
  36. copyright_deposit/gui/branding.py +83 -0
  37. copyright_deposit/gui/history.py +192 -0
  38. copyright_deposit/gui/main_window.py +617 -0
  39. copyright_deposit/gui/panels/__init__.py +1 -0
  40. copyright_deposit/gui/panels/estimate.py +166 -0
  41. copyright_deposit/gui/panels/files.py +635 -0
  42. copyright_deposit/gui/panels/identification.py +193 -0
  43. copyright_deposit/gui/panels/options.py +445 -0
  44. copyright_deposit/gui/panels/preflight.py +260 -0
  45. copyright_deposit/gui/workers.py +96 -0
@@ -0,0 +1,386 @@
1
+ """Pipeline orchestration.
2
+
3
+ Estimating and building share every stage up to and including layout, so
4
+ the page count shown in the GUI is the page count of the PDF - not an
5
+ approximation of it.
6
+
7
+ Results are cached on the file's SHA-256 plus the options that affect that
8
+ stage, which is what makes flipping a comment toggle re-estimate instantly
9
+ instead of re-reading and re-lexing the whole tree.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import dataclasses
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import Callable
18
+
19
+ from ..config import BuildSettings, TransformOptions
20
+ from . import deposit as deposit_mod
21
+ from . import layout as layout_mod
22
+ from . import lineranges, metrics, ordering, redaction
23
+ from .discovery import DiscoveredFile, DiscoveryResult, discover
24
+ from .encoding import read_source
25
+ from .layout import FileBlock, LayoutResult
26
+ from .ordering import OrderPlan
27
+ from .scanning import Finding, ScanReport
28
+ from .scanning import secrets as secrets_mod
29
+ from .scanning import thirdparty as thirdparty_mod
30
+ from .strip import StripResult, comments_only_lines, strip_source
31
+
32
+ Progress = Callable[[str, int, int], None]
33
+
34
+
35
+ def _noop(stage: str, current: int, total: int) -> None:
36
+ return None
37
+
38
+
39
+ @dataclass
40
+ class PreparedFile:
41
+ rel_path: str
42
+ abs_path: str
43
+ original_text: str
44
+ strip: StripResult
45
+ redaction: redaction.FileRedaction
46
+ selection: lineranges.Selection = field(default_factory=lineranges.Selection)
47
+ range_spec: str = ""
48
+ total_lines: int = 0
49
+ language: str = ""
50
+ encoding: str = ""
51
+ sha256: str = ""
52
+ warnings: list[str] = field(default_factory=list)
53
+
54
+ @property
55
+ def lines(self) -> list:
56
+ """The lines that actually reach the page."""
57
+ return self.selection.lines
58
+
59
+
60
+ @dataclass
61
+ class EstimateResult:
62
+ settings: BuildSettings
63
+ discovery: DiscoveryResult
64
+ plan: OrderPlan
65
+ layout: LayoutResult
66
+ selection: deposit_mod.DepositSelection
67
+ scan: ScanReport
68
+ redaction_report: redaction.RedactionReport
69
+ prepared: list[PreparedFile] = field(default_factory=list)
70
+ warnings: list[str] = field(default_factory=list)
71
+ geometry_note: str = ""
72
+
73
+ @property
74
+ def page_count(self) -> int:
75
+ return self.layout.page_count
76
+
77
+ @property
78
+ def deposit_pages(self) -> int:
79
+ return self.selection.deposited_pages
80
+
81
+ def omitted_files(self) -> list[str]:
82
+ return deposit_mod.files_entirely_omitted(self.selection, self.layout.file_ranges)
83
+
84
+ def split_files(self) -> list[str]:
85
+ return deposit_mod.files_partially_shown(self.selection, self.layout.file_ranges)
86
+
87
+
88
+ @dataclass
89
+ class BuildResult:
90
+ estimate: EstimateResult
91
+ outputs: list[str] = field(default_factory=list)
92
+ manifest_path: str = ""
93
+ summary_path: str = ""
94
+ blocked_by: list[Finding] = field(default_factory=list)
95
+
96
+ @property
97
+ def succeeded(self) -> bool:
98
+ return bool(self.outputs)
99
+
100
+
101
+ # What-if variants shown in the estimate panel, cheapest disclosure last.
102
+ WHAT_IF_VARIANTS: tuple[tuple[str, dict], ...] = (
103
+ ("Keep everything", {"strip_comments": False, "strip_docstrings": False, "collapse_blank_runs": False}),
104
+ ("Strip comments", {"strip_comments": True, "strip_docstrings": False, "collapse_blank_runs": False}),
105
+ ("Strip comments + docstrings", {"strip_comments": True, "strip_docstrings": True, "collapse_blank_runs": False}),
106
+ ("Strip all + collapse blanks", {"strip_comments": True, "strip_docstrings": True, "collapse_blank_runs": True}),
107
+ )
108
+
109
+
110
+ class Pipeline:
111
+ """Stateful across calls so repeated estimates stay cheap."""
112
+
113
+ def __init__(self) -> None:
114
+ self._discovery: dict[str, DiscoveryResult] = {}
115
+ self._text: dict[str, tuple[str, str, list[str]]] = {}
116
+ self._strip: dict[tuple[str, str], StripResult] = {}
117
+ self._scan: dict[tuple[str, str], tuple[list[Finding], list[Finding]]] = {}
118
+
119
+ # -- stages ------------------------------------------------------------
120
+
121
+ def discover(self, settings: BuildSettings, refresh: bool = False) -> DiscoveryResult:
122
+ key = settings.source_root + "|" + str(dataclasses.asdict(settings.discovery))
123
+ if refresh or key not in self._discovery:
124
+ self._discovery[key] = discover(settings.source_root, settings.discovery)
125
+ return self._discovery[key]
126
+
127
+ def read(self, file: DiscoveredFile) -> tuple[str, str, list[str]]:
128
+ cached = self._text.get(file.sha256)
129
+ if cached is None:
130
+ decoded = read_source(file.abs_path)
131
+ cached = (decoded.text, decoded.encoding, list(decoded.warnings))
132
+ self._text[file.sha256] = cached
133
+ return cached
134
+
135
+ def strip(self, file: DiscoveredFile, text: str, options: TransformOptions) -> StripResult:
136
+ key = (file.sha256, options.cache_key())
137
+ result = self._strip.get(key)
138
+ if result is None:
139
+ result = strip_source(text, file.rel_path, options)
140
+ self._strip[key] = result
141
+ return result
142
+
143
+ def scan(
144
+ self,
145
+ file: DiscoveredFile,
146
+ text: str,
147
+ owner: str,
148
+ settings: BuildSettings,
149
+ comment_spans: list,
150
+ ) -> tuple[list[Finding], list[Finding]]:
151
+ key = (file.sha256, owner)
152
+ cached = self._scan.get(key)
153
+ if cached is None:
154
+ # Secrets hide in code, so that scan reads everything; licence
155
+ # and authorship markers live in comments, so that scan reads
156
+ # only the comment layer and stays free of prose false hits.
157
+ found_secrets = secrets_mod.scan_text(file.rel_path, text) if settings.scan.scan_secrets else []
158
+ found_third: list[Finding] = []
159
+ if settings.scan.scan_third_party:
160
+ comment_lines = comments_only_lines(text, comment_spans)
161
+ found_third = thirdparty_mod.scan_text(file.rel_path, comment_lines, owner)
162
+ cached = (found_secrets, found_third)
163
+ self._scan[key] = cached
164
+ return cached
165
+
166
+ # -- estimate ----------------------------------------------------------
167
+
168
+ def estimate(
169
+ self,
170
+ settings: BuildSettings,
171
+ *,
172
+ transform: TransformOptions | None = None,
173
+ progress: Progress = _noop,
174
+ refresh: bool = False,
175
+ run_scans: bool = True,
176
+ ) -> EstimateResult:
177
+ transform = transform or settings.transform
178
+ warnings: list[str] = []
179
+
180
+ progress("Discovering files", 0, 1)
181
+ found = self.discover(settings, refresh=refresh)
182
+ warnings.extend(found.warnings)
183
+
184
+ plan = ordering.resolve_order(
185
+ settings.order_entries,
186
+ found.files,
187
+ include_unlisted=settings.include_unlisted,
188
+ excluded=settings.excluded,
189
+ )
190
+ warnings.extend(plan.warnings)
191
+ for entry in plan.unmatched:
192
+ warnings.append(f"Order entry '{entry}' matched no file.")
193
+
194
+ by_path = found.by_path()
195
+ ordered_files = [by_path[p] for p in plan.paths if p in by_path]
196
+
197
+ compiled, regex_warnings = redaction.compile_regexes(settings.redaction)
198
+ warnings.extend(regex_warnings)
199
+ manual = {m.replace("\\", "/") for m in settings.redaction.manual_lines}
200
+
201
+ scan_report = ScanReport()
202
+ redaction_report = redaction.RedactionReport()
203
+ prepared: list[PreparedFile] = []
204
+
205
+ total = len(ordered_files)
206
+ for index, file in enumerate(ordered_files, start=1):
207
+ progress("Reading and stripping", index, total)
208
+ text, encoding, read_warnings = self.read(file)
209
+ stripped = self.strip(file, text, transform)
210
+
211
+ # Line selection happens after stripping (so ranges refer to the
212
+ # editor's line numbers) and before redaction (so redaction
213
+ # indices line up with what is actually rendered).
214
+ range_spec = settings.line_ranges.get(file.rel_path, "")
215
+ ranges, range_warnings = lineranges.parse_ranges(range_spec, file.raw_line_count)
216
+ selection = lineranges.apply_selection(
217
+ stripped.lines, ranges, file.raw_line_count
218
+ )
219
+
220
+ file_redaction = redaction.compute_file_redaction(
221
+ file.rel_path, text, selection.lines, settings.redaction, compiled, manual
222
+ )
223
+ redaction_report.files.append(file_redaction)
224
+
225
+ if run_scans:
226
+ found_secrets, found_third = self.scan(
227
+ file,
228
+ text,
229
+ settings.header.copyright_owner,
230
+ settings,
231
+ stripped.comment_spans,
232
+ )
233
+ scan_report.secrets.extend(found_secrets)
234
+ scan_report.third_party.extend(found_third)
235
+
236
+ file_warnings = (
237
+ list(read_warnings)
238
+ + list(stripped.warnings)
239
+ + range_warnings
240
+ + selection.warnings
241
+ )
242
+ for message in file_warnings:
243
+ warnings.append(f"{file.rel_path}: {message}")
244
+
245
+ prepared.append(
246
+ PreparedFile(
247
+ rel_path=file.rel_path,
248
+ abs_path=file.abs_path,
249
+ original_text=text,
250
+ strip=stripped,
251
+ redaction=file_redaction,
252
+ selection=selection,
253
+ range_spec=range_spec,
254
+ total_lines=file.raw_line_count,
255
+ language=file.language,
256
+ encoding=encoding,
257
+ sha256=file.sha256,
258
+ warnings=file_warnings,
259
+ )
260
+ )
261
+
262
+ progress("Laying out pages", total, total)
263
+ blocks = [
264
+ FileBlock(
265
+ rel_path=p.rel_path,
266
+ lines=p.selection.lines,
267
+ language=p.language,
268
+ redactions=p.redaction.spans,
269
+ elisions=p.selection.elisions,
270
+ trailing_elision=p.selection.trailing,
271
+ range_label=p.selection.label,
272
+ )
273
+ for p in prepared
274
+ ]
275
+
276
+ geometry = metrics.build_geometry(
277
+ settings.layout, layout_mod.max_source_line_number(blocks)
278
+ )
279
+ warnings.extend(geometry.font.warnings)
280
+
281
+ result_layout = layout_mod.build_layout(
282
+ blocks, settings.header.lines(), settings.layout, geometry
283
+ )
284
+ warnings.extend(result_layout.warnings)
285
+
286
+ selection = deposit_mod.select_pages(result_layout.page_count, settings.deposit)
287
+ warnings.extend(redaction.check_compliance(redaction_report, selection.mode))
288
+
289
+ return EstimateResult(
290
+ settings=settings,
291
+ discovery=found,
292
+ plan=plan,
293
+ layout=result_layout,
294
+ selection=selection,
295
+ scan=scan_report,
296
+ redaction_report=redaction_report,
297
+ prepared=prepared,
298
+ warnings=warnings,
299
+ geometry_note=metrics.describe(geometry),
300
+ )
301
+
302
+ def what_if(
303
+ self,
304
+ settings: BuildSettings,
305
+ progress: Progress = _noop,
306
+ ) -> list[tuple[str, int, str]]:
307
+ """Page count under each comment policy.
308
+
309
+ Answers the question that actually matters: what do I change to get
310
+ under 50 pages and deposit the whole program?
311
+ """
312
+ rows: list[tuple[str, int, str]] = []
313
+ for index, (label, overrides) in enumerate(WHAT_IF_VARIANTS, start=1):
314
+ progress("Comparing policies", index, len(WHAT_IF_VARIANTS))
315
+ variant = dataclasses.replace(settings.transform, **overrides)
316
+ estimate = self.estimate(
317
+ settings, transform=variant, progress=_noop, run_scans=False
318
+ )
319
+ mode = (
320
+ "entire program"
321
+ if estimate.selection.mode == deposit_mod.MODE_ENTIRE
322
+ else "first 25 + last 25"
323
+ )
324
+ rows.append((label, estimate.page_count, mode))
325
+ return rows
326
+
327
+ # -- build -------------------------------------------------------------
328
+
329
+ def build(
330
+ self,
331
+ settings: BuildSettings,
332
+ *,
333
+ progress: Progress = _noop,
334
+ force: bool = False,
335
+ ) -> BuildResult:
336
+ from . import manifest as manifest_mod
337
+ from .render import render_pdf
338
+
339
+ estimate = self.estimate(settings, progress=progress)
340
+ result = BuildResult(estimate=estimate)
341
+
342
+ ignored = set(settings.scan.ignored_findings)
343
+ blocking = estimate.scan.blocking(ignored)
344
+ if blocking and settings.scan.block_on_secrets and not force:
345
+ result.blocked_by = blocking
346
+ return result
347
+
348
+ out_dir = Path(settings.output_dir or ".")
349
+ out_dir.mkdir(parents=True, exist_ok=True)
350
+ base = settings.output_basename or "deposit"
351
+ fingerprint = settings.content_fingerprint()
352
+
353
+ if settings.write_full_pdf:
354
+ progress("Rendering complete PDF", 1, 2)
355
+ full_path = out_dir / f"{base}_full.pdf"
356
+ render_pdf(
357
+ estimate.layout,
358
+ full_path,
359
+ settings.header,
360
+ settings.layout,
361
+ fingerprint=fingerprint,
362
+ )
363
+ result.outputs.append(str(full_path))
364
+
365
+ if settings.write_deposit_pdf:
366
+ progress("Rendering deposit copy", 2, 2)
367
+ deposit_path = out_dir / f"{base}_deposit.pdf"
368
+ render_pdf(
369
+ estimate.layout,
370
+ deposit_path,
371
+ settings.header,
372
+ settings.layout,
373
+ selection=estimate.selection,
374
+ fingerprint=fingerprint,
375
+ include_separator=settings.deposit.separator_page,
376
+ )
377
+ result.outputs.append(str(deposit_path))
378
+
379
+ if settings.write_manifest:
380
+ manifest_path, summary_path = manifest_mod.write_manifest(
381
+ estimate, out_dir, base, result.outputs
382
+ )
383
+ result.manifest_path = manifest_path
384
+ result.summary_path = summary_path
385
+
386
+ return result
@@ -0,0 +1,162 @@
1
+ """Trade-secret redaction (Compendium section 721.7).
2
+
3
+ Blocked-out material is drawn as a solid bar and the underlying glyphs are
4
+ never written into the PDF, so the text cannot be recovered by selecting or
5
+ extracting it. Redaction never changes the number of lines, so it cannot
6
+ shift pagination.
7
+
8
+ Region markers are located in the *original* text and tracked by original
9
+ line number, because the comment carrying the marker is usually removed by
10
+ the stripper before layout ever sees it.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from dataclasses import dataclass, field
17
+
18
+ from ..config import MAX_REDACTION_RATIO, RedactionRules
19
+ from .strip import SourceLine
20
+
21
+ Span = tuple[int, int]
22
+
23
+
24
+ @dataclass
25
+ class FileRedaction:
26
+ rel_path: str
27
+ spans: dict[int, list[Span]] = field(default_factory=dict) # index into lines
28
+ redacted_chars: int = 0
29
+ total_chars: int = 0
30
+ marker_lines: int = 0
31
+ regex_hits: int = 0
32
+
33
+
34
+ @dataclass
35
+ class RedactionReport:
36
+ files: list[FileRedaction] = field(default_factory=list)
37
+ warnings: list[str] = field(default_factory=list)
38
+
39
+ @property
40
+ def redacted_chars(self) -> int:
41
+ return sum(f.redacted_chars for f in self.files)
42
+
43
+ @property
44
+ def total_chars(self) -> int:
45
+ return sum(f.total_chars for f in self.files)
46
+
47
+ @property
48
+ def ratio(self) -> float:
49
+ return self.redacted_chars / self.total_chars if self.total_chars else 0.0
50
+
51
+ @property
52
+ def compliant(self) -> bool:
53
+ return self.ratio <= MAX_REDACTION_RATIO
54
+
55
+
56
+ def _merge(spans: list[Span]) -> list[Span]:
57
+ if not spans:
58
+ return []
59
+ spans = sorted(spans)
60
+ out = [spans[0]]
61
+ for a, b in spans[1:]:
62
+ last_a, last_b = out[-1]
63
+ if a <= last_b:
64
+ out[-1] = (last_a, max(last_b, b))
65
+ else:
66
+ out.append((a, b))
67
+ return out
68
+
69
+
70
+ def marker_line_numbers(original_text: str, rules: RedactionRules) -> set[int]:
71
+ """Original line numbers enclosed by BEGIN/END markers."""
72
+ if not rules.begin_marker or not rules.end_marker:
73
+ return set()
74
+ inside = False
75
+ result: set[int] = set()
76
+ for number, line in enumerate(original_text.split("\n"), start=1):
77
+ if rules.begin_marker in line:
78
+ inside = True
79
+ continue
80
+ if rules.end_marker in line:
81
+ inside = False
82
+ continue
83
+ if inside:
84
+ result.add(number)
85
+ return result
86
+
87
+
88
+ def compile_regexes(rules: RedactionRules) -> tuple[list[re.Pattern[str]], list[str]]:
89
+ compiled: list[re.Pattern[str]] = []
90
+ warnings: list[str] = []
91
+ for pattern in rules.regexes:
92
+ if not pattern.strip():
93
+ continue
94
+ try:
95
+ compiled.append(re.compile(pattern))
96
+ except re.error as exc:
97
+ warnings.append(f"Invalid redaction pattern '{pattern}': {exc}")
98
+ return compiled, warnings
99
+
100
+
101
+ def compute_file_redaction(
102
+ rel_path: str,
103
+ original_text: str,
104
+ lines: list[SourceLine],
105
+ rules: RedactionRules,
106
+ compiled: list[re.Pattern[str]] | None = None,
107
+ manual: set[str] | None = None,
108
+ ) -> FileRedaction:
109
+ out = FileRedaction(rel_path=rel_path)
110
+ out.total_chars = sum(len(line.text.strip()) for line in lines)
111
+ if not rules.enabled:
112
+ return out
113
+
114
+ if compiled is None:
115
+ compiled, _ = compile_regexes(rules)
116
+ if manual is None:
117
+ manual = {m.replace("\\", "/") for m in rules.manual_lines}
118
+
119
+ markers = marker_line_numbers(original_text, rules)
120
+
121
+ for index, line in enumerate(lines):
122
+ text = line.text
123
+ if not text.strip():
124
+ continue
125
+ spans: list[Span] = []
126
+ whole = line.number in markers or f"{rel_path}:{line.number}" in manual
127
+ if whole:
128
+ start = len(text) - len(text.lstrip())
129
+ spans.append((start, len(text)))
130
+ out.marker_lines += 1
131
+ else:
132
+ for pattern in compiled:
133
+ for match in pattern.finditer(text):
134
+ if match.end() > match.start():
135
+ spans.append((match.start(), match.end()))
136
+ out.regex_hits += 1
137
+ if spans:
138
+ merged = _merge(spans)
139
+ out.spans[index] = merged
140
+ out.redacted_chars += sum(b - a for a, b in merged)
141
+ return out
142
+
143
+
144
+ def check_compliance(report: RedactionReport, deposit_mode: str) -> list[str]:
145
+ """Warn when blocked-out material exceeds what section 721.7 allows."""
146
+ messages: list[str] = []
147
+ if report.redacted_chars == 0:
148
+ return messages
149
+ percent = report.ratio * 100
150
+ if not report.compliant:
151
+ messages.append(
152
+ f"Redaction covers {percent:.1f}% of the deposited text. "
153
+ f"Compendium sec. 721.7 caps blocked-out material at "
154
+ f"{MAX_REDACTION_RATIO * 100:.0f}% for this deposit option; "
155
+ "reduce the redacted regions or deposit more pages."
156
+ )
157
+ else:
158
+ messages.append(
159
+ f"Redaction covers {percent:.1f}% of the deposited text "
160
+ f"(within the {MAX_REDACTION_RATIO * 100:.0f}% limit)."
161
+ )
162
+ return messages