codendium 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codendium-1.0.0.dist-info/METADATA +332 -0
- codendium-1.0.0.dist-info/RECORD +45 -0
- codendium-1.0.0.dist-info/WHEEL +5 -0
- codendium-1.0.0.dist-info/entry_points.txt +6 -0
- codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
- codendium-1.0.0.dist-info/top_level.txt +1 -0
- copyright_deposit/__init__.py +15 -0
- copyright_deposit/__main__.py +34 -0
- copyright_deposit/assets/__init__.py +5 -0
- copyright_deposit/assets/fonts/README.md +31 -0
- copyright_deposit/assets/logo.svg +26 -0
- copyright_deposit/cli.py +314 -0
- copyright_deposit/config.py +269 -0
- copyright_deposit/core/__init__.py +1 -0
- copyright_deposit/core/deposit.py +117 -0
- copyright_deposit/core/discovery.py +260 -0
- copyright_deposit/core/encoding.py +136 -0
- copyright_deposit/core/languages.py +190 -0
- copyright_deposit/core/layout.py +349 -0
- copyright_deposit/core/lineranges.py +219 -0
- copyright_deposit/core/manifest.py +282 -0
- copyright_deposit/core/metrics.py +279 -0
- copyright_deposit/core/ordering.py +349 -0
- copyright_deposit/core/pipeline.py +386 -0
- copyright_deposit/core/redaction.py +162 -0
- copyright_deposit/core/render.py +242 -0
- copyright_deposit/core/scanning/__init__.py +61 -0
- copyright_deposit/core/scanning/secrets.py +181 -0
- copyright_deposit/core/scanning/thirdparty.py +190 -0
- copyright_deposit/core/strip/__init__.py +337 -0
- copyright_deposit/core/strip/cfamily_strip.py +235 -0
- copyright_deposit/core/strip/pygments_strip.py +85 -0
- copyright_deposit/core/strip/python_strip.py +131 -0
- copyright_deposit/gui/__init__.py +1 -0
- copyright_deposit/gui/app.py +34 -0
- copyright_deposit/gui/branding.py +83 -0
- copyright_deposit/gui/history.py +192 -0
- copyright_deposit/gui/main_window.py +617 -0
- copyright_deposit/gui/panels/__init__.py +1 -0
- copyright_deposit/gui/panels/estimate.py +166 -0
- copyright_deposit/gui/panels/files.py +635 -0
- copyright_deposit/gui/panels/identification.py +193 -0
- copyright_deposit/gui/panels/options.py +445 -0
- copyright_deposit/gui/panels/preflight.py +260 -0
- copyright_deposit/gui/workers.py +96 -0
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
"""Pipeline orchestration.
|
|
2
|
+
|
|
3
|
+
Estimating and building share every stage up to and including layout, so
|
|
4
|
+
the page count shown in the GUI is the page count of the PDF - not an
|
|
5
|
+
approximation of it.
|
|
6
|
+
|
|
7
|
+
Results are cached on the file's SHA-256 plus the options that affect that
|
|
8
|
+
stage, which is what makes flipping a comment toggle re-estimate instantly
|
|
9
|
+
instead of re-reading and re-lexing the whole tree.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import dataclasses
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Callable
|
|
18
|
+
|
|
19
|
+
from ..config import BuildSettings, TransformOptions
|
|
20
|
+
from . import deposit as deposit_mod
|
|
21
|
+
from . import layout as layout_mod
|
|
22
|
+
from . import lineranges, metrics, ordering, redaction
|
|
23
|
+
from .discovery import DiscoveredFile, DiscoveryResult, discover
|
|
24
|
+
from .encoding import read_source
|
|
25
|
+
from .layout import FileBlock, LayoutResult
|
|
26
|
+
from .ordering import OrderPlan
|
|
27
|
+
from .scanning import Finding, ScanReport
|
|
28
|
+
from .scanning import secrets as secrets_mod
|
|
29
|
+
from .scanning import thirdparty as thirdparty_mod
|
|
30
|
+
from .strip import StripResult, comments_only_lines, strip_source
|
|
31
|
+
|
|
32
|
+
Progress = Callable[[str, int, int], None]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _noop(stage: str, current: int, total: int) -> None:
|
|
36
|
+
return None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class PreparedFile:
|
|
41
|
+
rel_path: str
|
|
42
|
+
abs_path: str
|
|
43
|
+
original_text: str
|
|
44
|
+
strip: StripResult
|
|
45
|
+
redaction: redaction.FileRedaction
|
|
46
|
+
selection: lineranges.Selection = field(default_factory=lineranges.Selection)
|
|
47
|
+
range_spec: str = ""
|
|
48
|
+
total_lines: int = 0
|
|
49
|
+
language: str = ""
|
|
50
|
+
encoding: str = ""
|
|
51
|
+
sha256: str = ""
|
|
52
|
+
warnings: list[str] = field(default_factory=list)
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def lines(self) -> list:
|
|
56
|
+
"""The lines that actually reach the page."""
|
|
57
|
+
return self.selection.lines
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class EstimateResult:
|
|
62
|
+
settings: BuildSettings
|
|
63
|
+
discovery: DiscoveryResult
|
|
64
|
+
plan: OrderPlan
|
|
65
|
+
layout: LayoutResult
|
|
66
|
+
selection: deposit_mod.DepositSelection
|
|
67
|
+
scan: ScanReport
|
|
68
|
+
redaction_report: redaction.RedactionReport
|
|
69
|
+
prepared: list[PreparedFile] = field(default_factory=list)
|
|
70
|
+
warnings: list[str] = field(default_factory=list)
|
|
71
|
+
geometry_note: str = ""
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def page_count(self) -> int:
|
|
75
|
+
return self.layout.page_count
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def deposit_pages(self) -> int:
|
|
79
|
+
return self.selection.deposited_pages
|
|
80
|
+
|
|
81
|
+
def omitted_files(self) -> list[str]:
|
|
82
|
+
return deposit_mod.files_entirely_omitted(self.selection, self.layout.file_ranges)
|
|
83
|
+
|
|
84
|
+
def split_files(self) -> list[str]:
|
|
85
|
+
return deposit_mod.files_partially_shown(self.selection, self.layout.file_ranges)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass
|
|
89
|
+
class BuildResult:
|
|
90
|
+
estimate: EstimateResult
|
|
91
|
+
outputs: list[str] = field(default_factory=list)
|
|
92
|
+
manifest_path: str = ""
|
|
93
|
+
summary_path: str = ""
|
|
94
|
+
blocked_by: list[Finding] = field(default_factory=list)
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def succeeded(self) -> bool:
|
|
98
|
+
return bool(self.outputs)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# What-if variants shown in the estimate panel, cheapest disclosure last.
|
|
102
|
+
WHAT_IF_VARIANTS: tuple[tuple[str, dict], ...] = (
|
|
103
|
+
("Keep everything", {"strip_comments": False, "strip_docstrings": False, "collapse_blank_runs": False}),
|
|
104
|
+
("Strip comments", {"strip_comments": True, "strip_docstrings": False, "collapse_blank_runs": False}),
|
|
105
|
+
("Strip comments + docstrings", {"strip_comments": True, "strip_docstrings": True, "collapse_blank_runs": False}),
|
|
106
|
+
("Strip all + collapse blanks", {"strip_comments": True, "strip_docstrings": True, "collapse_blank_runs": True}),
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class Pipeline:
|
|
111
|
+
"""Stateful across calls so repeated estimates stay cheap."""
|
|
112
|
+
|
|
113
|
+
def __init__(self) -> None:
|
|
114
|
+
self._discovery: dict[str, DiscoveryResult] = {}
|
|
115
|
+
self._text: dict[str, tuple[str, str, list[str]]] = {}
|
|
116
|
+
self._strip: dict[tuple[str, str], StripResult] = {}
|
|
117
|
+
self._scan: dict[tuple[str, str], tuple[list[Finding], list[Finding]]] = {}
|
|
118
|
+
|
|
119
|
+
# -- stages ------------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
def discover(self, settings: BuildSettings, refresh: bool = False) -> DiscoveryResult:
|
|
122
|
+
key = settings.source_root + "|" + str(dataclasses.asdict(settings.discovery))
|
|
123
|
+
if refresh or key not in self._discovery:
|
|
124
|
+
self._discovery[key] = discover(settings.source_root, settings.discovery)
|
|
125
|
+
return self._discovery[key]
|
|
126
|
+
|
|
127
|
+
def read(self, file: DiscoveredFile) -> tuple[str, str, list[str]]:
|
|
128
|
+
cached = self._text.get(file.sha256)
|
|
129
|
+
if cached is None:
|
|
130
|
+
decoded = read_source(file.abs_path)
|
|
131
|
+
cached = (decoded.text, decoded.encoding, list(decoded.warnings))
|
|
132
|
+
self._text[file.sha256] = cached
|
|
133
|
+
return cached
|
|
134
|
+
|
|
135
|
+
def strip(self, file: DiscoveredFile, text: str, options: TransformOptions) -> StripResult:
|
|
136
|
+
key = (file.sha256, options.cache_key())
|
|
137
|
+
result = self._strip.get(key)
|
|
138
|
+
if result is None:
|
|
139
|
+
result = strip_source(text, file.rel_path, options)
|
|
140
|
+
self._strip[key] = result
|
|
141
|
+
return result
|
|
142
|
+
|
|
143
|
+
def scan(
|
|
144
|
+
self,
|
|
145
|
+
file: DiscoveredFile,
|
|
146
|
+
text: str,
|
|
147
|
+
owner: str,
|
|
148
|
+
settings: BuildSettings,
|
|
149
|
+
comment_spans: list,
|
|
150
|
+
) -> tuple[list[Finding], list[Finding]]:
|
|
151
|
+
key = (file.sha256, owner)
|
|
152
|
+
cached = self._scan.get(key)
|
|
153
|
+
if cached is None:
|
|
154
|
+
# Secrets hide in code, so that scan reads everything; licence
|
|
155
|
+
# and authorship markers live in comments, so that scan reads
|
|
156
|
+
# only the comment layer and stays free of prose false hits.
|
|
157
|
+
found_secrets = secrets_mod.scan_text(file.rel_path, text) if settings.scan.scan_secrets else []
|
|
158
|
+
found_third: list[Finding] = []
|
|
159
|
+
if settings.scan.scan_third_party:
|
|
160
|
+
comment_lines = comments_only_lines(text, comment_spans)
|
|
161
|
+
found_third = thirdparty_mod.scan_text(file.rel_path, comment_lines, owner)
|
|
162
|
+
cached = (found_secrets, found_third)
|
|
163
|
+
self._scan[key] = cached
|
|
164
|
+
return cached
|
|
165
|
+
|
|
166
|
+
# -- estimate ----------------------------------------------------------
|
|
167
|
+
|
|
168
|
+
def estimate(
|
|
169
|
+
self,
|
|
170
|
+
settings: BuildSettings,
|
|
171
|
+
*,
|
|
172
|
+
transform: TransformOptions | None = None,
|
|
173
|
+
progress: Progress = _noop,
|
|
174
|
+
refresh: bool = False,
|
|
175
|
+
run_scans: bool = True,
|
|
176
|
+
) -> EstimateResult:
|
|
177
|
+
transform = transform or settings.transform
|
|
178
|
+
warnings: list[str] = []
|
|
179
|
+
|
|
180
|
+
progress("Discovering files", 0, 1)
|
|
181
|
+
found = self.discover(settings, refresh=refresh)
|
|
182
|
+
warnings.extend(found.warnings)
|
|
183
|
+
|
|
184
|
+
plan = ordering.resolve_order(
|
|
185
|
+
settings.order_entries,
|
|
186
|
+
found.files,
|
|
187
|
+
include_unlisted=settings.include_unlisted,
|
|
188
|
+
excluded=settings.excluded,
|
|
189
|
+
)
|
|
190
|
+
warnings.extend(plan.warnings)
|
|
191
|
+
for entry in plan.unmatched:
|
|
192
|
+
warnings.append(f"Order entry '{entry}' matched no file.")
|
|
193
|
+
|
|
194
|
+
by_path = found.by_path()
|
|
195
|
+
ordered_files = [by_path[p] for p in plan.paths if p in by_path]
|
|
196
|
+
|
|
197
|
+
compiled, regex_warnings = redaction.compile_regexes(settings.redaction)
|
|
198
|
+
warnings.extend(regex_warnings)
|
|
199
|
+
manual = {m.replace("\\", "/") for m in settings.redaction.manual_lines}
|
|
200
|
+
|
|
201
|
+
scan_report = ScanReport()
|
|
202
|
+
redaction_report = redaction.RedactionReport()
|
|
203
|
+
prepared: list[PreparedFile] = []
|
|
204
|
+
|
|
205
|
+
total = len(ordered_files)
|
|
206
|
+
for index, file in enumerate(ordered_files, start=1):
|
|
207
|
+
progress("Reading and stripping", index, total)
|
|
208
|
+
text, encoding, read_warnings = self.read(file)
|
|
209
|
+
stripped = self.strip(file, text, transform)
|
|
210
|
+
|
|
211
|
+
# Line selection happens after stripping (so ranges refer to the
|
|
212
|
+
# editor's line numbers) and before redaction (so redaction
|
|
213
|
+
# indices line up with what is actually rendered).
|
|
214
|
+
range_spec = settings.line_ranges.get(file.rel_path, "")
|
|
215
|
+
ranges, range_warnings = lineranges.parse_ranges(range_spec, file.raw_line_count)
|
|
216
|
+
selection = lineranges.apply_selection(
|
|
217
|
+
stripped.lines, ranges, file.raw_line_count
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
file_redaction = redaction.compute_file_redaction(
|
|
221
|
+
file.rel_path, text, selection.lines, settings.redaction, compiled, manual
|
|
222
|
+
)
|
|
223
|
+
redaction_report.files.append(file_redaction)
|
|
224
|
+
|
|
225
|
+
if run_scans:
|
|
226
|
+
found_secrets, found_third = self.scan(
|
|
227
|
+
file,
|
|
228
|
+
text,
|
|
229
|
+
settings.header.copyright_owner,
|
|
230
|
+
settings,
|
|
231
|
+
stripped.comment_spans,
|
|
232
|
+
)
|
|
233
|
+
scan_report.secrets.extend(found_secrets)
|
|
234
|
+
scan_report.third_party.extend(found_third)
|
|
235
|
+
|
|
236
|
+
file_warnings = (
|
|
237
|
+
list(read_warnings)
|
|
238
|
+
+ list(stripped.warnings)
|
|
239
|
+
+ range_warnings
|
|
240
|
+
+ selection.warnings
|
|
241
|
+
)
|
|
242
|
+
for message in file_warnings:
|
|
243
|
+
warnings.append(f"{file.rel_path}: {message}")
|
|
244
|
+
|
|
245
|
+
prepared.append(
|
|
246
|
+
PreparedFile(
|
|
247
|
+
rel_path=file.rel_path,
|
|
248
|
+
abs_path=file.abs_path,
|
|
249
|
+
original_text=text,
|
|
250
|
+
strip=stripped,
|
|
251
|
+
redaction=file_redaction,
|
|
252
|
+
selection=selection,
|
|
253
|
+
range_spec=range_spec,
|
|
254
|
+
total_lines=file.raw_line_count,
|
|
255
|
+
language=file.language,
|
|
256
|
+
encoding=encoding,
|
|
257
|
+
sha256=file.sha256,
|
|
258
|
+
warnings=file_warnings,
|
|
259
|
+
)
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
progress("Laying out pages", total, total)
|
|
263
|
+
blocks = [
|
|
264
|
+
FileBlock(
|
|
265
|
+
rel_path=p.rel_path,
|
|
266
|
+
lines=p.selection.lines,
|
|
267
|
+
language=p.language,
|
|
268
|
+
redactions=p.redaction.spans,
|
|
269
|
+
elisions=p.selection.elisions,
|
|
270
|
+
trailing_elision=p.selection.trailing,
|
|
271
|
+
range_label=p.selection.label,
|
|
272
|
+
)
|
|
273
|
+
for p in prepared
|
|
274
|
+
]
|
|
275
|
+
|
|
276
|
+
geometry = metrics.build_geometry(
|
|
277
|
+
settings.layout, layout_mod.max_source_line_number(blocks)
|
|
278
|
+
)
|
|
279
|
+
warnings.extend(geometry.font.warnings)
|
|
280
|
+
|
|
281
|
+
result_layout = layout_mod.build_layout(
|
|
282
|
+
blocks, settings.header.lines(), settings.layout, geometry
|
|
283
|
+
)
|
|
284
|
+
warnings.extend(result_layout.warnings)
|
|
285
|
+
|
|
286
|
+
selection = deposit_mod.select_pages(result_layout.page_count, settings.deposit)
|
|
287
|
+
warnings.extend(redaction.check_compliance(redaction_report, selection.mode))
|
|
288
|
+
|
|
289
|
+
return EstimateResult(
|
|
290
|
+
settings=settings,
|
|
291
|
+
discovery=found,
|
|
292
|
+
plan=plan,
|
|
293
|
+
layout=result_layout,
|
|
294
|
+
selection=selection,
|
|
295
|
+
scan=scan_report,
|
|
296
|
+
redaction_report=redaction_report,
|
|
297
|
+
prepared=prepared,
|
|
298
|
+
warnings=warnings,
|
|
299
|
+
geometry_note=metrics.describe(geometry),
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
def what_if(
|
|
303
|
+
self,
|
|
304
|
+
settings: BuildSettings,
|
|
305
|
+
progress: Progress = _noop,
|
|
306
|
+
) -> list[tuple[str, int, str]]:
|
|
307
|
+
"""Page count under each comment policy.
|
|
308
|
+
|
|
309
|
+
Answers the question that actually matters: what do I change to get
|
|
310
|
+
under 50 pages and deposit the whole program?
|
|
311
|
+
"""
|
|
312
|
+
rows: list[tuple[str, int, str]] = []
|
|
313
|
+
for index, (label, overrides) in enumerate(WHAT_IF_VARIANTS, start=1):
|
|
314
|
+
progress("Comparing policies", index, len(WHAT_IF_VARIANTS))
|
|
315
|
+
variant = dataclasses.replace(settings.transform, **overrides)
|
|
316
|
+
estimate = self.estimate(
|
|
317
|
+
settings, transform=variant, progress=_noop, run_scans=False
|
|
318
|
+
)
|
|
319
|
+
mode = (
|
|
320
|
+
"entire program"
|
|
321
|
+
if estimate.selection.mode == deposit_mod.MODE_ENTIRE
|
|
322
|
+
else "first 25 + last 25"
|
|
323
|
+
)
|
|
324
|
+
rows.append((label, estimate.page_count, mode))
|
|
325
|
+
return rows
|
|
326
|
+
|
|
327
|
+
# -- build -------------------------------------------------------------
|
|
328
|
+
|
|
329
|
+
def build(
|
|
330
|
+
self,
|
|
331
|
+
settings: BuildSettings,
|
|
332
|
+
*,
|
|
333
|
+
progress: Progress = _noop,
|
|
334
|
+
force: bool = False,
|
|
335
|
+
) -> BuildResult:
|
|
336
|
+
from . import manifest as manifest_mod
|
|
337
|
+
from .render import render_pdf
|
|
338
|
+
|
|
339
|
+
estimate = self.estimate(settings, progress=progress)
|
|
340
|
+
result = BuildResult(estimate=estimate)
|
|
341
|
+
|
|
342
|
+
ignored = set(settings.scan.ignored_findings)
|
|
343
|
+
blocking = estimate.scan.blocking(ignored)
|
|
344
|
+
if blocking and settings.scan.block_on_secrets and not force:
|
|
345
|
+
result.blocked_by = blocking
|
|
346
|
+
return result
|
|
347
|
+
|
|
348
|
+
out_dir = Path(settings.output_dir or ".")
|
|
349
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
350
|
+
base = settings.output_basename or "deposit"
|
|
351
|
+
fingerprint = settings.content_fingerprint()
|
|
352
|
+
|
|
353
|
+
if settings.write_full_pdf:
|
|
354
|
+
progress("Rendering complete PDF", 1, 2)
|
|
355
|
+
full_path = out_dir / f"{base}_full.pdf"
|
|
356
|
+
render_pdf(
|
|
357
|
+
estimate.layout,
|
|
358
|
+
full_path,
|
|
359
|
+
settings.header,
|
|
360
|
+
settings.layout,
|
|
361
|
+
fingerprint=fingerprint,
|
|
362
|
+
)
|
|
363
|
+
result.outputs.append(str(full_path))
|
|
364
|
+
|
|
365
|
+
if settings.write_deposit_pdf:
|
|
366
|
+
progress("Rendering deposit copy", 2, 2)
|
|
367
|
+
deposit_path = out_dir / f"{base}_deposit.pdf"
|
|
368
|
+
render_pdf(
|
|
369
|
+
estimate.layout,
|
|
370
|
+
deposit_path,
|
|
371
|
+
settings.header,
|
|
372
|
+
settings.layout,
|
|
373
|
+
selection=estimate.selection,
|
|
374
|
+
fingerprint=fingerprint,
|
|
375
|
+
include_separator=settings.deposit.separator_page,
|
|
376
|
+
)
|
|
377
|
+
result.outputs.append(str(deposit_path))
|
|
378
|
+
|
|
379
|
+
if settings.write_manifest:
|
|
380
|
+
manifest_path, summary_path = manifest_mod.write_manifest(
|
|
381
|
+
estimate, out_dir, base, result.outputs
|
|
382
|
+
)
|
|
383
|
+
result.manifest_path = manifest_path
|
|
384
|
+
result.summary_path = summary_path
|
|
385
|
+
|
|
386
|
+
return result
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Trade-secret redaction (Compendium section 721.7).
|
|
2
|
+
|
|
3
|
+
Blocked-out material is drawn as a solid bar and the underlying glyphs are
|
|
4
|
+
never written into the PDF, so the text cannot be recovered by selecting or
|
|
5
|
+
extracting it. Redaction never changes the number of lines, so it cannot
|
|
6
|
+
shift pagination.
|
|
7
|
+
|
|
8
|
+
Region markers are located in the *original* text and tracked by original
|
|
9
|
+
line number, because the comment carrying the marker is usually removed by
|
|
10
|
+
the stripper before layout ever sees it.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
|
|
18
|
+
from ..config import MAX_REDACTION_RATIO, RedactionRules
|
|
19
|
+
from .strip import SourceLine
|
|
20
|
+
|
|
21
|
+
Span = tuple[int, int]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class FileRedaction:
|
|
26
|
+
rel_path: str
|
|
27
|
+
spans: dict[int, list[Span]] = field(default_factory=dict) # index into lines
|
|
28
|
+
redacted_chars: int = 0
|
|
29
|
+
total_chars: int = 0
|
|
30
|
+
marker_lines: int = 0
|
|
31
|
+
regex_hits: int = 0
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class RedactionReport:
|
|
36
|
+
files: list[FileRedaction] = field(default_factory=list)
|
|
37
|
+
warnings: list[str] = field(default_factory=list)
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def redacted_chars(self) -> int:
|
|
41
|
+
return sum(f.redacted_chars for f in self.files)
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def total_chars(self) -> int:
|
|
45
|
+
return sum(f.total_chars for f in self.files)
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def ratio(self) -> float:
|
|
49
|
+
return self.redacted_chars / self.total_chars if self.total_chars else 0.0
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def compliant(self) -> bool:
|
|
53
|
+
return self.ratio <= MAX_REDACTION_RATIO
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _merge(spans: list[Span]) -> list[Span]:
|
|
57
|
+
if not spans:
|
|
58
|
+
return []
|
|
59
|
+
spans = sorted(spans)
|
|
60
|
+
out = [spans[0]]
|
|
61
|
+
for a, b in spans[1:]:
|
|
62
|
+
last_a, last_b = out[-1]
|
|
63
|
+
if a <= last_b:
|
|
64
|
+
out[-1] = (last_a, max(last_b, b))
|
|
65
|
+
else:
|
|
66
|
+
out.append((a, b))
|
|
67
|
+
return out
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def marker_line_numbers(original_text: str, rules: RedactionRules) -> set[int]:
|
|
71
|
+
"""Original line numbers enclosed by BEGIN/END markers."""
|
|
72
|
+
if not rules.begin_marker or not rules.end_marker:
|
|
73
|
+
return set()
|
|
74
|
+
inside = False
|
|
75
|
+
result: set[int] = set()
|
|
76
|
+
for number, line in enumerate(original_text.split("\n"), start=1):
|
|
77
|
+
if rules.begin_marker in line:
|
|
78
|
+
inside = True
|
|
79
|
+
continue
|
|
80
|
+
if rules.end_marker in line:
|
|
81
|
+
inside = False
|
|
82
|
+
continue
|
|
83
|
+
if inside:
|
|
84
|
+
result.add(number)
|
|
85
|
+
return result
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def compile_regexes(rules: RedactionRules) -> tuple[list[re.Pattern[str]], list[str]]:
|
|
89
|
+
compiled: list[re.Pattern[str]] = []
|
|
90
|
+
warnings: list[str] = []
|
|
91
|
+
for pattern in rules.regexes:
|
|
92
|
+
if not pattern.strip():
|
|
93
|
+
continue
|
|
94
|
+
try:
|
|
95
|
+
compiled.append(re.compile(pattern))
|
|
96
|
+
except re.error as exc:
|
|
97
|
+
warnings.append(f"Invalid redaction pattern '{pattern}': {exc}")
|
|
98
|
+
return compiled, warnings
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def compute_file_redaction(
|
|
102
|
+
rel_path: str,
|
|
103
|
+
original_text: str,
|
|
104
|
+
lines: list[SourceLine],
|
|
105
|
+
rules: RedactionRules,
|
|
106
|
+
compiled: list[re.Pattern[str]] | None = None,
|
|
107
|
+
manual: set[str] | None = None,
|
|
108
|
+
) -> FileRedaction:
|
|
109
|
+
out = FileRedaction(rel_path=rel_path)
|
|
110
|
+
out.total_chars = sum(len(line.text.strip()) for line in lines)
|
|
111
|
+
if not rules.enabled:
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
if compiled is None:
|
|
115
|
+
compiled, _ = compile_regexes(rules)
|
|
116
|
+
if manual is None:
|
|
117
|
+
manual = {m.replace("\\", "/") for m in rules.manual_lines}
|
|
118
|
+
|
|
119
|
+
markers = marker_line_numbers(original_text, rules)
|
|
120
|
+
|
|
121
|
+
for index, line in enumerate(lines):
|
|
122
|
+
text = line.text
|
|
123
|
+
if not text.strip():
|
|
124
|
+
continue
|
|
125
|
+
spans: list[Span] = []
|
|
126
|
+
whole = line.number in markers or f"{rel_path}:{line.number}" in manual
|
|
127
|
+
if whole:
|
|
128
|
+
start = len(text) - len(text.lstrip())
|
|
129
|
+
spans.append((start, len(text)))
|
|
130
|
+
out.marker_lines += 1
|
|
131
|
+
else:
|
|
132
|
+
for pattern in compiled:
|
|
133
|
+
for match in pattern.finditer(text):
|
|
134
|
+
if match.end() > match.start():
|
|
135
|
+
spans.append((match.start(), match.end()))
|
|
136
|
+
out.regex_hits += 1
|
|
137
|
+
if spans:
|
|
138
|
+
merged = _merge(spans)
|
|
139
|
+
out.spans[index] = merged
|
|
140
|
+
out.redacted_chars += sum(b - a for a, b in merged)
|
|
141
|
+
return out
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def check_compliance(report: RedactionReport, deposit_mode: str) -> list[str]:
|
|
145
|
+
"""Warn when blocked-out material exceeds what section 721.7 allows."""
|
|
146
|
+
messages: list[str] = []
|
|
147
|
+
if report.redacted_chars == 0:
|
|
148
|
+
return messages
|
|
149
|
+
percent = report.ratio * 100
|
|
150
|
+
if not report.compliant:
|
|
151
|
+
messages.append(
|
|
152
|
+
f"Redaction covers {percent:.1f}% of the deposited text. "
|
|
153
|
+
f"Compendium sec. 721.7 caps blocked-out material at "
|
|
154
|
+
f"{MAX_REDACTION_RATIO * 100:.0f}% for this deposit option; "
|
|
155
|
+
"reduce the redacted regions or deposit more pages."
|
|
156
|
+
)
|
|
157
|
+
else:
|
|
158
|
+
messages.append(
|
|
159
|
+
f"Redaction covers {percent:.1f}% of the deposited text "
|
|
160
|
+
f"(within the {MAX_REDACTION_RATIO * 100:.0f}% limit)."
|
|
161
|
+
)
|
|
162
|
+
return messages
|