codendium 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codendium-1.0.0.dist-info/METADATA +332 -0
- codendium-1.0.0.dist-info/RECORD +45 -0
- codendium-1.0.0.dist-info/WHEEL +5 -0
- codendium-1.0.0.dist-info/entry_points.txt +6 -0
- codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
- codendium-1.0.0.dist-info/top_level.txt +1 -0
- copyright_deposit/__init__.py +15 -0
- copyright_deposit/__main__.py +34 -0
- copyright_deposit/assets/__init__.py +5 -0
- copyright_deposit/assets/fonts/README.md +31 -0
- copyright_deposit/assets/logo.svg +26 -0
- copyright_deposit/cli.py +314 -0
- copyright_deposit/config.py +269 -0
- copyright_deposit/core/__init__.py +1 -0
- copyright_deposit/core/deposit.py +117 -0
- copyright_deposit/core/discovery.py +260 -0
- copyright_deposit/core/encoding.py +136 -0
- copyright_deposit/core/languages.py +190 -0
- copyright_deposit/core/layout.py +349 -0
- copyright_deposit/core/lineranges.py +219 -0
- copyright_deposit/core/manifest.py +282 -0
- copyright_deposit/core/metrics.py +279 -0
- copyright_deposit/core/ordering.py +349 -0
- copyright_deposit/core/pipeline.py +386 -0
- copyright_deposit/core/redaction.py +162 -0
- copyright_deposit/core/render.py +242 -0
- copyright_deposit/core/scanning/__init__.py +61 -0
- copyright_deposit/core/scanning/secrets.py +181 -0
- copyright_deposit/core/scanning/thirdparty.py +190 -0
- copyright_deposit/core/strip/__init__.py +337 -0
- copyright_deposit/core/strip/cfamily_strip.py +235 -0
- copyright_deposit/core/strip/pygments_strip.py +85 -0
- copyright_deposit/core/strip/python_strip.py +131 -0
- copyright_deposit/gui/__init__.py +1 -0
- copyright_deposit/gui/app.py +34 -0
- copyright_deposit/gui/branding.py +83 -0
- copyright_deposit/gui/history.py +192 -0
- copyright_deposit/gui/main_window.py +617 -0
- copyright_deposit/gui/panels/__init__.py +1 -0
- copyright_deposit/gui/panels/estimate.py +166 -0
- copyright_deposit/gui/panels/files.py +635 -0
- copyright_deposit/gui/panels/identification.py +193 -0
- copyright_deposit/gui/panels/options.py +445 -0
- copyright_deposit/gui/panels/preflight.py +260 -0
- copyright_deposit/gui/workers.py +96 -0
|
@@ -0,0 +1,337 @@
|
|
|
1
|
+
"""Comment and docstring removal.
|
|
2
|
+
|
|
3
|
+
Design
|
|
4
|
+
------
|
|
5
|
+
Strippers never rewrite code. They *classify* byte ranges of the original
|
|
6
|
+
text as comment/docstring, and a single shared routine removes those
|
|
7
|
+
ranges. Removal is therefore provably subtractive: nothing that was not
|
|
8
|
+
classified as a comment can ever be altered.
|
|
9
|
+
|
|
10
|
+
Dispatch, in order of trustworthiness:
|
|
11
|
+
|
|
12
|
+
* Python -> ``tokenize`` + ``ast``. Authoritative, and the result is
|
|
13
|
+
checked by comparing the ASTs of the original and stripped source.
|
|
14
|
+
* Others -> pygments, whose token stream is verified to reconstruct the
|
|
15
|
+
source exactly before it is trusted.
|
|
16
|
+
* Fallback -> a hand-written C-family state machine (also used when no
|
|
17
|
+
lexer exists or the pygments round-trip fails).
|
|
18
|
+
|
|
19
|
+
The pygments path deliberately keeps ``Comment.Preproc`` tokens: C lexers
|
|
20
|
+
classify ``#include`` and ``#define`` as comments, and deleting those
|
|
21
|
+
would silently gut the deposit.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import re
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
|
|
29
|
+
from ...config import LEGAL_HEADER_PATTERN, TransformOptions
|
|
30
|
+
from .. import languages
|
|
31
|
+
from ..languages import FAMILY_C, FAMILY_HASH, FAMILY_PYTHON, Language
|
|
32
|
+
|
|
33
|
+
KIND_COMMENT = "comment"
|
|
34
|
+
KIND_DOCSTRING = "docstring"
|
|
35
|
+
KIND_SHEBANG = "shebang"
|
|
36
|
+
|
|
37
|
+
Span = tuple[int, int, str] # (start offset, end offset, kind)
|
|
38
|
+
|
|
39
|
+
_LEGAL_RE = re.compile(LEGAL_HEADER_PATTERN, re.IGNORECASE)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class SourceLine:
|
|
44
|
+
"""One physical line of the stripped source."""
|
|
45
|
+
|
|
46
|
+
number: int # 1-based line number in the ORIGINAL file
|
|
47
|
+
text: str
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class StripResult:
|
|
52
|
+
lines: list[SourceLine] = field(default_factory=list)
|
|
53
|
+
method: str = "none"
|
|
54
|
+
removed_lines: int = 0
|
|
55
|
+
warnings: list[str] = field(default_factory=list)
|
|
56
|
+
# Every classified comment/docstring span, before the keep-policy
|
|
57
|
+
# toggles are applied. The third-party scanner reads these so it only
|
|
58
|
+
# inspects real comments instead of matching prose inside code.
|
|
59
|
+
comment_spans: list[Span] = field(default_factory=list)
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def line_count(self) -> int:
|
|
63
|
+
return len(self.lines)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# ---------------------------------------------------------------------------
|
|
67
|
+
# Offset helpers
|
|
68
|
+
# ---------------------------------------------------------------------------
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def line_start_offsets(text: str) -> list[int]:
|
|
72
|
+
"""Offset of the first character of each 1-based line (index 0 unused)."""
|
|
73
|
+
starts = [0, 0]
|
|
74
|
+
for i, ch in enumerate(text):
|
|
75
|
+
if ch == "\n":
|
|
76
|
+
starts.append(i + 1)
|
|
77
|
+
return starts
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def rowcol_to_offset(starts: list[int], row: int, col: int) -> int:
|
|
81
|
+
if row < 1 or row >= len(starts):
|
|
82
|
+
return len(starts) and starts[-1]
|
|
83
|
+
return starts[row] + col
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# ---------------------------------------------------------------------------
|
|
87
|
+
# Span post-processing
|
|
88
|
+
# ---------------------------------------------------------------------------
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def merge_spans(spans: list[Span]) -> list[Span]:
|
|
92
|
+
"""Sort and coalesce overlapping spans."""
|
|
93
|
+
if not spans:
|
|
94
|
+
return []
|
|
95
|
+
spans = sorted(spans, key=lambda s: (s[0], s[1]))
|
|
96
|
+
merged: list[Span] = [spans[0]]
|
|
97
|
+
for start, end, kind in spans[1:]:
|
|
98
|
+
last_start, last_end, last_kind = merged[-1]
|
|
99
|
+
if start <= last_end:
|
|
100
|
+
merged[-1] = (last_start, max(last_end, end), last_kind or kind)
|
|
101
|
+
else:
|
|
102
|
+
merged.append((start, end, kind))
|
|
103
|
+
return merged
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _filter_spans(text: str, spans: list[Span], options: TransformOptions) -> list[Span]:
|
|
107
|
+
"""Apply the keep-policy toggles to classified spans."""
|
|
108
|
+
kept: list[Span] = []
|
|
109
|
+
for start, end, kind in spans:
|
|
110
|
+
if kind == KIND_SHEBANG:
|
|
111
|
+
if options.preserve_shebang:
|
|
112
|
+
continue
|
|
113
|
+
if not options.strip_comments:
|
|
114
|
+
continue
|
|
115
|
+
kept.append((start, end, kind))
|
|
116
|
+
continue
|
|
117
|
+
if kind == KIND_DOCSTRING:
|
|
118
|
+
if options.strip_docstrings:
|
|
119
|
+
kept.append((start, end, kind))
|
|
120
|
+
continue
|
|
121
|
+
if options.strip_comments:
|
|
122
|
+
kept.append((start, end, kind))
|
|
123
|
+
if not options.preserve_legal_headers:
|
|
124
|
+
return kept
|
|
125
|
+
return _keep_legal_header(text, kept)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _keep_legal_header(text: str, spans: list[Span]) -> list[Span]:
|
|
129
|
+
"""Drop the leading comment block from the kill list if it reads as legal.
|
|
130
|
+
|
|
131
|
+
Only comments appearing before the first line of real code qualify: a
|
|
132
|
+
licence notice buried mid-file is not the header the Office asks for.
|
|
133
|
+
"""
|
|
134
|
+
if not spans:
|
|
135
|
+
return spans
|
|
136
|
+
result: list[Span] = []
|
|
137
|
+
for index, (start, end, kind) in enumerate(spans):
|
|
138
|
+
if kind == KIND_DOCSTRING and index > 0:
|
|
139
|
+
result.append((start, end, kind))
|
|
140
|
+
continue
|
|
141
|
+
before = text[:start]
|
|
142
|
+
# "Leading" means only whitespace and other comments precede it.
|
|
143
|
+
stripped_before = before
|
|
144
|
+
for prev_start, prev_end, _ in spans[:index]:
|
|
145
|
+
stripped_before = (
|
|
146
|
+
stripped_before[:prev_start] + " " * (prev_end - prev_start) + stripped_before[prev_end:]
|
|
147
|
+
)
|
|
148
|
+
if stripped_before.strip():
|
|
149
|
+
result.append((start, end, kind))
|
|
150
|
+
continue
|
|
151
|
+
if _LEGAL_RE.search(text[start:end]):
|
|
152
|
+
continue # preserved
|
|
153
|
+
result.append((start, end, kind))
|
|
154
|
+
return result
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
# ---------------------------------------------------------------------------
|
|
158
|
+
# Removal
|
|
159
|
+
# ---------------------------------------------------------------------------
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def apply_spans(text: str, spans: list[Span]) -> list[SourceLine]:
|
|
163
|
+
"""Remove the spans and rebuild lines, tracking original line numbers.
|
|
164
|
+
|
|
165
|
+
Removal joins fragments across a multi-line comment, which is what a
|
|
166
|
+
reader expects: ``int x = /* note\\n more */ 5;`` becomes ``int x = 5;``.
|
|
167
|
+
"""
|
|
168
|
+
spans = merge_spans(spans)
|
|
169
|
+
lines: list[SourceLine] = []
|
|
170
|
+
buf: list[str] = []
|
|
171
|
+
orig_line = 1
|
|
172
|
+
line_started_at = 1
|
|
173
|
+
span_index = 0
|
|
174
|
+
i = 0
|
|
175
|
+
length = len(text)
|
|
176
|
+
|
|
177
|
+
while i < length:
|
|
178
|
+
if span_index < len(spans) and i == spans[span_index][0]:
|
|
179
|
+
_, end, _ = spans[span_index]
|
|
180
|
+
# Count newlines swallowed by the comment so numbering stays true.
|
|
181
|
+
orig_line += text.count("\n", i, end)
|
|
182
|
+
i = end
|
|
183
|
+
span_index += 1
|
|
184
|
+
continue
|
|
185
|
+
ch = text[i]
|
|
186
|
+
if ch == "\n":
|
|
187
|
+
lines.append(SourceLine(line_started_at, "".join(buf)))
|
|
188
|
+
buf = []
|
|
189
|
+
orig_line += 1
|
|
190
|
+
line_started_at = orig_line
|
|
191
|
+
i += 1
|
|
192
|
+
continue
|
|
193
|
+
buf.append(ch)
|
|
194
|
+
i += 1
|
|
195
|
+
|
|
196
|
+
if buf:
|
|
197
|
+
lines.append(SourceLine(line_started_at, "".join(buf)))
|
|
198
|
+
return lines
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def comments_only_lines(text: str, spans: list[Span]) -> list[str]:
|
|
202
|
+
"""The file with every non-comment character blanked out.
|
|
203
|
+
|
|
204
|
+
Line numbering is preserved exactly, so a scanner can report a hit at
|
|
205
|
+
its true location while never seeing a character of actual code.
|
|
206
|
+
"""
|
|
207
|
+
keep = bytearray(len(text))
|
|
208
|
+
for start, end, _kind in spans:
|
|
209
|
+
for i in range(max(0, start), min(len(text), end)):
|
|
210
|
+
keep[i] = 1
|
|
211
|
+
out = [
|
|
212
|
+
ch if (ch == "\n" or keep[i]) else " "
|
|
213
|
+
for i, ch in enumerate(text)
|
|
214
|
+
]
|
|
215
|
+
return "".join(out).split("\n")
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _blank_original_lines(text: str) -> set[int]:
|
|
219
|
+
return {n for n, raw in enumerate(text.split("\n"), start=1) if not raw.strip()}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def postprocess(
|
|
223
|
+
lines: list[SourceLine],
|
|
224
|
+
original_text: str,
|
|
225
|
+
options: TransformOptions,
|
|
226
|
+
) -> list[SourceLine]:
|
|
227
|
+
"""Tabs, trailing whitespace, comment-only line removal, blank collapsing."""
|
|
228
|
+
originally_blank = _blank_original_lines(original_text)
|
|
229
|
+
|
|
230
|
+
staged: list[SourceLine] = []
|
|
231
|
+
for line in lines:
|
|
232
|
+
text = line.text
|
|
233
|
+
if options.expand_tabs:
|
|
234
|
+
text = text.expandtabs(options.tab_width)
|
|
235
|
+
if options.trim_trailing_whitespace:
|
|
236
|
+
text = text.rstrip()
|
|
237
|
+
if not text.strip() and line.number not in originally_blank:
|
|
238
|
+
# The line held nothing but a comment; drop it entirely.
|
|
239
|
+
continue
|
|
240
|
+
staged.append(SourceLine(line.number, text))
|
|
241
|
+
|
|
242
|
+
if options.collapse_blank_runs:
|
|
243
|
+
collapsed: list[SourceLine] = []
|
|
244
|
+
run = 0
|
|
245
|
+
for line in staged:
|
|
246
|
+
if line.text.strip():
|
|
247
|
+
run = 0
|
|
248
|
+
collapsed.append(line)
|
|
249
|
+
continue
|
|
250
|
+
run += 1
|
|
251
|
+
if run <= max(0, options.max_blank_run):
|
|
252
|
+
collapsed.append(line)
|
|
253
|
+
staged = collapsed
|
|
254
|
+
|
|
255
|
+
while staged and not staged[0].text.strip():
|
|
256
|
+
staged.pop(0)
|
|
257
|
+
while staged and not staged[-1].text.strip():
|
|
258
|
+
staged.pop()
|
|
259
|
+
return staged
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
# ---------------------------------------------------------------------------
|
|
263
|
+
# Dispatch
|
|
264
|
+
# ---------------------------------------------------------------------------
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def find_spans(text: str, language: Language) -> tuple[list[Span], str, list[str]]:
|
|
268
|
+
"""Classify comment/docstring ranges. Returns (spans, method, warnings)."""
|
|
269
|
+
from . import cfamily_strip, pygments_strip, python_strip
|
|
270
|
+
|
|
271
|
+
if language.family == FAMILY_PYTHON:
|
|
272
|
+
spans, warnings = python_strip.find_spans(text)
|
|
273
|
+
if spans is not None:
|
|
274
|
+
return spans, "python-tokenize", warnings
|
|
275
|
+
|
|
276
|
+
spans, warnings = pygments_strip.find_spans(text, language)
|
|
277
|
+
if spans is not None:
|
|
278
|
+
return spans, "pygments", warnings
|
|
279
|
+
|
|
280
|
+
if language.family == FAMILY_C:
|
|
281
|
+
spans, w2 = cfamily_strip.find_spans(text, language)
|
|
282
|
+
return spans, "c-state-machine", warnings + w2
|
|
283
|
+
|
|
284
|
+
if language.family in (FAMILY_HASH, FAMILY_PYTHON) or language.line_comments:
|
|
285
|
+
spans, w2 = cfamily_strip.find_spans(text, language)
|
|
286
|
+
return spans, "line-scanner", warnings + w2
|
|
287
|
+
|
|
288
|
+
return [], "none", warnings + ["No comment syntax known; source kept verbatim."]
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def strip_source(
|
|
292
|
+
text: str,
|
|
293
|
+
path: str,
|
|
294
|
+
options: TransformOptions,
|
|
295
|
+
language: Language | None = None,
|
|
296
|
+
) -> StripResult:
|
|
297
|
+
"""Strip `text` according to `options`, returning renderable lines."""
|
|
298
|
+
language = language or languages.detect(path)
|
|
299
|
+
result = StripResult()
|
|
300
|
+
|
|
301
|
+
# Spans are always classified, even when nothing will be removed: the
|
|
302
|
+
# third-party scanner needs to know which regions are comments.
|
|
303
|
+
all_spans, method, warnings = find_spans(text, language)
|
|
304
|
+
all_spans = merge_spans(all_spans)
|
|
305
|
+
result.comment_spans = all_spans
|
|
306
|
+
result.method = method
|
|
307
|
+
result.warnings.extend(warnings)
|
|
308
|
+
|
|
309
|
+
if not (options.strip_comments or options.strip_docstrings):
|
|
310
|
+
result.method = "verbatim"
|
|
311
|
+
raw_lines = [SourceLine(n, t) for n, t in enumerate(text.split("\n"), start=1)]
|
|
312
|
+
result.lines = postprocess(raw_lines, text, options)
|
|
313
|
+
return result
|
|
314
|
+
|
|
315
|
+
spans = _filter_spans(text, all_spans, options)
|
|
316
|
+
stripped_lines = apply_spans(text, spans)
|
|
317
|
+
|
|
318
|
+
# Safety net: for Python the transform must not change the parse tree.
|
|
319
|
+
if language.family == FAMILY_PYTHON and spans:
|
|
320
|
+
from . import python_strip
|
|
321
|
+
|
|
322
|
+
rebuilt = "\n".join(line.text for line in stripped_lines)
|
|
323
|
+
ok, message = python_strip.verify_equivalent(text, rebuilt, options)
|
|
324
|
+
if not ok:
|
|
325
|
+
result.warnings.append(
|
|
326
|
+
f"Comment removal changed the parse tree ({message}); "
|
|
327
|
+
"the original source was kept for this file."
|
|
328
|
+
)
|
|
329
|
+
raw_lines = [SourceLine(n, t) for n, t in enumerate(text.split("\n"), start=1)]
|
|
330
|
+
result.lines = postprocess(raw_lines, text, options)
|
|
331
|
+
result.method = "verbatim (verification failed)"
|
|
332
|
+
return result
|
|
333
|
+
|
|
334
|
+
original_count = text.count("\n") + 1
|
|
335
|
+
result.lines = postprocess(stripped_lines, text, options)
|
|
336
|
+
result.removed_lines = max(0, original_count - len(result.lines))
|
|
337
|
+
return result
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
"""Hand-written comment scanner used when pygments cannot be trusted.
|
|
2
|
+
|
|
3
|
+
A single pass over the text tracks whether we are in code, a string, a
|
|
4
|
+
character literal, a raw/verbatim string, or a comment. Only regions
|
|
5
|
+
entered through a comment token are reported, so the scanner can never
|
|
6
|
+
classify code as a comment by accident.
|
|
7
|
+
|
|
8
|
+
The awkward cases it exists to survive:
|
|
9
|
+
|
|
10
|
+
* ``"/* not a comment */"`` and ``"// not a comment"`` inside strings
|
|
11
|
+
* C++11 raw strings ``R"tag( */ )tag"``
|
|
12
|
+
* C# verbatim strings ``@"C:\\path"`` with ``""`` escapes
|
|
13
|
+
* Rust ``r#"..."#``, Go and JavaScript backtick strings
|
|
14
|
+
* C line-splicing, where a ``\\`` continues a ``//`` comment onto the next line
|
|
15
|
+
* JavaScript regex literals such as ``/a\\/\\/b/`` that contain ``//``
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from ..languages import FAMILY_C, Language
|
|
21
|
+
from . import KIND_COMMENT, KIND_SHEBANG, Span
|
|
22
|
+
|
|
23
|
+
# After one of these, a '/' in JavaScript begins a regex literal rather
|
|
24
|
+
# than a division. This is the standard disambiguation heuristic.
|
|
25
|
+
_REGEX_PRECEDERS = set("(,=:[!&|?{};+-*%^~<>") | {""}
|
|
26
|
+
_REGEX_KEYWORDS = ("return", "typeof", "instanceof", "in", "of", "new", "delete", "case", "do", "else", "yield", "await")
|
|
27
|
+
|
|
28
|
+
_RAW_PREFIXES = ("u8R", "LR", "uR", "UR", "R")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _line_comment_end(text: str, start: int, splice: bool) -> int:
|
|
32
|
+
"""End offset of a line comment (exclusive of the newline)."""
|
|
33
|
+
i = start
|
|
34
|
+
n = len(text)
|
|
35
|
+
while i < n:
|
|
36
|
+
ch = text[i]
|
|
37
|
+
if ch == "\n":
|
|
38
|
+
if splice and i > start and text[i - 1] == "\\":
|
|
39
|
+
i += 1
|
|
40
|
+
continue
|
|
41
|
+
return i
|
|
42
|
+
i += 1
|
|
43
|
+
return n
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _skip_quoted(text: str, start: int, quote: str, escapes: bool) -> int:
|
|
47
|
+
"""End offset (exclusive) of a quoted run beginning at `start`."""
|
|
48
|
+
i = start + 1
|
|
49
|
+
n = len(text)
|
|
50
|
+
while i < n:
|
|
51
|
+
ch = text[i]
|
|
52
|
+
if escapes and ch == "\\":
|
|
53
|
+
i += 2
|
|
54
|
+
continue
|
|
55
|
+
if ch == quote:
|
|
56
|
+
return i + 1
|
|
57
|
+
i += 1
|
|
58
|
+
return n
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _skip_cpp_raw(text: str, quote_pos: int) -> int | None:
|
|
62
|
+
"""Handle R"delim( ... )delim" starting at the quote character."""
|
|
63
|
+
i = quote_pos + 1
|
|
64
|
+
n = len(text)
|
|
65
|
+
delim_start = i
|
|
66
|
+
while i < n and text[i] not in "( \t\n\\":
|
|
67
|
+
i += 1
|
|
68
|
+
if i >= n or text[i] != "(":
|
|
69
|
+
return None
|
|
70
|
+
delim = text[delim_start:i]
|
|
71
|
+
closer = ")" + delim + '"'
|
|
72
|
+
end = text.find(closer, i + 1)
|
|
73
|
+
return n if end == -1 else end + len(closer)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _has_raw_prefix(text: str, quote_pos: int) -> bool:
|
|
77
|
+
for prefix in _RAW_PREFIXES:
|
|
78
|
+
start = quote_pos - len(prefix)
|
|
79
|
+
if start < 0 or text[start:quote_pos] != prefix:
|
|
80
|
+
continue
|
|
81
|
+
before = text[start - 1] if start > 0 else ""
|
|
82
|
+
if before.isalnum() or before == "_":
|
|
83
|
+
continue
|
|
84
|
+
return True
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _skip_rust_raw(text: str, r_pos: int) -> int | None:
|
|
89
|
+
"""Handle r"..." and r#"..."# starting at the 'r'."""
|
|
90
|
+
i = r_pos + 1
|
|
91
|
+
hashes = 0
|
|
92
|
+
while i < len(text) and text[i] == "#":
|
|
93
|
+
hashes += 1
|
|
94
|
+
i += 1
|
|
95
|
+
if i >= len(text) or text[i] != '"':
|
|
96
|
+
return None
|
|
97
|
+
closer = '"' + "#" * hashes
|
|
98
|
+
end = text.find(closer, i + 1)
|
|
99
|
+
return len(text) if end == -1 else end + len(closer)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _skip_regex(text: str, start: int) -> int:
|
|
103
|
+
"""End offset of a JavaScript regex literal beginning at '/'."""
|
|
104
|
+
i = start + 1
|
|
105
|
+
n = len(text)
|
|
106
|
+
in_class = False
|
|
107
|
+
while i < n:
|
|
108
|
+
ch = text[i]
|
|
109
|
+
if ch == "\\":
|
|
110
|
+
i += 2
|
|
111
|
+
continue
|
|
112
|
+
if ch == "\n":
|
|
113
|
+
return start + 1 # unterminated: treat the '/' as ordinary
|
|
114
|
+
if ch == "[":
|
|
115
|
+
in_class = True
|
|
116
|
+
elif ch == "]":
|
|
117
|
+
in_class = False
|
|
118
|
+
elif ch == "/" and not in_class:
|
|
119
|
+
return i + 1
|
|
120
|
+
i += 1
|
|
121
|
+
return start + 1
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _regex_allowed(text: str, index: int) -> bool:
|
|
125
|
+
j = index - 1
|
|
126
|
+
while j >= 0 and text[j] in " \t\r\n":
|
|
127
|
+
j -= 1
|
|
128
|
+
if j < 0:
|
|
129
|
+
return True
|
|
130
|
+
ch = text[j]
|
|
131
|
+
if ch in _REGEX_PRECEDERS:
|
|
132
|
+
return True
|
|
133
|
+
if ch.isalnum() or ch == "_":
|
|
134
|
+
end = j + 1
|
|
135
|
+
while j >= 0 and (text[j].isalnum() or text[j] == "_"):
|
|
136
|
+
j -= 1
|
|
137
|
+
return text[j + 1 : end] in _REGEX_KEYWORDS
|
|
138
|
+
return False
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def find_spans(text: str, language: Language) -> tuple[list[Span], list[str]]:
|
|
142
|
+
alias = language.pygments_alias
|
|
143
|
+
is_js = alias in {"javascript", "jsx", "typescript", "tsx"}
|
|
144
|
+
is_cpp = language.raw_strings and alias in {"cpp", "cuda"}
|
|
145
|
+
is_rust = alias == "rust"
|
|
146
|
+
is_csharp = alias == "csharp"
|
|
147
|
+
backticks = language.raw_strings or is_js or alias == "go"
|
|
148
|
+
splice = language.family == FAMILY_C
|
|
149
|
+
|
|
150
|
+
line_tokens = sorted(language.line_comments, key=len, reverse=True)
|
|
151
|
+
block_tokens = language.block_comments
|
|
152
|
+
|
|
153
|
+
spans: list[Span] = []
|
|
154
|
+
warnings: list[str] = []
|
|
155
|
+
i = 0
|
|
156
|
+
n = len(text)
|
|
157
|
+
|
|
158
|
+
while i < n:
|
|
159
|
+
ch = text[i]
|
|
160
|
+
|
|
161
|
+
# --- comments -----------------------------------------------------
|
|
162
|
+
matched = False
|
|
163
|
+
for open_tok, close_tok in block_tokens:
|
|
164
|
+
if text.startswith(open_tok, i):
|
|
165
|
+
end = text.find(close_tok, i + len(open_tok))
|
|
166
|
+
if end == -1:
|
|
167
|
+
end = n
|
|
168
|
+
warnings.append("Unterminated block comment; stripped to end of file.")
|
|
169
|
+
else:
|
|
170
|
+
end += len(close_tok)
|
|
171
|
+
spans.append((i, end, KIND_COMMENT))
|
|
172
|
+
i = end
|
|
173
|
+
matched = True
|
|
174
|
+
break
|
|
175
|
+
if matched:
|
|
176
|
+
continue
|
|
177
|
+
|
|
178
|
+
for tok in line_tokens:
|
|
179
|
+
if not text.startswith(tok, i):
|
|
180
|
+
continue
|
|
181
|
+
end = _line_comment_end(text, i, splice)
|
|
182
|
+
kind = KIND_SHEBANG if i == 0 and text.startswith("#!") else KIND_COMMENT
|
|
183
|
+
spans.append((i, end, kind))
|
|
184
|
+
i = end
|
|
185
|
+
matched = True
|
|
186
|
+
break
|
|
187
|
+
if matched:
|
|
188
|
+
continue
|
|
189
|
+
|
|
190
|
+
# --- literals that must not be scanned for comment tokens ---------
|
|
191
|
+
if ch == '"':
|
|
192
|
+
if is_cpp and _has_raw_prefix(text, i):
|
|
193
|
+
end = _skip_cpp_raw(text, i)
|
|
194
|
+
i = end if end is not None else _skip_quoted(text, i, '"', True)
|
|
195
|
+
continue
|
|
196
|
+
if is_csharp and i > 0 and text[i - 1] == "@":
|
|
197
|
+
j = i + 1
|
|
198
|
+
while j < n:
|
|
199
|
+
if text[j] == '"':
|
|
200
|
+
if j + 1 < n and text[j + 1] == '"':
|
|
201
|
+
j += 2
|
|
202
|
+
continue
|
|
203
|
+
j += 1
|
|
204
|
+
break
|
|
205
|
+
j += 1
|
|
206
|
+
i = j
|
|
207
|
+
continue
|
|
208
|
+
if text.startswith('"""', i): # Java/Kotlin/Swift text blocks
|
|
209
|
+
end = text.find('"""', i + 3)
|
|
210
|
+
i = n if end == -1 else end + 3
|
|
211
|
+
continue
|
|
212
|
+
i = _skip_quoted(text, i, '"', True)
|
|
213
|
+
continue
|
|
214
|
+
|
|
215
|
+
if ch == "'" and language.char_literals:
|
|
216
|
+
i = _skip_quoted(text, i, "'", True)
|
|
217
|
+
continue
|
|
218
|
+
|
|
219
|
+
if ch == "`" and backticks:
|
|
220
|
+
i = _skip_quoted(text, i, "`", alias != "go")
|
|
221
|
+
continue
|
|
222
|
+
|
|
223
|
+
if is_rust and ch == "r":
|
|
224
|
+
end = _skip_rust_raw(text, i)
|
|
225
|
+
if end is not None:
|
|
226
|
+
i = end
|
|
227
|
+
continue
|
|
228
|
+
|
|
229
|
+
if is_js and ch == "/" and _regex_allowed(text, i):
|
|
230
|
+
i = _skip_regex(text, i)
|
|
231
|
+
continue
|
|
232
|
+
|
|
233
|
+
i += 1
|
|
234
|
+
|
|
235
|
+
return spans, warnings
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Universal comment classification via pygments.
|
|
2
|
+
|
|
3
|
+
Two safeguards make this trustworthy enough to be the default path for
|
|
4
|
+
non-Python languages:
|
|
5
|
+
|
|
6
|
+
1. **Round-trip verification.** The concatenated token values must equal
|
|
7
|
+
the source exactly. If a lexer drops or rewrites anything, we refuse
|
|
8
|
+
the result and the caller falls back to the state machine.
|
|
9
|
+
2. **Preprocessor tokens are never treated as comments.** pygments maps
|
|
10
|
+
``#include`` / ``#define`` to ``Comment.Preproc``; removing those would
|
|
11
|
+
quietly delete real code from the deposit.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from pygments.token import Comment, String, Token
|
|
17
|
+
|
|
18
|
+
from ..languages import Language
|
|
19
|
+
from . import KIND_COMMENT, KIND_DOCSTRING, KIND_SHEBANG, Span
|
|
20
|
+
|
|
21
|
+
_PREPROC = (Token.Comment.Preproc, Token.Comment.PreprocFile)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _lexer_for(language: Language):
|
|
25
|
+
from pygments.lexers import get_lexer_by_name
|
|
26
|
+
|
|
27
|
+
if not language.pygments_alias:
|
|
28
|
+
return None
|
|
29
|
+
try:
|
|
30
|
+
return get_lexer_by_name(language.pygments_alias, stripnl=False, ensurenl=False)
|
|
31
|
+
except Exception:
|
|
32
|
+
return None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _is_preproc(ttype) -> bool:
|
|
36
|
+
return any(ttype in p for p in _PREPROC)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _classify(ttype) -> str | None:
|
|
40
|
+
if ttype in Token.Comment.Hashbang:
|
|
41
|
+
return KIND_SHEBANG
|
|
42
|
+
if ttype in String.Doc:
|
|
43
|
+
return KIND_DOCSTRING
|
|
44
|
+
if ttype in Comment and not _is_preproc(ttype):
|
|
45
|
+
return KIND_COMMENT
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def find_spans(text: str, language: Language) -> tuple[list[Span] | None, list[str]]:
|
|
50
|
+
"""Return classified spans, or ``None`` if pygments cannot be trusted."""
|
|
51
|
+
lexer = _lexer_for(language)
|
|
52
|
+
if lexer is None:
|
|
53
|
+
return None, []
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
tokens = list(lexer.get_tokens_unprocessed(text))
|
|
57
|
+
except Exception as exc: # a lexer crash must never abort a build
|
|
58
|
+
return None, [f"Lexer for {language.name} failed ({exc.__class__.__name__})."]
|
|
59
|
+
|
|
60
|
+
# Round-trip check: the token stream must reconstruct the source.
|
|
61
|
+
rebuilt_length = 0
|
|
62
|
+
for index, _ttype, value in tokens:
|
|
63
|
+
if index != rebuilt_length:
|
|
64
|
+
return None, [
|
|
65
|
+
f"Lexer output for {language.name} did not reconstruct the source; "
|
|
66
|
+
"using the fallback comment scanner."
|
|
67
|
+
]
|
|
68
|
+
rebuilt_length += len(value)
|
|
69
|
+
if rebuilt_length != len(text):
|
|
70
|
+
return None, [
|
|
71
|
+
f"Lexer output for {language.name} was truncated; using the fallback scanner."
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
spans: list[Span] = []
|
|
75
|
+
for index, ttype, value in tokens:
|
|
76
|
+
kind = _classify(ttype)
|
|
77
|
+
if kind is None or not value:
|
|
78
|
+
continue
|
|
79
|
+
end = index + len(value)
|
|
80
|
+
# A trailing newline belongs to the layout, not to the comment.
|
|
81
|
+
while end > index and text[end - 1] == "\n":
|
|
82
|
+
end -= 1
|
|
83
|
+
if end > index:
|
|
84
|
+
spans.append((index, end, kind))
|
|
85
|
+
return spans, []
|