codendium 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. codendium-1.0.0.dist-info/METADATA +332 -0
  2. codendium-1.0.0.dist-info/RECORD +45 -0
  3. codendium-1.0.0.dist-info/WHEEL +5 -0
  4. codendium-1.0.0.dist-info/entry_points.txt +6 -0
  5. codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
  6. codendium-1.0.0.dist-info/top_level.txt +1 -0
  7. copyright_deposit/__init__.py +15 -0
  8. copyright_deposit/__main__.py +34 -0
  9. copyright_deposit/assets/__init__.py +5 -0
  10. copyright_deposit/assets/fonts/README.md +31 -0
  11. copyright_deposit/assets/logo.svg +26 -0
  12. copyright_deposit/cli.py +314 -0
  13. copyright_deposit/config.py +269 -0
  14. copyright_deposit/core/__init__.py +1 -0
  15. copyright_deposit/core/deposit.py +117 -0
  16. copyright_deposit/core/discovery.py +260 -0
  17. copyright_deposit/core/encoding.py +136 -0
  18. copyright_deposit/core/languages.py +190 -0
  19. copyright_deposit/core/layout.py +349 -0
  20. copyright_deposit/core/lineranges.py +219 -0
  21. copyright_deposit/core/manifest.py +282 -0
  22. copyright_deposit/core/metrics.py +279 -0
  23. copyright_deposit/core/ordering.py +349 -0
  24. copyright_deposit/core/pipeline.py +386 -0
  25. copyright_deposit/core/redaction.py +162 -0
  26. copyright_deposit/core/render.py +242 -0
  27. copyright_deposit/core/scanning/__init__.py +61 -0
  28. copyright_deposit/core/scanning/secrets.py +181 -0
  29. copyright_deposit/core/scanning/thirdparty.py +190 -0
  30. copyright_deposit/core/strip/__init__.py +337 -0
  31. copyright_deposit/core/strip/cfamily_strip.py +235 -0
  32. copyright_deposit/core/strip/pygments_strip.py +85 -0
  33. copyright_deposit/core/strip/python_strip.py +131 -0
  34. copyright_deposit/gui/__init__.py +1 -0
  35. copyright_deposit/gui/app.py +34 -0
  36. copyright_deposit/gui/branding.py +83 -0
  37. copyright_deposit/gui/history.py +192 -0
  38. copyright_deposit/gui/main_window.py +617 -0
  39. copyright_deposit/gui/panels/__init__.py +1 -0
  40. copyright_deposit/gui/panels/estimate.py +166 -0
  41. copyright_deposit/gui/panels/files.py +635 -0
  42. copyright_deposit/gui/panels/identification.py +193 -0
  43. copyright_deposit/gui/panels/options.py +445 -0
  44. copyright_deposit/gui/panels/preflight.py +260 -0
  45. copyright_deposit/gui/workers.py +96 -0
@@ -0,0 +1,337 @@
1
+ """Comment and docstring removal.
2
+
3
+ Design
4
+ ------
5
+ Strippers never rewrite code. They *classify* byte ranges of the original
6
+ text as comment/docstring, and a single shared routine removes those
7
+ ranges. Removal is therefore provably subtractive: nothing that was not
8
+ classified as a comment can ever be altered.
9
+
10
+ Dispatch, in order of trustworthiness:
11
+
12
+ * Python -> ``tokenize`` + ``ast``. Authoritative, and the result is
13
+ checked by comparing the ASTs of the original and stripped source.
14
+ * Others -> pygments, whose token stream is verified to reconstruct the
15
+ source exactly before it is trusted.
16
+ * Fallback -> a hand-written C-family state machine (also used when no
17
+ lexer exists or the pygments round-trip fails).
18
+
19
+ The pygments path deliberately keeps ``Comment.Preproc`` tokens: C lexers
20
+ classify ``#include`` and ``#define`` as comments, and deleting those
21
+ would silently gut the deposit.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import re
27
+ from dataclasses import dataclass, field
28
+
29
+ from ...config import LEGAL_HEADER_PATTERN, TransformOptions
30
+ from .. import languages
31
+ from ..languages import FAMILY_C, FAMILY_HASH, FAMILY_PYTHON, Language
32
+
33
+ KIND_COMMENT = "comment"
34
+ KIND_DOCSTRING = "docstring"
35
+ KIND_SHEBANG = "shebang"
36
+
37
+ Span = tuple[int, int, str] # (start offset, end offset, kind)
38
+
39
+ _LEGAL_RE = re.compile(LEGAL_HEADER_PATTERN, re.IGNORECASE)
40
+
41
+
42
+ @dataclass
43
+ class SourceLine:
44
+ """One physical line of the stripped source."""
45
+
46
+ number: int # 1-based line number in the ORIGINAL file
47
+ text: str
48
+
49
+
50
+ @dataclass
51
+ class StripResult:
52
+ lines: list[SourceLine] = field(default_factory=list)
53
+ method: str = "none"
54
+ removed_lines: int = 0
55
+ warnings: list[str] = field(default_factory=list)
56
+ # Every classified comment/docstring span, before the keep-policy
57
+ # toggles are applied. The third-party scanner reads these so it only
58
+ # inspects real comments instead of matching prose inside code.
59
+ comment_spans: list[Span] = field(default_factory=list)
60
+
61
+ @property
62
+ def line_count(self) -> int:
63
+ return len(self.lines)
64
+
65
+
66
+ # ---------------------------------------------------------------------------
67
+ # Offset helpers
68
+ # ---------------------------------------------------------------------------
69
+
70
+
71
+ def line_start_offsets(text: str) -> list[int]:
72
+ """Offset of the first character of each 1-based line (index 0 unused)."""
73
+ starts = [0, 0]
74
+ for i, ch in enumerate(text):
75
+ if ch == "\n":
76
+ starts.append(i + 1)
77
+ return starts
78
+
79
+
80
+ def rowcol_to_offset(starts: list[int], row: int, col: int) -> int:
81
+ if row < 1 or row >= len(starts):
82
+ return len(starts) and starts[-1]
83
+ return starts[row] + col
84
+
85
+
86
+ # ---------------------------------------------------------------------------
87
+ # Span post-processing
88
+ # ---------------------------------------------------------------------------
89
+
90
+
91
+ def merge_spans(spans: list[Span]) -> list[Span]:
92
+ """Sort and coalesce overlapping spans."""
93
+ if not spans:
94
+ return []
95
+ spans = sorted(spans, key=lambda s: (s[0], s[1]))
96
+ merged: list[Span] = [spans[0]]
97
+ for start, end, kind in spans[1:]:
98
+ last_start, last_end, last_kind = merged[-1]
99
+ if start <= last_end:
100
+ merged[-1] = (last_start, max(last_end, end), last_kind or kind)
101
+ else:
102
+ merged.append((start, end, kind))
103
+ return merged
104
+
105
+
106
+ def _filter_spans(text: str, spans: list[Span], options: TransformOptions) -> list[Span]:
107
+ """Apply the keep-policy toggles to classified spans."""
108
+ kept: list[Span] = []
109
+ for start, end, kind in spans:
110
+ if kind == KIND_SHEBANG:
111
+ if options.preserve_shebang:
112
+ continue
113
+ if not options.strip_comments:
114
+ continue
115
+ kept.append((start, end, kind))
116
+ continue
117
+ if kind == KIND_DOCSTRING:
118
+ if options.strip_docstrings:
119
+ kept.append((start, end, kind))
120
+ continue
121
+ if options.strip_comments:
122
+ kept.append((start, end, kind))
123
+ if not options.preserve_legal_headers:
124
+ return kept
125
+ return _keep_legal_header(text, kept)
126
+
127
+
128
+ def _keep_legal_header(text: str, spans: list[Span]) -> list[Span]:
129
+ """Drop the leading comment block from the kill list if it reads as legal.
130
+
131
+ Only comments appearing before the first line of real code qualify: a
132
+ licence notice buried mid-file is not the header the Office asks for.
133
+ """
134
+ if not spans:
135
+ return spans
136
+ result: list[Span] = []
137
+ for index, (start, end, kind) in enumerate(spans):
138
+ if kind == KIND_DOCSTRING and index > 0:
139
+ result.append((start, end, kind))
140
+ continue
141
+ before = text[:start]
142
+ # "Leading" means only whitespace and other comments precede it.
143
+ stripped_before = before
144
+ for prev_start, prev_end, _ in spans[:index]:
145
+ stripped_before = (
146
+ stripped_before[:prev_start] + " " * (prev_end - prev_start) + stripped_before[prev_end:]
147
+ )
148
+ if stripped_before.strip():
149
+ result.append((start, end, kind))
150
+ continue
151
+ if _LEGAL_RE.search(text[start:end]):
152
+ continue # preserved
153
+ result.append((start, end, kind))
154
+ return result
155
+
156
+
157
+ # ---------------------------------------------------------------------------
158
+ # Removal
159
+ # ---------------------------------------------------------------------------
160
+
161
+
162
+ def apply_spans(text: str, spans: list[Span]) -> list[SourceLine]:
163
+ """Remove the spans and rebuild lines, tracking original line numbers.
164
+
165
+ Removal joins fragments across a multi-line comment, which is what a
166
+ reader expects: ``int x = /* note\\n more */ 5;`` becomes ``int x = 5;``.
167
+ """
168
+ spans = merge_spans(spans)
169
+ lines: list[SourceLine] = []
170
+ buf: list[str] = []
171
+ orig_line = 1
172
+ line_started_at = 1
173
+ span_index = 0
174
+ i = 0
175
+ length = len(text)
176
+
177
+ while i < length:
178
+ if span_index < len(spans) and i == spans[span_index][0]:
179
+ _, end, _ = spans[span_index]
180
+ # Count newlines swallowed by the comment so numbering stays true.
181
+ orig_line += text.count("\n", i, end)
182
+ i = end
183
+ span_index += 1
184
+ continue
185
+ ch = text[i]
186
+ if ch == "\n":
187
+ lines.append(SourceLine(line_started_at, "".join(buf)))
188
+ buf = []
189
+ orig_line += 1
190
+ line_started_at = orig_line
191
+ i += 1
192
+ continue
193
+ buf.append(ch)
194
+ i += 1
195
+
196
+ if buf:
197
+ lines.append(SourceLine(line_started_at, "".join(buf)))
198
+ return lines
199
+
200
+
201
+ def comments_only_lines(text: str, spans: list[Span]) -> list[str]:
202
+ """The file with every non-comment character blanked out.
203
+
204
+ Line numbering is preserved exactly, so a scanner can report a hit at
205
+ its true location while never seeing a character of actual code.
206
+ """
207
+ keep = bytearray(len(text))
208
+ for start, end, _kind in spans:
209
+ for i in range(max(0, start), min(len(text), end)):
210
+ keep[i] = 1
211
+ out = [
212
+ ch if (ch == "\n" or keep[i]) else " "
213
+ for i, ch in enumerate(text)
214
+ ]
215
+ return "".join(out).split("\n")
216
+
217
+
218
+ def _blank_original_lines(text: str) -> set[int]:
219
+ return {n for n, raw in enumerate(text.split("\n"), start=1) if not raw.strip()}
220
+
221
+
222
+ def postprocess(
223
+ lines: list[SourceLine],
224
+ original_text: str,
225
+ options: TransformOptions,
226
+ ) -> list[SourceLine]:
227
+ """Tabs, trailing whitespace, comment-only line removal, blank collapsing."""
228
+ originally_blank = _blank_original_lines(original_text)
229
+
230
+ staged: list[SourceLine] = []
231
+ for line in lines:
232
+ text = line.text
233
+ if options.expand_tabs:
234
+ text = text.expandtabs(options.tab_width)
235
+ if options.trim_trailing_whitespace:
236
+ text = text.rstrip()
237
+ if not text.strip() and line.number not in originally_blank:
238
+ # The line held nothing but a comment; drop it entirely.
239
+ continue
240
+ staged.append(SourceLine(line.number, text))
241
+
242
+ if options.collapse_blank_runs:
243
+ collapsed: list[SourceLine] = []
244
+ run = 0
245
+ for line in staged:
246
+ if line.text.strip():
247
+ run = 0
248
+ collapsed.append(line)
249
+ continue
250
+ run += 1
251
+ if run <= max(0, options.max_blank_run):
252
+ collapsed.append(line)
253
+ staged = collapsed
254
+
255
+ while staged and not staged[0].text.strip():
256
+ staged.pop(0)
257
+ while staged and not staged[-1].text.strip():
258
+ staged.pop()
259
+ return staged
260
+
261
+
262
+ # ---------------------------------------------------------------------------
263
+ # Dispatch
264
+ # ---------------------------------------------------------------------------
265
+
266
+
267
+ def find_spans(text: str, language: Language) -> tuple[list[Span], str, list[str]]:
268
+ """Classify comment/docstring ranges. Returns (spans, method, warnings)."""
269
+ from . import cfamily_strip, pygments_strip, python_strip
270
+
271
+ if language.family == FAMILY_PYTHON:
272
+ spans, warnings = python_strip.find_spans(text)
273
+ if spans is not None:
274
+ return spans, "python-tokenize", warnings
275
+
276
+ spans, warnings = pygments_strip.find_spans(text, language)
277
+ if spans is not None:
278
+ return spans, "pygments", warnings
279
+
280
+ if language.family == FAMILY_C:
281
+ spans, w2 = cfamily_strip.find_spans(text, language)
282
+ return spans, "c-state-machine", warnings + w2
283
+
284
+ if language.family in (FAMILY_HASH, FAMILY_PYTHON) or language.line_comments:
285
+ spans, w2 = cfamily_strip.find_spans(text, language)
286
+ return spans, "line-scanner", warnings + w2
287
+
288
+ return [], "none", warnings + ["No comment syntax known; source kept verbatim."]
289
+
290
+
291
+ def strip_source(
292
+ text: str,
293
+ path: str,
294
+ options: TransformOptions,
295
+ language: Language | None = None,
296
+ ) -> StripResult:
297
+ """Strip `text` according to `options`, returning renderable lines."""
298
+ language = language or languages.detect(path)
299
+ result = StripResult()
300
+
301
+ # Spans are always classified, even when nothing will be removed: the
302
+ # third-party scanner needs to know which regions are comments.
303
+ all_spans, method, warnings = find_spans(text, language)
304
+ all_spans = merge_spans(all_spans)
305
+ result.comment_spans = all_spans
306
+ result.method = method
307
+ result.warnings.extend(warnings)
308
+
309
+ if not (options.strip_comments or options.strip_docstrings):
310
+ result.method = "verbatim"
311
+ raw_lines = [SourceLine(n, t) for n, t in enumerate(text.split("\n"), start=1)]
312
+ result.lines = postprocess(raw_lines, text, options)
313
+ return result
314
+
315
+ spans = _filter_spans(text, all_spans, options)
316
+ stripped_lines = apply_spans(text, spans)
317
+
318
+ # Safety net: for Python the transform must not change the parse tree.
319
+ if language.family == FAMILY_PYTHON and spans:
320
+ from . import python_strip
321
+
322
+ rebuilt = "\n".join(line.text for line in stripped_lines)
323
+ ok, message = python_strip.verify_equivalent(text, rebuilt, options)
324
+ if not ok:
325
+ result.warnings.append(
326
+ f"Comment removal changed the parse tree ({message}); "
327
+ "the original source was kept for this file."
328
+ )
329
+ raw_lines = [SourceLine(n, t) for n, t in enumerate(text.split("\n"), start=1)]
330
+ result.lines = postprocess(raw_lines, text, options)
331
+ result.method = "verbatim (verification failed)"
332
+ return result
333
+
334
+ original_count = text.count("\n") + 1
335
+ result.lines = postprocess(stripped_lines, text, options)
336
+ result.removed_lines = max(0, original_count - len(result.lines))
337
+ return result
@@ -0,0 +1,235 @@
1
+ """Hand-written comment scanner used when pygments cannot be trusted.
2
+
3
+ A single pass over the text tracks whether we are in code, a string, a
4
+ character literal, a raw/verbatim string, or a comment. Only regions
5
+ entered through a comment token are reported, so the scanner can never
6
+ classify code as a comment by accident.
7
+
8
+ The awkward cases it exists to survive:
9
+
10
+ * ``"/* not a comment */"`` and ``"// not a comment"`` inside strings
11
+ * C++11 raw strings ``R"tag( */ )tag"``
12
+ * C# verbatim strings ``@"C:\\path"`` with ``""`` escapes
13
+ * Rust ``r#"..."#``, Go and JavaScript backtick strings
14
+ * C line-splicing, where a ``\\`` continues a ``//`` comment onto the next line
15
+ * JavaScript regex literals such as ``/a\\/\\/b/`` that contain ``//``
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from ..languages import FAMILY_C, Language
21
+ from . import KIND_COMMENT, KIND_SHEBANG, Span
22
+
23
+ # After one of these, a '/' in JavaScript begins a regex literal rather
24
+ # than a division. This is the standard disambiguation heuristic.
25
+ _REGEX_PRECEDERS = set("(,=:[!&|?{};+-*%^~<>") | {""}
26
+ _REGEX_KEYWORDS = ("return", "typeof", "instanceof", "in", "of", "new", "delete", "case", "do", "else", "yield", "await")
27
+
28
+ _RAW_PREFIXES = ("u8R", "LR", "uR", "UR", "R")
29
+
30
+
31
+ def _line_comment_end(text: str, start: int, splice: bool) -> int:
32
+ """End offset of a line comment (exclusive of the newline)."""
33
+ i = start
34
+ n = len(text)
35
+ while i < n:
36
+ ch = text[i]
37
+ if ch == "\n":
38
+ if splice and i > start and text[i - 1] == "\\":
39
+ i += 1
40
+ continue
41
+ return i
42
+ i += 1
43
+ return n
44
+
45
+
46
+ def _skip_quoted(text: str, start: int, quote: str, escapes: bool) -> int:
47
+ """End offset (exclusive) of a quoted run beginning at `start`."""
48
+ i = start + 1
49
+ n = len(text)
50
+ while i < n:
51
+ ch = text[i]
52
+ if escapes and ch == "\\":
53
+ i += 2
54
+ continue
55
+ if ch == quote:
56
+ return i + 1
57
+ i += 1
58
+ return n
59
+
60
+
61
+ def _skip_cpp_raw(text: str, quote_pos: int) -> int | None:
62
+ """Handle R"delim( ... )delim" starting at the quote character."""
63
+ i = quote_pos + 1
64
+ n = len(text)
65
+ delim_start = i
66
+ while i < n and text[i] not in "( \t\n\\":
67
+ i += 1
68
+ if i >= n or text[i] != "(":
69
+ return None
70
+ delim = text[delim_start:i]
71
+ closer = ")" + delim + '"'
72
+ end = text.find(closer, i + 1)
73
+ return n if end == -1 else end + len(closer)
74
+
75
+
76
+ def _has_raw_prefix(text: str, quote_pos: int) -> bool:
77
+ for prefix in _RAW_PREFIXES:
78
+ start = quote_pos - len(prefix)
79
+ if start < 0 or text[start:quote_pos] != prefix:
80
+ continue
81
+ before = text[start - 1] if start > 0 else ""
82
+ if before.isalnum() or before == "_":
83
+ continue
84
+ return True
85
+ return False
86
+
87
+
88
+ def _skip_rust_raw(text: str, r_pos: int) -> int | None:
89
+ """Handle r"..." and r#"..."# starting at the 'r'."""
90
+ i = r_pos + 1
91
+ hashes = 0
92
+ while i < len(text) and text[i] == "#":
93
+ hashes += 1
94
+ i += 1
95
+ if i >= len(text) or text[i] != '"':
96
+ return None
97
+ closer = '"' + "#" * hashes
98
+ end = text.find(closer, i + 1)
99
+ return len(text) if end == -1 else end + len(closer)
100
+
101
+
102
+ def _skip_regex(text: str, start: int) -> int:
103
+ """End offset of a JavaScript regex literal beginning at '/'."""
104
+ i = start + 1
105
+ n = len(text)
106
+ in_class = False
107
+ while i < n:
108
+ ch = text[i]
109
+ if ch == "\\":
110
+ i += 2
111
+ continue
112
+ if ch == "\n":
113
+ return start + 1 # unterminated: treat the '/' as ordinary
114
+ if ch == "[":
115
+ in_class = True
116
+ elif ch == "]":
117
+ in_class = False
118
+ elif ch == "/" and not in_class:
119
+ return i + 1
120
+ i += 1
121
+ return start + 1
122
+
123
+
124
+ def _regex_allowed(text: str, index: int) -> bool:
125
+ j = index - 1
126
+ while j >= 0 and text[j] in " \t\r\n":
127
+ j -= 1
128
+ if j < 0:
129
+ return True
130
+ ch = text[j]
131
+ if ch in _REGEX_PRECEDERS:
132
+ return True
133
+ if ch.isalnum() or ch == "_":
134
+ end = j + 1
135
+ while j >= 0 and (text[j].isalnum() or text[j] == "_"):
136
+ j -= 1
137
+ return text[j + 1 : end] in _REGEX_KEYWORDS
138
+ return False
139
+
140
+
141
+ def find_spans(text: str, language: Language) -> tuple[list[Span], list[str]]:
142
+ alias = language.pygments_alias
143
+ is_js = alias in {"javascript", "jsx", "typescript", "tsx"}
144
+ is_cpp = language.raw_strings and alias in {"cpp", "cuda"}
145
+ is_rust = alias == "rust"
146
+ is_csharp = alias == "csharp"
147
+ backticks = language.raw_strings or is_js or alias == "go"
148
+ splice = language.family == FAMILY_C
149
+
150
+ line_tokens = sorted(language.line_comments, key=len, reverse=True)
151
+ block_tokens = language.block_comments
152
+
153
+ spans: list[Span] = []
154
+ warnings: list[str] = []
155
+ i = 0
156
+ n = len(text)
157
+
158
+ while i < n:
159
+ ch = text[i]
160
+
161
+ # --- comments -----------------------------------------------------
162
+ matched = False
163
+ for open_tok, close_tok in block_tokens:
164
+ if text.startswith(open_tok, i):
165
+ end = text.find(close_tok, i + len(open_tok))
166
+ if end == -1:
167
+ end = n
168
+ warnings.append("Unterminated block comment; stripped to end of file.")
169
+ else:
170
+ end += len(close_tok)
171
+ spans.append((i, end, KIND_COMMENT))
172
+ i = end
173
+ matched = True
174
+ break
175
+ if matched:
176
+ continue
177
+
178
+ for tok in line_tokens:
179
+ if not text.startswith(tok, i):
180
+ continue
181
+ end = _line_comment_end(text, i, splice)
182
+ kind = KIND_SHEBANG if i == 0 and text.startswith("#!") else KIND_COMMENT
183
+ spans.append((i, end, kind))
184
+ i = end
185
+ matched = True
186
+ break
187
+ if matched:
188
+ continue
189
+
190
+ # --- literals that must not be scanned for comment tokens ---------
191
+ if ch == '"':
192
+ if is_cpp and _has_raw_prefix(text, i):
193
+ end = _skip_cpp_raw(text, i)
194
+ i = end if end is not None else _skip_quoted(text, i, '"', True)
195
+ continue
196
+ if is_csharp and i > 0 and text[i - 1] == "@":
197
+ j = i + 1
198
+ while j < n:
199
+ if text[j] == '"':
200
+ if j + 1 < n and text[j + 1] == '"':
201
+ j += 2
202
+ continue
203
+ j += 1
204
+ break
205
+ j += 1
206
+ i = j
207
+ continue
208
+ if text.startswith('"""', i): # Java/Kotlin/Swift text blocks
209
+ end = text.find('"""', i + 3)
210
+ i = n if end == -1 else end + 3
211
+ continue
212
+ i = _skip_quoted(text, i, '"', True)
213
+ continue
214
+
215
+ if ch == "'" and language.char_literals:
216
+ i = _skip_quoted(text, i, "'", True)
217
+ continue
218
+
219
+ if ch == "`" and backticks:
220
+ i = _skip_quoted(text, i, "`", alias != "go")
221
+ continue
222
+
223
+ if is_rust and ch == "r":
224
+ end = _skip_rust_raw(text, i)
225
+ if end is not None:
226
+ i = end
227
+ continue
228
+
229
+ if is_js and ch == "/" and _regex_allowed(text, i):
230
+ i = _skip_regex(text, i)
231
+ continue
232
+
233
+ i += 1
234
+
235
+ return spans, warnings
@@ -0,0 +1,85 @@
1
+ """Universal comment classification via pygments.
2
+
3
+ Two safeguards make this trustworthy enough to be the default path for
4
+ non-Python languages:
5
+
6
+ 1. **Round-trip verification.** The concatenated token values must equal
7
+ the source exactly. If a lexer drops or rewrites anything, we refuse
8
+ the result and the caller falls back to the state machine.
9
+ 2. **Preprocessor tokens are never treated as comments.** pygments maps
10
+ ``#include`` / ``#define`` to ``Comment.Preproc``; removing those would
11
+ quietly delete real code from the deposit.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from pygments.token import Comment, String, Token
17
+
18
+ from ..languages import Language
19
+ from . import KIND_COMMENT, KIND_DOCSTRING, KIND_SHEBANG, Span
20
+
21
+ _PREPROC = (Token.Comment.Preproc, Token.Comment.PreprocFile)
22
+
23
+
24
+ def _lexer_for(language: Language):
25
+ from pygments.lexers import get_lexer_by_name
26
+
27
+ if not language.pygments_alias:
28
+ return None
29
+ try:
30
+ return get_lexer_by_name(language.pygments_alias, stripnl=False, ensurenl=False)
31
+ except Exception:
32
+ return None
33
+
34
+
35
+ def _is_preproc(ttype) -> bool:
36
+ return any(ttype in p for p in _PREPROC)
37
+
38
+
39
+ def _classify(ttype) -> str | None:
40
+ if ttype in Token.Comment.Hashbang:
41
+ return KIND_SHEBANG
42
+ if ttype in String.Doc:
43
+ return KIND_DOCSTRING
44
+ if ttype in Comment and not _is_preproc(ttype):
45
+ return KIND_COMMENT
46
+ return None
47
+
48
+
49
+ def find_spans(text: str, language: Language) -> tuple[list[Span] | None, list[str]]:
50
+ """Return classified spans, or ``None`` if pygments cannot be trusted."""
51
+ lexer = _lexer_for(language)
52
+ if lexer is None:
53
+ return None, []
54
+
55
+ try:
56
+ tokens = list(lexer.get_tokens_unprocessed(text))
57
+ except Exception as exc: # a lexer crash must never abort a build
58
+ return None, [f"Lexer for {language.name} failed ({exc.__class__.__name__})."]
59
+
60
+ # Round-trip check: the token stream must reconstruct the source.
61
+ rebuilt_length = 0
62
+ for index, _ttype, value in tokens:
63
+ if index != rebuilt_length:
64
+ return None, [
65
+ f"Lexer output for {language.name} did not reconstruct the source; "
66
+ "using the fallback comment scanner."
67
+ ]
68
+ rebuilt_length += len(value)
69
+ if rebuilt_length != len(text):
70
+ return None, [
71
+ f"Lexer output for {language.name} was truncated; using the fallback scanner."
72
+ ]
73
+
74
+ spans: list[Span] = []
75
+ for index, ttype, value in tokens:
76
+ kind = _classify(ttype)
77
+ if kind is None or not value:
78
+ continue
79
+ end = index + len(value)
80
+ # A trailing newline belongs to the layout, not to the comment.
81
+ while end > index and text[end - 1] == "\n":
82
+ end -= 1
83
+ if end > index:
84
+ spans.append((index, end, kind))
85
+ return spans, []