codendium 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. codendium-1.0.0.dist-info/METADATA +332 -0
  2. codendium-1.0.0.dist-info/RECORD +45 -0
  3. codendium-1.0.0.dist-info/WHEEL +5 -0
  4. codendium-1.0.0.dist-info/entry_points.txt +6 -0
  5. codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
  6. codendium-1.0.0.dist-info/top_level.txt +1 -0
  7. copyright_deposit/__init__.py +15 -0
  8. copyright_deposit/__main__.py +34 -0
  9. copyright_deposit/assets/__init__.py +5 -0
  10. copyright_deposit/assets/fonts/README.md +31 -0
  11. copyright_deposit/assets/logo.svg +26 -0
  12. copyright_deposit/cli.py +314 -0
  13. copyright_deposit/config.py +269 -0
  14. copyright_deposit/core/__init__.py +1 -0
  15. copyright_deposit/core/deposit.py +117 -0
  16. copyright_deposit/core/discovery.py +260 -0
  17. copyright_deposit/core/encoding.py +136 -0
  18. copyright_deposit/core/languages.py +190 -0
  19. copyright_deposit/core/layout.py +349 -0
  20. copyright_deposit/core/lineranges.py +219 -0
  21. copyright_deposit/core/manifest.py +282 -0
  22. copyright_deposit/core/metrics.py +279 -0
  23. copyright_deposit/core/ordering.py +349 -0
  24. copyright_deposit/core/pipeline.py +386 -0
  25. copyright_deposit/core/redaction.py +162 -0
  26. copyright_deposit/core/render.py +242 -0
  27. copyright_deposit/core/scanning/__init__.py +61 -0
  28. copyright_deposit/core/scanning/secrets.py +181 -0
  29. copyright_deposit/core/scanning/thirdparty.py +190 -0
  30. copyright_deposit/core/strip/__init__.py +337 -0
  31. copyright_deposit/core/strip/cfamily_strip.py +235 -0
  32. copyright_deposit/core/strip/pygments_strip.py +85 -0
  33. copyright_deposit/core/strip/python_strip.py +131 -0
  34. copyright_deposit/gui/__init__.py +1 -0
  35. copyright_deposit/gui/app.py +34 -0
  36. copyright_deposit/gui/branding.py +83 -0
  37. copyright_deposit/gui/history.py +192 -0
  38. copyright_deposit/gui/main_window.py +617 -0
  39. copyright_deposit/gui/panels/__init__.py +1 -0
  40. copyright_deposit/gui/panels/estimate.py +166 -0
  41. copyright_deposit/gui/panels/files.py +635 -0
  42. copyright_deposit/gui/panels/identification.py +193 -0
  43. copyright_deposit/gui/panels/options.py +445 -0
  44. copyright_deposit/gui/panels/preflight.py +260 -0
  45. copyright_deposit/gui/workers.py +96 -0
@@ -0,0 +1,349 @@
1
+ """File ordering: the user's list, resolved; or one suggested from the code.
2
+
3
+ Order matters legally. The Compendium asks for the first and last 25 pages
4
+ of the program, so whatever sits at the front of the deposit is what the
5
+ Office actually reads. The resolver therefore never guesses silently: an
6
+ entry that could mean two different files is reported as an ambiguity.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import ast
12
+ import re
13
+ from dataclasses import dataclass, field
14
+ from pathlib import PurePosixPath
15
+
16
+ from .discovery import DiscoveredFile, default_sort_key
17
+
18
+ SOURCE_LISTED = "listed"
19
+ SOURCE_GLOB = "glob"
20
+ SOURCE_UNLISTED = "unlisted"
21
+
22
+ _GLOB_CHARS = set("*?[")
23
+
24
+ # Filenames that conventionally begin a program.
25
+ _ENTRY_NAMES = (
26
+ "__main__.py", "main.py", "app.py", "run.py", "manage.py", "cli.py",
27
+ "main.cpp", "main.c", "main.cc", "winmain.cpp", "program.cs", "main.go",
28
+ "main.rs", "index.js", "index.ts", "main.js", "main.ts", "main.java",
29
+ )
30
+ _ENTRY_PATTERNS = (
31
+ re.compile(r"^\s*if\s+__name__\s*==\s*['\"]__main__['\"]", re.MULTILINE),
32
+ re.compile(r"\b(?:int|void)\s+(?:main|WinMain|wmain)\s*\(", re.MULTILINE),
33
+ re.compile(r"\bfunc\s+main\s*\(", re.MULTILINE),
34
+ re.compile(r"\bfn\s+main\s*\(", re.MULTILINE),
35
+ re.compile(r"\bstatic\s+void\s+Main\s*\(", re.MULTILINE),
36
+ )
37
+
38
+ _INCLUDE_RE = re.compile(r'^\s*#\s*include\s+"([^"]+)"', re.MULTILINE)
39
+
40
+ _HEADER_SUFFIXES = (".h", ".hpp", ".hh", ".hxx")
41
+ _IMPL_SUFFIXES = (".c", ".cpp", ".cc", ".cxx", ".m", ".mm")
42
+
43
+
44
+ @dataclass
45
+ class OrderedItem:
46
+ rel_path: str
47
+ source: str = SOURCE_LISTED
48
+ entry: str = "" # the order-list line that produced it
49
+
50
+
51
+ @dataclass
52
+ class Ambiguity:
53
+ entry: str
54
+ candidates: list[str]
55
+ chosen: str = ""
56
+
57
+
58
+ @dataclass
59
+ class OrderPlan:
60
+ items: list[OrderedItem] = field(default_factory=list)
61
+ ambiguities: list[Ambiguity] = field(default_factory=list)
62
+ unmatched: list[str] = field(default_factory=list)
63
+ excluded: list[str] = field(default_factory=list)
64
+ warnings: list[str] = field(default_factory=list)
65
+
66
+ @property
67
+ def paths(self) -> list[str]:
68
+ return [item.rel_path for item in self.items]
69
+
70
+
71
+ # ---------------------------------------------------------------------------
72
+ # Order-list text format
73
+ # ---------------------------------------------------------------------------
74
+
75
+
76
+ def parse_order_text(text: str) -> list[str]:
77
+ """One entry per line; blank lines and '#' comments ignored."""
78
+ entries: list[str] = []
79
+ for raw in text.splitlines():
80
+ line = raw.strip()
81
+ if not line or line.startswith("#"):
82
+ continue
83
+ entries.append(line.replace("\\", "/"))
84
+ return entries
85
+
86
+
87
+ def format_order_text(paths: list[str], header: bool = True) -> str:
88
+ lines: list[str] = []
89
+ if header:
90
+ lines.append("# Deposit file order - one path per line, top to bottom.")
91
+ lines.append("# Blank lines and lines starting with '#' are ignored.")
92
+ lines.append("")
93
+ lines.extend(paths)
94
+ return "\n".join(lines) + "\n"
95
+
96
+
97
+ # ---------------------------------------------------------------------------
98
+ # Resolution
99
+ # ---------------------------------------------------------------------------
100
+
101
+
102
+ def _match_entry(entry: str, files: list[DiscoveredFile]) -> tuple[list[str], str]:
103
+ """Return (matching rel paths, match kind) for one order-list entry."""
104
+ normalized = entry.strip().replace("\\", "/").lstrip("./")
105
+ by_path = {f.rel_path: f for f in files}
106
+
107
+ if normalized in by_path:
108
+ return [normalized], SOURCE_LISTED
109
+
110
+ lowered = normalized.lower()
111
+ exact_ci = [f.rel_path for f in files if f.rel_path.lower() == lowered]
112
+ if exact_ci:
113
+ return exact_ci, SOURCE_LISTED
114
+
115
+ if _GLOB_CHARS & set(normalized):
116
+ import fnmatch
117
+
118
+ hits = [f.rel_path for f in files if fnmatch.fnmatch(f.rel_path, normalized)]
119
+ if not hits:
120
+ hits = [f.rel_path for f in files if fnmatch.fnmatch(f.rel_path.lower(), lowered)]
121
+ return sorted(hits), SOURCE_GLOB
122
+
123
+ # Bare filename, or a trailing path fragment such as "core/layout.py".
124
+ suffix = "/" + normalized
125
+ hits = [
126
+ f.rel_path
127
+ for f in files
128
+ if f.rel_path == normalized or f.rel_path.endswith(suffix)
129
+ ]
130
+ if not hits:
131
+ hits = [
132
+ f.rel_path
133
+ for f in files
134
+ if f.rel_path.lower() == lowered or f.rel_path.lower().endswith(suffix.lower())
135
+ ]
136
+ return sorted(hits), SOURCE_LISTED
137
+
138
+
139
+ def resolve_order(
140
+ entries: list[str],
141
+ files: list[DiscoveredFile],
142
+ *,
143
+ include_unlisted: bool = True,
144
+ excluded: list[str] | None = None,
145
+ disambiguations: dict[str, str] | None = None,
146
+ ) -> OrderPlan:
147
+ """Turn order-list entries plus discovered files into a final sequence."""
148
+ plan = OrderPlan()
149
+ excluded_set = {p.replace("\\", "/") for p in (excluded or [])}
150
+ disambiguations = disambiguations or {}
151
+
152
+ available = [f for f in files if f.rel_path not in excluded_set]
153
+ plan.excluded = sorted(excluded_set & {f.rel_path for f in files})
154
+
155
+ placed: set[str] = set()
156
+
157
+ for entry in entries:
158
+ hits, kind = _match_entry(entry, available)
159
+ hits = [h for h in hits if h not in placed]
160
+ if not hits:
161
+ if any(_match_entry(entry, files)[0]):
162
+ plan.warnings.append(f"Order entry '{entry}' matched only excluded or already-placed files.")
163
+ else:
164
+ plan.unmatched.append(entry)
165
+ continue
166
+
167
+ if len(hits) > 1 and kind != SOURCE_GLOB:
168
+ chosen = disambiguations.get(entry)
169
+ if chosen and chosen in hits:
170
+ plan.ambiguities.append(Ambiguity(entry, hits, chosen))
171
+ hits = [chosen]
172
+ else:
173
+ plan.ambiguities.append(Ambiguity(entry, hits))
174
+ plan.warnings.append(
175
+ f"Order entry '{entry}' matches {len(hits)} files; all were included in path order."
176
+ )
177
+
178
+ for path in hits:
179
+ if path in placed:
180
+ continue
181
+ placed.add(path)
182
+ plan.items.append(OrderedItem(path, kind, entry))
183
+
184
+ if include_unlisted:
185
+ leftovers = [f for f in available if f.rel_path not in placed]
186
+ for f in sorted(leftovers, key=default_sort_key):
187
+ plan.items.append(OrderedItem(f.rel_path, SOURCE_UNLISTED))
188
+
189
+ return plan
190
+
191
+
192
+ # ---------------------------------------------------------------------------
193
+ # Auto-suggested order
194
+ # ---------------------------------------------------------------------------
195
+
196
+
197
+ def _python_module_index(files: list[DiscoveredFile]) -> dict[str, str]:
198
+ """Map dotted module names to relative paths."""
199
+ index: dict[str, str] = {}
200
+ for f in files:
201
+ if not f.rel_path.endswith((".py", ".pyi")):
202
+ continue
203
+ parts = list(PurePosixPath(f.rel_path).with_suffix("").parts)
204
+ if parts and parts[-1] == "__init__":
205
+ parts.pop()
206
+ if not parts:
207
+ continue
208
+ dotted = ".".join(parts)
209
+ index.setdefault(dotted, f.rel_path)
210
+ # Also index without the top-level package so 'core.layout' resolves
211
+ # inside a src/ or package-rooted layout.
212
+ for start in range(1, len(parts)):
213
+ index.setdefault(".".join(parts[start:]), f.rel_path)
214
+ return index
215
+
216
+
217
+ def _python_dependencies(text: str, rel_path: str, index: dict[str, str]) -> list[str]:
218
+ try:
219
+ tree = ast.parse(text)
220
+ except SyntaxError:
221
+ return []
222
+ package_parts = list(PurePosixPath(rel_path).parent.parts)
223
+ deps: list[str] = []
224
+
225
+ def add(name: str) -> None:
226
+ target = index.get(name)
227
+ if target and target != rel_path and target not in deps:
228
+ deps.append(target)
229
+
230
+ for node in ast.walk(tree):
231
+ if isinstance(node, ast.Import):
232
+ for alias in node.names:
233
+ add(alias.name)
234
+ elif isinstance(node, ast.ImportFrom):
235
+ if node.level:
236
+ base = package_parts[: len(package_parts) - node.level + 1]
237
+ prefix = ".".join(base)
238
+ module = f"{prefix}.{node.module}" if node.module else prefix
239
+ else:
240
+ module = node.module or ""
241
+ if module:
242
+ add(module)
243
+ for alias in node.names:
244
+ add(f"{module}.{alias.name}" if module else alias.name)
245
+ return deps
246
+
247
+
248
+ def _include_dependencies(text: str, rel_path: str, files_by_path: dict[str, str]) -> list[str]:
249
+ """Resolve #include "..." relative to the including file, then to the root."""
250
+ here = PurePosixPath(rel_path).parent
251
+ deps: list[str] = []
252
+ for match in _INCLUDE_RE.finditer(text):
253
+ target = match.group(1).replace("\\", "/")
254
+ candidates = [str(here / target).replace("\\", "/").lstrip("./"), target]
255
+ resolved = next((c for c in candidates if c in files_by_path), None)
256
+ if resolved is None:
257
+ suffix = "/" + PurePosixPath(target).name
258
+ resolved = next(
259
+ (p for p in files_by_path if p.endswith(suffix)),
260
+ None,
261
+ )
262
+ if resolved and resolved != rel_path and resolved not in deps:
263
+ deps.append(resolved)
264
+ return deps
265
+
266
+
267
+ def _paired_header(rel_path: str, files_by_path: dict[str, str]) -> str | None:
268
+ """The declaration that belongs with an implementation file."""
269
+ p = PurePosixPath(rel_path)
270
+ if p.suffix.lower() not in _IMPL_SUFFIXES:
271
+ return None
272
+ for suffix in _HEADER_SUFFIXES:
273
+ candidate = str(p.with_suffix(suffix)).replace("\\", "/")
274
+ if candidate in files_by_path:
275
+ return candidate
276
+ return None
277
+
278
+
279
+ def suggest_order(
280
+ files: list[DiscoveredFile],
281
+ read_text=None,
282
+ ) -> list[str]:
283
+ """Propose an order: entry points first, then their dependencies.
284
+
285
+ Gives the deposit the 'objectively identifiable beginning and end' the
286
+ Compendium asks for, instead of an arbitrary alphabetical walk.
287
+ """
288
+ if not files:
289
+ return []
290
+ if read_text is None:
291
+ from .encoding import read_source
292
+
293
+ def read_text(f: DiscoveredFile) -> str: # type: ignore[misc]
294
+ try:
295
+ return read_source(f.abs_path).text
296
+ except OSError:
297
+ return ""
298
+
299
+ files_by_path = {f.rel_path: f.abs_path for f in files}
300
+ texts: dict[str, str] = {}
301
+ for f in files:
302
+ texts[f.rel_path] = read_text(f)
303
+
304
+ module_index = _python_module_index(files)
305
+
306
+ dependencies: dict[str, list[str]] = {}
307
+ for f in files:
308
+ text = texts.get(f.rel_path, "")
309
+ if f.rel_path.endswith((".py", ".pyi")):
310
+ dependencies[f.rel_path] = _python_dependencies(text, f.rel_path, module_index)
311
+ else:
312
+ dependencies[f.rel_path] = _include_dependencies(text, f.rel_path, files_by_path)
313
+
314
+ # Rank entry points: known filenames first, then a main() signature.
315
+ def entry_rank(f: DiscoveredFile) -> tuple[int, int, str]:
316
+ name = PurePosixPath(f.rel_path).name.lower()
317
+ depth = len(PurePosixPath(f.rel_path).parts) - 1
318
+ if name in _ENTRY_NAMES:
319
+ return (_ENTRY_NAMES.index(name), depth, f.rel_path.lower())
320
+ text = texts.get(f.rel_path, "")
321
+ if any(p.search(text) for p in _ENTRY_PATTERNS):
322
+ return (len(_ENTRY_NAMES), depth, f.rel_path.lower())
323
+ return (len(_ENTRY_NAMES) + 1, depth, f.rel_path.lower())
324
+
325
+ ordered: list[str] = []
326
+ seen: set[str] = set()
327
+
328
+ def visit(path: str, stack: set[str]) -> None:
329
+ if path in seen or path in stack:
330
+ return
331
+ stack.add(path)
332
+ header = _paired_header(path, files_by_path)
333
+ if header and header not in seen:
334
+ visit(header, stack)
335
+ seen.add(path)
336
+ ordered.append(path)
337
+ for dep in dependencies.get(path, []):
338
+ visit(dep, stack)
339
+ stack.discard(path)
340
+
341
+ for f in sorted(files, key=entry_rank):
342
+ visit(f.rel_path, set())
343
+
344
+ # Anything unreachable (no entry point referenced it) keeps stable order.
345
+ for f in sorted(files, key=default_sort_key):
346
+ if f.rel_path not in seen:
347
+ seen.add(f.rel_path)
348
+ ordered.append(f.rel_path)
349
+ return ordered