codendium 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codendium-1.0.0.dist-info/METADATA +332 -0
- codendium-1.0.0.dist-info/RECORD +45 -0
- codendium-1.0.0.dist-info/WHEEL +5 -0
- codendium-1.0.0.dist-info/entry_points.txt +6 -0
- codendium-1.0.0.dist-info/licenses/LICENSE +201 -0
- codendium-1.0.0.dist-info/top_level.txt +1 -0
- copyright_deposit/__init__.py +15 -0
- copyright_deposit/__main__.py +34 -0
- copyright_deposit/assets/__init__.py +5 -0
- copyright_deposit/assets/fonts/README.md +31 -0
- copyright_deposit/assets/logo.svg +26 -0
- copyright_deposit/cli.py +314 -0
- copyright_deposit/config.py +269 -0
- copyright_deposit/core/__init__.py +1 -0
- copyright_deposit/core/deposit.py +117 -0
- copyright_deposit/core/discovery.py +260 -0
- copyright_deposit/core/encoding.py +136 -0
- copyright_deposit/core/languages.py +190 -0
- copyright_deposit/core/layout.py +349 -0
- copyright_deposit/core/lineranges.py +219 -0
- copyright_deposit/core/manifest.py +282 -0
- copyright_deposit/core/metrics.py +279 -0
- copyright_deposit/core/ordering.py +349 -0
- copyright_deposit/core/pipeline.py +386 -0
- copyright_deposit/core/redaction.py +162 -0
- copyright_deposit/core/render.py +242 -0
- copyright_deposit/core/scanning/__init__.py +61 -0
- copyright_deposit/core/scanning/secrets.py +181 -0
- copyright_deposit/core/scanning/thirdparty.py +190 -0
- copyright_deposit/core/strip/__init__.py +337 -0
- copyright_deposit/core/strip/cfamily_strip.py +235 -0
- copyright_deposit/core/strip/pygments_strip.py +85 -0
- copyright_deposit/core/strip/python_strip.py +131 -0
- copyright_deposit/gui/__init__.py +1 -0
- copyright_deposit/gui/app.py +34 -0
- copyright_deposit/gui/branding.py +83 -0
- copyright_deposit/gui/history.py +192 -0
- copyright_deposit/gui/main_window.py +617 -0
- copyright_deposit/gui/panels/__init__.py +1 -0
- copyright_deposit/gui/panels/estimate.py +166 -0
- copyright_deposit/gui/panels/files.py +635 -0
- copyright_deposit/gui/panels/identification.py +193 -0
- copyright_deposit/gui/panels/options.py +445 -0
- copyright_deposit/gui/panels/preflight.py +260 -0
- copyright_deposit/gui/workers.py +96 -0
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
"""File ordering: the user's list, resolved; or one suggested from the code.
|
|
2
|
+
|
|
3
|
+
Order matters legally. The Compendium asks for the first and last 25 pages
|
|
4
|
+
of the program, so whatever sits at the front of the deposit is what the
|
|
5
|
+
Office actually reads. The resolver therefore never guesses silently: an
|
|
6
|
+
entry that could mean two different files is reported as an ambiguity.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import ast
|
|
12
|
+
import re
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import PurePosixPath
|
|
15
|
+
|
|
16
|
+
from .discovery import DiscoveredFile, default_sort_key
|
|
17
|
+
|
|
18
|
+
SOURCE_LISTED = "listed"
|
|
19
|
+
SOURCE_GLOB = "glob"
|
|
20
|
+
SOURCE_UNLISTED = "unlisted"
|
|
21
|
+
|
|
22
|
+
_GLOB_CHARS = set("*?[")
|
|
23
|
+
|
|
24
|
+
# Filenames that conventionally begin a program.
|
|
25
|
+
_ENTRY_NAMES = (
|
|
26
|
+
"__main__.py", "main.py", "app.py", "run.py", "manage.py", "cli.py",
|
|
27
|
+
"main.cpp", "main.c", "main.cc", "winmain.cpp", "program.cs", "main.go",
|
|
28
|
+
"main.rs", "index.js", "index.ts", "main.js", "main.ts", "main.java",
|
|
29
|
+
)
|
|
30
|
+
_ENTRY_PATTERNS = (
|
|
31
|
+
re.compile(r"^\s*if\s+__name__\s*==\s*['\"]__main__['\"]", re.MULTILINE),
|
|
32
|
+
re.compile(r"\b(?:int|void)\s+(?:main|WinMain|wmain)\s*\(", re.MULTILINE),
|
|
33
|
+
re.compile(r"\bfunc\s+main\s*\(", re.MULTILINE),
|
|
34
|
+
re.compile(r"\bfn\s+main\s*\(", re.MULTILINE),
|
|
35
|
+
re.compile(r"\bstatic\s+void\s+Main\s*\(", re.MULTILINE),
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
_INCLUDE_RE = re.compile(r'^\s*#\s*include\s+"([^"]+)"', re.MULTILINE)
|
|
39
|
+
|
|
40
|
+
_HEADER_SUFFIXES = (".h", ".hpp", ".hh", ".hxx")
|
|
41
|
+
_IMPL_SUFFIXES = (".c", ".cpp", ".cc", ".cxx", ".m", ".mm")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass
|
|
45
|
+
class OrderedItem:
|
|
46
|
+
rel_path: str
|
|
47
|
+
source: str = SOURCE_LISTED
|
|
48
|
+
entry: str = "" # the order-list line that produced it
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class Ambiguity:
|
|
53
|
+
entry: str
|
|
54
|
+
candidates: list[str]
|
|
55
|
+
chosen: str = ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class OrderPlan:
|
|
60
|
+
items: list[OrderedItem] = field(default_factory=list)
|
|
61
|
+
ambiguities: list[Ambiguity] = field(default_factory=list)
|
|
62
|
+
unmatched: list[str] = field(default_factory=list)
|
|
63
|
+
excluded: list[str] = field(default_factory=list)
|
|
64
|
+
warnings: list[str] = field(default_factory=list)
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def paths(self) -> list[str]:
|
|
68
|
+
return [item.rel_path for item in self.items]
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# ---------------------------------------------------------------------------
|
|
72
|
+
# Order-list text format
|
|
73
|
+
# ---------------------------------------------------------------------------
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def parse_order_text(text: str) -> list[str]:
|
|
77
|
+
"""One entry per line; blank lines and '#' comments ignored."""
|
|
78
|
+
entries: list[str] = []
|
|
79
|
+
for raw in text.splitlines():
|
|
80
|
+
line = raw.strip()
|
|
81
|
+
if not line or line.startswith("#"):
|
|
82
|
+
continue
|
|
83
|
+
entries.append(line.replace("\\", "/"))
|
|
84
|
+
return entries
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def format_order_text(paths: list[str], header: bool = True) -> str:
|
|
88
|
+
lines: list[str] = []
|
|
89
|
+
if header:
|
|
90
|
+
lines.append("# Deposit file order - one path per line, top to bottom.")
|
|
91
|
+
lines.append("# Blank lines and lines starting with '#' are ignored.")
|
|
92
|
+
lines.append("")
|
|
93
|
+
lines.extend(paths)
|
|
94
|
+
return "\n".join(lines) + "\n"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# ---------------------------------------------------------------------------
|
|
98
|
+
# Resolution
|
|
99
|
+
# ---------------------------------------------------------------------------
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _match_entry(entry: str, files: list[DiscoveredFile]) -> tuple[list[str], str]:
|
|
103
|
+
"""Return (matching rel paths, match kind) for one order-list entry."""
|
|
104
|
+
normalized = entry.strip().replace("\\", "/").lstrip("./")
|
|
105
|
+
by_path = {f.rel_path: f for f in files}
|
|
106
|
+
|
|
107
|
+
if normalized in by_path:
|
|
108
|
+
return [normalized], SOURCE_LISTED
|
|
109
|
+
|
|
110
|
+
lowered = normalized.lower()
|
|
111
|
+
exact_ci = [f.rel_path for f in files if f.rel_path.lower() == lowered]
|
|
112
|
+
if exact_ci:
|
|
113
|
+
return exact_ci, SOURCE_LISTED
|
|
114
|
+
|
|
115
|
+
if _GLOB_CHARS & set(normalized):
|
|
116
|
+
import fnmatch
|
|
117
|
+
|
|
118
|
+
hits = [f.rel_path for f in files if fnmatch.fnmatch(f.rel_path, normalized)]
|
|
119
|
+
if not hits:
|
|
120
|
+
hits = [f.rel_path for f in files if fnmatch.fnmatch(f.rel_path.lower(), lowered)]
|
|
121
|
+
return sorted(hits), SOURCE_GLOB
|
|
122
|
+
|
|
123
|
+
# Bare filename, or a trailing path fragment such as "core/layout.py".
|
|
124
|
+
suffix = "/" + normalized
|
|
125
|
+
hits = [
|
|
126
|
+
f.rel_path
|
|
127
|
+
for f in files
|
|
128
|
+
if f.rel_path == normalized or f.rel_path.endswith(suffix)
|
|
129
|
+
]
|
|
130
|
+
if not hits:
|
|
131
|
+
hits = [
|
|
132
|
+
f.rel_path
|
|
133
|
+
for f in files
|
|
134
|
+
if f.rel_path.lower() == lowered or f.rel_path.lower().endswith(suffix.lower())
|
|
135
|
+
]
|
|
136
|
+
return sorted(hits), SOURCE_LISTED
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def resolve_order(
|
|
140
|
+
entries: list[str],
|
|
141
|
+
files: list[DiscoveredFile],
|
|
142
|
+
*,
|
|
143
|
+
include_unlisted: bool = True,
|
|
144
|
+
excluded: list[str] | None = None,
|
|
145
|
+
disambiguations: dict[str, str] | None = None,
|
|
146
|
+
) -> OrderPlan:
|
|
147
|
+
"""Turn order-list entries plus discovered files into a final sequence."""
|
|
148
|
+
plan = OrderPlan()
|
|
149
|
+
excluded_set = {p.replace("\\", "/") for p in (excluded or [])}
|
|
150
|
+
disambiguations = disambiguations or {}
|
|
151
|
+
|
|
152
|
+
available = [f for f in files if f.rel_path not in excluded_set]
|
|
153
|
+
plan.excluded = sorted(excluded_set & {f.rel_path for f in files})
|
|
154
|
+
|
|
155
|
+
placed: set[str] = set()
|
|
156
|
+
|
|
157
|
+
for entry in entries:
|
|
158
|
+
hits, kind = _match_entry(entry, available)
|
|
159
|
+
hits = [h for h in hits if h not in placed]
|
|
160
|
+
if not hits:
|
|
161
|
+
if any(_match_entry(entry, files)[0]):
|
|
162
|
+
plan.warnings.append(f"Order entry '{entry}' matched only excluded or already-placed files.")
|
|
163
|
+
else:
|
|
164
|
+
plan.unmatched.append(entry)
|
|
165
|
+
continue
|
|
166
|
+
|
|
167
|
+
if len(hits) > 1 and kind != SOURCE_GLOB:
|
|
168
|
+
chosen = disambiguations.get(entry)
|
|
169
|
+
if chosen and chosen in hits:
|
|
170
|
+
plan.ambiguities.append(Ambiguity(entry, hits, chosen))
|
|
171
|
+
hits = [chosen]
|
|
172
|
+
else:
|
|
173
|
+
plan.ambiguities.append(Ambiguity(entry, hits))
|
|
174
|
+
plan.warnings.append(
|
|
175
|
+
f"Order entry '{entry}' matches {len(hits)} files; all were included in path order."
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
for path in hits:
|
|
179
|
+
if path in placed:
|
|
180
|
+
continue
|
|
181
|
+
placed.add(path)
|
|
182
|
+
plan.items.append(OrderedItem(path, kind, entry))
|
|
183
|
+
|
|
184
|
+
if include_unlisted:
|
|
185
|
+
leftovers = [f for f in available if f.rel_path not in placed]
|
|
186
|
+
for f in sorted(leftovers, key=default_sort_key):
|
|
187
|
+
plan.items.append(OrderedItem(f.rel_path, SOURCE_UNLISTED))
|
|
188
|
+
|
|
189
|
+
return plan
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# ---------------------------------------------------------------------------
|
|
193
|
+
# Auto-suggested order
|
|
194
|
+
# ---------------------------------------------------------------------------
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _python_module_index(files: list[DiscoveredFile]) -> dict[str, str]:
|
|
198
|
+
"""Map dotted module names to relative paths."""
|
|
199
|
+
index: dict[str, str] = {}
|
|
200
|
+
for f in files:
|
|
201
|
+
if not f.rel_path.endswith((".py", ".pyi")):
|
|
202
|
+
continue
|
|
203
|
+
parts = list(PurePosixPath(f.rel_path).with_suffix("").parts)
|
|
204
|
+
if parts and parts[-1] == "__init__":
|
|
205
|
+
parts.pop()
|
|
206
|
+
if not parts:
|
|
207
|
+
continue
|
|
208
|
+
dotted = ".".join(parts)
|
|
209
|
+
index.setdefault(dotted, f.rel_path)
|
|
210
|
+
# Also index without the top-level package so 'core.layout' resolves
|
|
211
|
+
# inside a src/ or package-rooted layout.
|
|
212
|
+
for start in range(1, len(parts)):
|
|
213
|
+
index.setdefault(".".join(parts[start:]), f.rel_path)
|
|
214
|
+
return index
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _python_dependencies(text: str, rel_path: str, index: dict[str, str]) -> list[str]:
|
|
218
|
+
try:
|
|
219
|
+
tree = ast.parse(text)
|
|
220
|
+
except SyntaxError:
|
|
221
|
+
return []
|
|
222
|
+
package_parts = list(PurePosixPath(rel_path).parent.parts)
|
|
223
|
+
deps: list[str] = []
|
|
224
|
+
|
|
225
|
+
def add(name: str) -> None:
|
|
226
|
+
target = index.get(name)
|
|
227
|
+
if target and target != rel_path and target not in deps:
|
|
228
|
+
deps.append(target)
|
|
229
|
+
|
|
230
|
+
for node in ast.walk(tree):
|
|
231
|
+
if isinstance(node, ast.Import):
|
|
232
|
+
for alias in node.names:
|
|
233
|
+
add(alias.name)
|
|
234
|
+
elif isinstance(node, ast.ImportFrom):
|
|
235
|
+
if node.level:
|
|
236
|
+
base = package_parts[: len(package_parts) - node.level + 1]
|
|
237
|
+
prefix = ".".join(base)
|
|
238
|
+
module = f"{prefix}.{node.module}" if node.module else prefix
|
|
239
|
+
else:
|
|
240
|
+
module = node.module or ""
|
|
241
|
+
if module:
|
|
242
|
+
add(module)
|
|
243
|
+
for alias in node.names:
|
|
244
|
+
add(f"{module}.{alias.name}" if module else alias.name)
|
|
245
|
+
return deps
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _include_dependencies(text: str, rel_path: str, files_by_path: dict[str, str]) -> list[str]:
|
|
249
|
+
"""Resolve #include "..." relative to the including file, then to the root."""
|
|
250
|
+
here = PurePosixPath(rel_path).parent
|
|
251
|
+
deps: list[str] = []
|
|
252
|
+
for match in _INCLUDE_RE.finditer(text):
|
|
253
|
+
target = match.group(1).replace("\\", "/")
|
|
254
|
+
candidates = [str(here / target).replace("\\", "/").lstrip("./"), target]
|
|
255
|
+
resolved = next((c for c in candidates if c in files_by_path), None)
|
|
256
|
+
if resolved is None:
|
|
257
|
+
suffix = "/" + PurePosixPath(target).name
|
|
258
|
+
resolved = next(
|
|
259
|
+
(p for p in files_by_path if p.endswith(suffix)),
|
|
260
|
+
None,
|
|
261
|
+
)
|
|
262
|
+
if resolved and resolved != rel_path and resolved not in deps:
|
|
263
|
+
deps.append(resolved)
|
|
264
|
+
return deps
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _paired_header(rel_path: str, files_by_path: dict[str, str]) -> str | None:
|
|
268
|
+
"""The declaration that belongs with an implementation file."""
|
|
269
|
+
p = PurePosixPath(rel_path)
|
|
270
|
+
if p.suffix.lower() not in _IMPL_SUFFIXES:
|
|
271
|
+
return None
|
|
272
|
+
for suffix in _HEADER_SUFFIXES:
|
|
273
|
+
candidate = str(p.with_suffix(suffix)).replace("\\", "/")
|
|
274
|
+
if candidate in files_by_path:
|
|
275
|
+
return candidate
|
|
276
|
+
return None
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def suggest_order(
|
|
280
|
+
files: list[DiscoveredFile],
|
|
281
|
+
read_text=None,
|
|
282
|
+
) -> list[str]:
|
|
283
|
+
"""Propose an order: entry points first, then their dependencies.
|
|
284
|
+
|
|
285
|
+
Gives the deposit the 'objectively identifiable beginning and end' the
|
|
286
|
+
Compendium asks for, instead of an arbitrary alphabetical walk.
|
|
287
|
+
"""
|
|
288
|
+
if not files:
|
|
289
|
+
return []
|
|
290
|
+
if read_text is None:
|
|
291
|
+
from .encoding import read_source
|
|
292
|
+
|
|
293
|
+
def read_text(f: DiscoveredFile) -> str: # type: ignore[misc]
|
|
294
|
+
try:
|
|
295
|
+
return read_source(f.abs_path).text
|
|
296
|
+
except OSError:
|
|
297
|
+
return ""
|
|
298
|
+
|
|
299
|
+
files_by_path = {f.rel_path: f.abs_path for f in files}
|
|
300
|
+
texts: dict[str, str] = {}
|
|
301
|
+
for f in files:
|
|
302
|
+
texts[f.rel_path] = read_text(f)
|
|
303
|
+
|
|
304
|
+
module_index = _python_module_index(files)
|
|
305
|
+
|
|
306
|
+
dependencies: dict[str, list[str]] = {}
|
|
307
|
+
for f in files:
|
|
308
|
+
text = texts.get(f.rel_path, "")
|
|
309
|
+
if f.rel_path.endswith((".py", ".pyi")):
|
|
310
|
+
dependencies[f.rel_path] = _python_dependencies(text, f.rel_path, module_index)
|
|
311
|
+
else:
|
|
312
|
+
dependencies[f.rel_path] = _include_dependencies(text, f.rel_path, files_by_path)
|
|
313
|
+
|
|
314
|
+
# Rank entry points: known filenames first, then a main() signature.
|
|
315
|
+
def entry_rank(f: DiscoveredFile) -> tuple[int, int, str]:
|
|
316
|
+
name = PurePosixPath(f.rel_path).name.lower()
|
|
317
|
+
depth = len(PurePosixPath(f.rel_path).parts) - 1
|
|
318
|
+
if name in _ENTRY_NAMES:
|
|
319
|
+
return (_ENTRY_NAMES.index(name), depth, f.rel_path.lower())
|
|
320
|
+
text = texts.get(f.rel_path, "")
|
|
321
|
+
if any(p.search(text) for p in _ENTRY_PATTERNS):
|
|
322
|
+
return (len(_ENTRY_NAMES), depth, f.rel_path.lower())
|
|
323
|
+
return (len(_ENTRY_NAMES) + 1, depth, f.rel_path.lower())
|
|
324
|
+
|
|
325
|
+
ordered: list[str] = []
|
|
326
|
+
seen: set[str] = set()
|
|
327
|
+
|
|
328
|
+
def visit(path: str, stack: set[str]) -> None:
|
|
329
|
+
if path in seen or path in stack:
|
|
330
|
+
return
|
|
331
|
+
stack.add(path)
|
|
332
|
+
header = _paired_header(path, files_by_path)
|
|
333
|
+
if header and header not in seen:
|
|
334
|
+
visit(header, stack)
|
|
335
|
+
seen.add(path)
|
|
336
|
+
ordered.append(path)
|
|
337
|
+
for dep in dependencies.get(path, []):
|
|
338
|
+
visit(dep, stack)
|
|
339
|
+
stack.discard(path)
|
|
340
|
+
|
|
341
|
+
for f in sorted(files, key=entry_rank):
|
|
342
|
+
visit(f.rel_path, set())
|
|
343
|
+
|
|
344
|
+
# Anything unreachable (no entry point referenced it) keeps stable order.
|
|
345
|
+
for f in sorted(files, key=default_sort_key):
|
|
346
|
+
if f.rel_path not in seen:
|
|
347
|
+
seen.add(f.rel_path)
|
|
348
|
+
ordered.append(f.rel_path)
|
|
349
|
+
return ordered
|