codeecho 1.0.0__py3-none-any.whl → 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codeecho/__init__.py +22 -22
- codeecho/__main__.py +310 -289
- codeecho/db.py +289 -289
- codeecho/detector.py +288 -246
- codeecho/extractor.py +142 -139
- codeecho/fingerprint.py +38 -38
- codeecho/logging.ini +7 -1
- codeecho/models.py +53 -52
- codeecho/normalizer.py +519 -519
- codeecho/parser.py +95 -94
- codeecho/reporter/__init__.py +6 -6
- codeecho/reporter/html_reporter.py +212 -212
- codeecho/reporter/json_reporter.py +83 -82
- codeecho/scanner.py +126 -109
- {codeecho-1.0.0.dist-info → codeecho-1.1.0.dist-info}/METADATA +12 -11
- codeecho-1.1.0.dist-info/RECORD +20 -0
- {codeecho-1.0.0.dist-info → codeecho-1.1.0.dist-info}/WHEEL +1 -1
- {codeecho-1.0.0.dist-info → codeecho-1.1.0.dist-info}/licenses/LICENSE +21 -21
- codeecho-1.0.0.dist-info/RECORD +0 -20
- {codeecho-1.0.0.dist-info → codeecho-1.1.0.dist-info}/entry_points.txt +0 -0
codeecho/__init__.py
CHANGED
|
@@ -1,22 +1,22 @@
|
|
|
1
|
-
"""
|
|
2
|
-
codeecho - A developer tool that scans your codebase to detect and highlight ECHOES
|
|
3
|
-
of duplicated or near-duplicated code so you can refactor toward cleaner, more
|
|
4
|
-
maintainable designs.
|
|
5
|
-
"""
|
|
6
|
-
|
|
7
|
-
from env_dir_bootstrap import EnvDirBootstrap
|
|
8
|
-
from logenrich import setup_logger
|
|
9
|
-
|
|
10
|
-
__version__ = "1.
|
|
11
|
-
|
|
12
|
-
_bootstrapper = EnvDirBootstrap(
|
|
13
|
-
env_var="CODEECHO_CONFIG_DIR",
|
|
14
|
-
resources=["logging.ini", ".ignore"],
|
|
15
|
-
package="codeecho",
|
|
16
|
-
)
|
|
17
|
-
|
|
18
|
-
_bootstrapper.setup()
|
|
19
|
-
|
|
20
|
-
CONF_DIR = str(_bootstrapper.get_dir())
|
|
21
|
-
|
|
22
|
-
setup_logger("codeecho", conf_dir=CONF_DIR)
|
|
1
|
+
"""
|
|
2
|
+
codeecho - A developer tool that scans your codebase to detect and highlight ECHOES
|
|
3
|
+
of duplicated or near-duplicated code so you can refactor toward cleaner, more
|
|
4
|
+
maintainable designs.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from env_dir_bootstrap import EnvDirBootstrap
|
|
8
|
+
from logenrich import setup_logger
|
|
9
|
+
|
|
10
|
+
__version__ = "1.1.0"
|
|
11
|
+
|
|
12
|
+
_bootstrapper = EnvDirBootstrap(
|
|
13
|
+
env_var="CODEECHO_CONFIG_DIR",
|
|
14
|
+
resources=["logging.ini", ".ignore"],
|
|
15
|
+
package="codeecho",
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
_bootstrapper.setup()
|
|
19
|
+
|
|
20
|
+
CONF_DIR = str(_bootstrapper.get_dir())
|
|
21
|
+
|
|
22
|
+
setup_logger("codeecho", conf_dir=CONF_DIR)
|
codeecho/__main__.py
CHANGED
|
@@ -1,289 +1,310 @@
|
|
|
1
|
-
"""
|
|
2
|
-
CLI entry point for codeecho.
|
|
3
|
-
|
|
4
|
-
Invoked via::
|
|
5
|
-
|
|
6
|
-
poetry run python -m codeecho [OPTIONS] PATH
|
|
7
|
-
"""
|
|
8
|
-
|
|
9
|
-
import
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
import
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
from
|
|
16
|
-
from rich.
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
from
|
|
26
|
-
|
|
27
|
-
from codeecho
|
|
28
|
-
from codeecho
|
|
29
|
-
from codeecho.
|
|
30
|
-
from codeecho.
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
"
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
"
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
"
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
"
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
"""
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
"
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
# ── Phase
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
)
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
)
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
""
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
1
|
+
"""
|
|
2
|
+
CLI entry point for codeecho.
|
|
3
|
+
|
|
4
|
+
Invoked via::
|
|
5
|
+
|
|
6
|
+
poetry run python -m codeecho [OPTIONS] PATH
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import uuid
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import click
|
|
15
|
+
from braincraft import IgnoreFile
|
|
16
|
+
from rich.console import Console
|
|
17
|
+
from rich.panel import Panel
|
|
18
|
+
from rich.progress import (
|
|
19
|
+
BarColumn,
|
|
20
|
+
MofNCompleteColumn,
|
|
21
|
+
Progress,
|
|
22
|
+
SpinnerColumn,
|
|
23
|
+
TextColumn,
|
|
24
|
+
)
|
|
25
|
+
from rich.table import Table
|
|
26
|
+
|
|
27
|
+
from codeecho import __version__, CONF_DIR
|
|
28
|
+
from codeecho import extractor, fingerprint, parser as ts_parser, scanner
|
|
29
|
+
from codeecho.db import SessionDB, get_db_path
|
|
30
|
+
from codeecho.detector import detect
|
|
31
|
+
from codeecho.models import ScanResult
|
|
32
|
+
from codeecho.reporter import html_reporter, json_reporter
|
|
33
|
+
|
|
34
|
+
_console = Console()
|
|
35
|
+
|
|
36
|
+
_FORMAT_CHOICES = click.Choice(["json", "html", "both"])
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _parse_types(types_str: str) -> set[int]:
|
|
40
|
+
"""Parse ``--types`` value into a set of integers."""
|
|
41
|
+
if types_str.strip().lower() == "all":
|
|
42
|
+
return {1, 2, 3}
|
|
43
|
+
result: set[int] = set()
|
|
44
|
+
for token in types_str.split(","):
|
|
45
|
+
token = token.strip()
|
|
46
|
+
if token.isdigit() and token in {"1", "2", "3"}:
|
|
47
|
+
result.add(int(token))
|
|
48
|
+
if not result:
|
|
49
|
+
raise click.BadParameter(
|
|
50
|
+
f"Invalid types value: {types_str!r}. Use 'all' or e.g. '1,2,3'."
|
|
51
|
+
)
|
|
52
|
+
return result
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@click.command(
|
|
56
|
+
name="codeecho", context_settings={"help_option_names": ["-h", "--help"]}
|
|
57
|
+
)
|
|
58
|
+
@click.version_option(
|
|
59
|
+
version=__version__, prog_name="codeecho", message="%(prog)s v%(version)s"
|
|
60
|
+
)
|
|
61
|
+
@click.argument(
|
|
62
|
+
"paths",
|
|
63
|
+
nargs=-1,
|
|
64
|
+
type=click.Path(exists=True, file_okay=True, dir_okay=True, path_type=Path),
|
|
65
|
+
)
|
|
66
|
+
@click.option(
|
|
67
|
+
"--types",
|
|
68
|
+
default="all",
|
|
69
|
+
show_default=True,
|
|
70
|
+
metavar="TYPES",
|
|
71
|
+
help="Clone types to detect: comma-separated (e.g. '1,2') or 'all'.",
|
|
72
|
+
)
|
|
73
|
+
@click.option(
|
|
74
|
+
"--threshold",
|
|
75
|
+
default=0.8,
|
|
76
|
+
show_default=True,
|
|
77
|
+
type=click.FloatRange(0.0, 1.0),
|
|
78
|
+
help="Jaccard similarity threshold for Type-3 (near-duplicate) detection.",
|
|
79
|
+
)
|
|
80
|
+
@click.option(
|
|
81
|
+
"--output",
|
|
82
|
+
default="codeecho-output",
|
|
83
|
+
show_default=True,
|
|
84
|
+
metavar="NAME",
|
|
85
|
+
help="Base name (without extension) for output file(s).",
|
|
86
|
+
)
|
|
87
|
+
@click.option(
|
|
88
|
+
"--output-dir",
|
|
89
|
+
default=None,
|
|
90
|
+
show_default=False,
|
|
91
|
+
type=click.Path(file_okay=False, path_type=Path),
|
|
92
|
+
help="Directory where output file(s) will be written. [default: <cwd>/reports]",
|
|
93
|
+
)
|
|
94
|
+
@click.option(
|
|
95
|
+
"--db-dir",
|
|
96
|
+
"db_dir",
|
|
97
|
+
default=None,
|
|
98
|
+
show_default=False,
|
|
99
|
+
type=click.Path(file_okay=False, path_type=Path),
|
|
100
|
+
help="Directory for the SQLite scratch database. [default: ~/.codeecho]",
|
|
101
|
+
)
|
|
102
|
+
@click.option(
|
|
103
|
+
"--format",
|
|
104
|
+
"fmt",
|
|
105
|
+
default="both",
|
|
106
|
+
show_default=True,
|
|
107
|
+
type=_FORMAT_CHOICES,
|
|
108
|
+
help="Output format.",
|
|
109
|
+
)
|
|
110
|
+
@click.option(
|
|
111
|
+
"--min-tokens",
|
|
112
|
+
default=10,
|
|
113
|
+
show_default=True,
|
|
114
|
+
type=click.IntRange(1),
|
|
115
|
+
help="Minimum token count for a fragment to be considered.",
|
|
116
|
+
)
|
|
117
|
+
@click.option(
|
|
118
|
+
"--exclude",
|
|
119
|
+
multiple=True,
|
|
120
|
+
metavar="PATTERN",
|
|
121
|
+
help="Glob pattern(s) to exclude from scanning (repeatable).",
|
|
122
|
+
)
|
|
123
|
+
def main( # pylint: disable=too-many-arguments,too-many-positional-arguments,too-many-locals,invalid-name
|
|
124
|
+
paths: tuple[Path, ...],
|
|
125
|
+
types: str,
|
|
126
|
+
threshold: float,
|
|
127
|
+
output: str,
|
|
128
|
+
output_dir: Path | None,
|
|
129
|
+
db_dir: Path | None,
|
|
130
|
+
fmt: str,
|
|
131
|
+
min_tokens: int,
|
|
132
|
+
exclude: tuple[str, ...],
|
|
133
|
+
) -> None:
|
|
134
|
+
"""Scan one or more PATH(s) for duplicate and near-duplicate code.
|
|
135
|
+
|
|
136
|
+
Each PATH may be a file or a directory. Generates JSON and/or HTML reports,
|
|
137
|
+
then removes the intermediate session data from the embedded database.
|
|
138
|
+
"""
|
|
139
|
+
_console.print(
|
|
140
|
+
Panel(
|
|
141
|
+
f"[bold cyan]codeecho[/bold cyan] [dim]v{__version__}[/dim] — Code Duplicate Scanner",
|
|
142
|
+
border_style="dim",
|
|
143
|
+
)
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
if not paths:
|
|
147
|
+
raise click.UsageError("At least one PATH is required.")
|
|
148
|
+
|
|
149
|
+
detect_types = _parse_types(types)
|
|
150
|
+
if output_dir is None:
|
|
151
|
+
output_dir = Path.cwd() / "reports"
|
|
152
|
+
session_id = str(uuid.uuid4())
|
|
153
|
+
config = {
|
|
154
|
+
"types": types,
|
|
155
|
+
"threshold": threshold,
|
|
156
|
+
"min_tokens": min_tokens,
|
|
157
|
+
"exclude": list(exclude),
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
161
|
+
|
|
162
|
+
resolved = [p.resolve() for p in paths]
|
|
163
|
+
|
|
164
|
+
db_path = Path(get_db_path(str(db_dir) if db_dir else None))
|
|
165
|
+
with SessionDB(db_path=db_path) as session_db:
|
|
166
|
+
session_db.create_session(
|
|
167
|
+
session_id, json.dumps([str(p) for p in resolved]), config
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
# ── Phase 1: File discovery ─────────────────────────────────────────
|
|
171
|
+
for p in resolved:
|
|
172
|
+
_console.print(f"[dim]Scanning:[/dim] [bold]{p}[/bold]")
|
|
173
|
+
try:
|
|
174
|
+
base_dir = Path(os.path.commonpath(resolved))
|
|
175
|
+
if not base_dir.is_dir():
|
|
176
|
+
base_dir = base_dir.parent
|
|
177
|
+
except ValueError:
|
|
178
|
+
base_dir = Path.cwd()
|
|
179
|
+
_ignore_file = IgnoreFile(Path(CONF_DIR) / ".ignore", base_dir=base_dir)
|
|
180
|
+
files = scanner.scan(
|
|
181
|
+
tuple(resolved), exclude_patterns=exclude, ignore_file=_ignore_file
|
|
182
|
+
)
|
|
183
|
+
total_fragments = 0
|
|
184
|
+
|
|
185
|
+
if not files:
|
|
186
|
+
_console.print("[yellow]No supported source files found.[/yellow]")
|
|
187
|
+
session_db.delete_session(session_id)
|
|
188
|
+
return
|
|
189
|
+
|
|
190
|
+
# ── Phase 2: Parse → Extract → Hash ────────────────────────────────
|
|
191
|
+
total_fragments = _process_files(files, session_db, session_id, min_tokens)
|
|
192
|
+
|
|
193
|
+
# ── Phase 3: Clone detection ────────────────────────────────────────
|
|
194
|
+
cnt1, cnt2, cnt3 = _detect_clones(
|
|
195
|
+
session_db, session_id, detect_types, threshold
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
result = ScanResult(
|
|
199
|
+
session_id=session_id,
|
|
200
|
+
version=__version__,
|
|
201
|
+
scan_path=[str(p) for p in resolved],
|
|
202
|
+
files_scanned=len(files),
|
|
203
|
+
fragments_extracted=total_fragments,
|
|
204
|
+
type1_groups=cnt1,
|
|
205
|
+
type2_groups=cnt2,
|
|
206
|
+
type3_groups=cnt3,
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
# ── Phase 4: Report generation ──────────────────────────────────────
|
|
210
|
+
written = _write_reports(session_db, result, output_dir, output, fmt)
|
|
211
|
+
|
|
212
|
+
# ── Phase 5: Clean up session ───────────────────────────────────────
|
|
213
|
+
session_db.delete_session(session_id)
|
|
214
|
+
|
|
215
|
+
# ── Summary table (printed after DB is closed) ──────────────────────────
|
|
216
|
+
_print_summary(result, written)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _detect_clones(
|
|
220
|
+
session_db: SessionDB,
|
|
221
|
+
session_id: str,
|
|
222
|
+
detect_types: set[int],
|
|
223
|
+
threshold: float,
|
|
224
|
+
) -> tuple[int, int, int]:
|
|
225
|
+
"""Run clone detection and return (type1_count, type2_count, type3_count)."""
|
|
226
|
+
_console.print("[dim]Detecting clones…[/dim]")
|
|
227
|
+
return detect(session_db, session_id, detect_types, threshold)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _write_reports(
|
|
231
|
+
session_db: SessionDB,
|
|
232
|
+
result: ScanResult,
|
|
233
|
+
output_dir: Path,
|
|
234
|
+
output: str,
|
|
235
|
+
fmt: str,
|
|
236
|
+
) -> list[Path]:
|
|
237
|
+
"""Write the requested report formats and return a list of written paths."""
|
|
238
|
+
written: list[Path] = []
|
|
239
|
+
if fmt in ("json", "both"):
|
|
240
|
+
written.append(
|
|
241
|
+
json_reporter.write(session_db, result, output_dir / f"{output}.json")
|
|
242
|
+
)
|
|
243
|
+
if fmt in ("html", "both"):
|
|
244
|
+
written.append(
|
|
245
|
+
html_reporter.write(session_db, result, output_dir / f"{output}.html")
|
|
246
|
+
)
|
|
247
|
+
return written
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _print_summary(result: ScanResult, written: list[Path]) -> None:
|
|
251
|
+
"""Print the scan summary table and list of saved report paths."""
|
|
252
|
+
table = Table(title="Scan Summary", show_header=True, header_style="bold magenta")
|
|
253
|
+
table.add_column("Metric", style="dim", min_width=26)
|
|
254
|
+
table.add_column("Value", justify="right", style="bold")
|
|
255
|
+
table.add_row("Files scanned", str(result.files_scanned))
|
|
256
|
+
table.add_row("Fragments extracted", str(result.fragments_extracted))
|
|
257
|
+
table.add_row("[red]Type-1[/red] clone groups (exact)", str(result.type1_groups))
|
|
258
|
+
table.add_row(
|
|
259
|
+
"[yellow]Type-2[/yellow] clone groups (structural)", str(result.type2_groups)
|
|
260
|
+
)
|
|
261
|
+
table.add_row(
|
|
262
|
+
"[green]Type-3[/green] clone groups (near-duplicate)", str(result.type3_groups)
|
|
263
|
+
)
|
|
264
|
+
_console.print(table)
|
|
265
|
+
_console.print("\n[bold]Reports saved:[/bold]")
|
|
266
|
+
for dest in written:
|
|
267
|
+
_console.print(f" [cyan]•[/cyan] {dest}")
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _process_files(
|
|
271
|
+
files: list[tuple[Path, str]],
|
|
272
|
+
session_db: SessionDB,
|
|
273
|
+
session_id: str,
|
|
274
|
+
min_tokens: int,
|
|
275
|
+
) -> int:
|
|
276
|
+
"""Parse, extract, and hash every file; bulk-insert fragments into *session_db*.
|
|
277
|
+
|
|
278
|
+
Returns:
|
|
279
|
+
Total number of fragments extracted.
|
|
280
|
+
"""
|
|
281
|
+
total = 0
|
|
282
|
+
with Progress(
|
|
283
|
+
SpinnerColumn(),
|
|
284
|
+
TextColumn("[progress.description]{task.description}"),
|
|
285
|
+
BarColumn(),
|
|
286
|
+
MofNCompleteColumn(),
|
|
287
|
+
console=_console,
|
|
288
|
+
transient=True,
|
|
289
|
+
) as progress:
|
|
290
|
+
task = progress.add_task("Processing files…", total=len(files))
|
|
291
|
+
for file_path, language in files:
|
|
292
|
+
try:
|
|
293
|
+
source_bytes = file_path.read_bytes()
|
|
294
|
+
tree = ts_parser.parse(source_bytes, language)
|
|
295
|
+
if tree is not None:
|
|
296
|
+
frags = extractor.extract_fragments(
|
|
297
|
+
tree, source_bytes, file_path, language, session_id, min_tokens
|
|
298
|
+
)
|
|
299
|
+
fingerprint.hash_all(frags)
|
|
300
|
+
session_db.insert_many_fragments(frags)
|
|
301
|
+
total += len(frags)
|
|
302
|
+
except Exception as exc: # pylint: disable=broad-exception-caught
|
|
303
|
+
_console.print(f"[red]Error processing {file_path.name}: {exc}[/red]")
|
|
304
|
+
finally:
|
|
305
|
+
progress.advance(task)
|
|
306
|
+
return total
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
if __name__ == "__main__":
|
|
310
|
+
main() # pylint: disable=no-value-for-parameter
|