codeecho 1.0.1__py3-none-any.whl → 1.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codeecho/__init__.py +6 -2
- codeecho/__main__.py +250 -35
- codeecho/basis.py +101 -0
- codeecho/config.ini +2 -0
- codeecho/config.py +41 -0
- codeecho/db.py +42 -11
- codeecho/detector.py +44 -10
- codeecho/extractor.py +10 -3
- codeecho/fingerprint.py +9 -2
- codeecho/logging.ini +7 -1
- codeecho/models.py +7 -2
- codeecho/normalizer.py +30 -6
- codeecho/parser.py +18 -8
- codeecho/reporter/html_reporter.py +43 -11
- codeecho/reporter/json_reporter.py +46 -10
- codeecho/scanner.py +36 -12
- {codeecho-1.0.1.dist-info → codeecho-1.2.0.dist-info}/METADATA +41 -16
- codeecho-1.2.0.dist-info/RECORD +23 -0
- codeecho-1.0.1.dist-info/RECORD +0 -20
- {codeecho-1.0.1.dist-info → codeecho-1.2.0.dist-info}/WHEEL +0 -0
- {codeecho-1.0.1.dist-info → codeecho-1.2.0.dist-info}/entry_points.txt +0 -0
- {codeecho-1.0.1.dist-info → codeecho-1.2.0.dist-info}/licenses/LICENSE +0 -0
codeecho/__init__.py
CHANGED
|
@@ -2,21 +2,25 @@
|
|
|
2
2
|
codeecho - A developer tool that scans your codebase to detect and highlight ECHOES
|
|
3
3
|
of duplicated or near-duplicated code so you can refactor toward cleaner, more
|
|
4
4
|
maintainable designs.
|
|
5
|
+
|
|
6
|
+
:author: Ron Webb
|
|
7
|
+
:since: 1.0.0
|
|
5
8
|
"""
|
|
6
9
|
|
|
7
10
|
from env_dir_bootstrap import EnvDirBootstrap
|
|
8
11
|
from logenrich import setup_logger
|
|
9
12
|
|
|
10
|
-
__version__ = "1.0
|
|
13
|
+
__version__ = "1.2.0"
|
|
11
14
|
|
|
12
15
|
_bootstrapper = EnvDirBootstrap(
|
|
13
16
|
env_var="CODEECHO_CONFIG_DIR",
|
|
14
|
-
resources=["logging.ini", ".ignore"],
|
|
17
|
+
resources=["logging.ini", ".ignore", "config.ini"],
|
|
15
18
|
package="codeecho",
|
|
16
19
|
)
|
|
17
20
|
|
|
18
21
|
_bootstrapper.setup()
|
|
19
22
|
|
|
20
23
|
CONF_DIR = str(_bootstrapper.get_dir())
|
|
24
|
+
DEFAULT_IGNORE_PATH = _bootstrapper.resolve(".ignore")
|
|
21
25
|
|
|
22
26
|
setup_logger("codeecho", conf_dir=CONF_DIR)
|
codeecho/__main__.py
CHANGED
|
@@ -4,8 +4,14 @@ CLI entry point for codeecho.
|
|
|
4
4
|
Invoked via::
|
|
5
5
|
|
|
6
6
|
poetry run python -m codeecho [OPTIONS] PATH
|
|
7
|
+
|
|
8
|
+
:author: Ron Webb
|
|
9
|
+
:since: 1.0.0
|
|
7
10
|
"""
|
|
8
11
|
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import os
|
|
9
15
|
import uuid
|
|
10
16
|
from pathlib import Path
|
|
11
17
|
|
|
@@ -22,20 +28,27 @@ from rich.progress import (
|
|
|
22
28
|
)
|
|
23
29
|
from rich.table import Table
|
|
24
30
|
|
|
25
|
-
from
|
|
26
|
-
from
|
|
27
|
-
from
|
|
28
|
-
from
|
|
29
|
-
from
|
|
30
|
-
from
|
|
31
|
+
from . import __version__, CONF_DIR, DEFAULT_IGNORE_PATH
|
|
32
|
+
from . import basis as basis_module
|
|
33
|
+
from . import extractor, fingerprint, parser as ts_parser, scanner
|
|
34
|
+
from .config import Config
|
|
35
|
+
from .db import SessionDB, get_db_path
|
|
36
|
+
from .detector import detect
|
|
37
|
+
from .models import ScanResult
|
|
38
|
+
from .reporter import html_reporter, json_reporter
|
|
31
39
|
|
|
32
40
|
_console = Console()
|
|
41
|
+
_config = Config()
|
|
42
|
+
_logger = logging.getLogger("codeecho.__main__")
|
|
33
43
|
|
|
34
44
|
_FORMAT_CHOICES = click.Choice(["json", "html", "both"])
|
|
35
45
|
|
|
36
46
|
|
|
37
47
|
def _parse_types(types_str: str) -> set[int]:
|
|
38
|
-
"""Parse ``--types`` value into a set of integers.
|
|
48
|
+
"""Parse ``--types`` value into a set of integers.
|
|
49
|
+
|
|
50
|
+
:since: 1.0.0
|
|
51
|
+
"""
|
|
39
52
|
if types_str.strip().lower() == "all":
|
|
40
53
|
return {1, 2, 3}
|
|
41
54
|
result: set[int] = set()
|
|
@@ -50,6 +63,120 @@ def _parse_types(types_str: str) -> set[int]:
|
|
|
50
63
|
return result
|
|
51
64
|
|
|
52
65
|
|
|
66
|
+
def _read_target_list(list_file: Path) -> list[Path]:
|
|
67
|
+
"""Read one target path per line from *list_file*.
|
|
68
|
+
|
|
69
|
+
Blank lines and lines starting with ``#`` are skipped.
|
|
70
|
+
|
|
71
|
+
:since: 1.2.0
|
|
72
|
+
"""
|
|
73
|
+
result: list[Path] = []
|
|
74
|
+
for line in list_file.read_text(encoding="utf-8").splitlines():
|
|
75
|
+
stripped = line.strip()
|
|
76
|
+
if stripped and not stripped.startswith("#"):
|
|
77
|
+
result.append(Path(stripped))
|
|
78
|
+
return result
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _load_ignore_file(ignore_path: Path, base_dir: Path) -> IgnoreFile | None:
|
|
82
|
+
"""Load an :class:`IgnoreFile` from *ignore_path*, logging failures.
|
|
83
|
+
|
|
84
|
+
Returns ``None`` on a missing or non-UTF-8 file instead of raising.
|
|
85
|
+
|
|
86
|
+
:since: 1.2.0
|
|
87
|
+
"""
|
|
88
|
+
try:
|
|
89
|
+
return IgnoreFile(ignore_path, base_dir=base_dir)
|
|
90
|
+
except FileNotFoundError:
|
|
91
|
+
_logger.warning("Ignore file not found at %s", ignore_path)
|
|
92
|
+
return None
|
|
93
|
+
except UnicodeDecodeError as exc:
|
|
94
|
+
_logger.warning("Ignore file at %s is not valid UTF-8: %s", ignore_path, exc)
|
|
95
|
+
return None
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _build_ignore(base_dir: Path) -> IgnoreFile | None:
|
|
99
|
+
"""Build a path-ignore matcher anchored at *base_dir*.
|
|
100
|
+
|
|
101
|
+
The ignore filename is resolved from ``config.ini``'s ``[override]
|
|
102
|
+
ignore-file`` setting under :data:`CONF_DIR`; when that file is missing,
|
|
103
|
+
the bundled default ``.ignore`` is tried instead.
|
|
104
|
+
|
|
105
|
+
:since: 1.2.0
|
|
106
|
+
"""
|
|
107
|
+
custom_path = Path(CONF_DIR) / _config.get_ignore_file()
|
|
108
|
+
ignore = _load_ignore_file(custom_path, base_dir)
|
|
109
|
+
if ignore is not None:
|
|
110
|
+
return ignore
|
|
111
|
+
if custom_path == Path(DEFAULT_IGNORE_PATH):
|
|
112
|
+
return None
|
|
113
|
+
return _load_ignore_file(Path(DEFAULT_IGNORE_PATH), base_dir)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _resolve_target_paths(
|
|
117
|
+
paths: tuple[Path, ...], target_list: bool
|
|
118
|
+
) -> tuple[Path, ...]:
|
|
119
|
+
"""Validate *paths* and expand ``--target-list`` into concrete target paths.
|
|
120
|
+
|
|
121
|
+
:since: 1.2.0
|
|
122
|
+
"""
|
|
123
|
+
if not paths:
|
|
124
|
+
raise click.UsageError("At least one PATH is required.")
|
|
125
|
+
|
|
126
|
+
if not target_list:
|
|
127
|
+
return paths
|
|
128
|
+
|
|
129
|
+
if len(paths) != 1 or not paths[0].is_file():
|
|
130
|
+
raise click.UsageError(
|
|
131
|
+
"--target-list requires PATH to be a single existing file."
|
|
132
|
+
)
|
|
133
|
+
expanded = tuple(_read_target_list(paths[0]))
|
|
134
|
+
if not expanded:
|
|
135
|
+
raise click.UsageError("--target-list file contains no target paths.")
|
|
136
|
+
return expanded
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _read_basis_targets(basis_path: Path | None) -> list[Path]:
|
|
140
|
+
"""Read the basis path(s) listed in *basis_path*, if given.
|
|
141
|
+
|
|
142
|
+
Absolute entries are resolved (normalising case/symlinks); relative entries
|
|
143
|
+
(e.g. a bare filename) are left as-is so they can be matched by filename/suffix
|
|
144
|
+
against the scanned files instead of being resolved against the current
|
|
145
|
+
working directory.
|
|
146
|
+
|
|
147
|
+
:since: 1.2.0
|
|
148
|
+
"""
|
|
149
|
+
if basis_path is None:
|
|
150
|
+
return []
|
|
151
|
+
targets = [
|
|
152
|
+
p.resolve() if p.is_absolute() else p for p in _read_target_list(basis_path)
|
|
153
|
+
]
|
|
154
|
+
if not targets:
|
|
155
|
+
raise click.UsageError("--basis file contains no target paths.")
|
|
156
|
+
return targets
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _discover_files(
|
|
160
|
+
resolved: list[Path], exclude: tuple[str, ...]
|
|
161
|
+
) -> list[tuple[Path, str]]:
|
|
162
|
+
"""Print scan targets and return the discovered ``(file, language)`` pairs.
|
|
163
|
+
|
|
164
|
+
:since: 1.2.0
|
|
165
|
+
"""
|
|
166
|
+
for target_path in resolved:
|
|
167
|
+
_console.print(f"[dim]Scanning:[/dim] [bold]{target_path}[/bold]")
|
|
168
|
+
try:
|
|
169
|
+
base_dir = Path(os.path.commonpath(resolved))
|
|
170
|
+
if not base_dir.is_dir():
|
|
171
|
+
base_dir = base_dir.parent
|
|
172
|
+
except ValueError:
|
|
173
|
+
base_dir = Path.cwd()
|
|
174
|
+
ignore_file = _build_ignore(base_dir)
|
|
175
|
+
return scanner.scan(
|
|
176
|
+
tuple(resolved), exclude_patterns=exclude, ignore_file=ignore_file
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
|
|
53
180
|
@click.command(
|
|
54
181
|
name="codeecho", context_settings={"help_option_names": ["-h", "--help"]}
|
|
55
182
|
)
|
|
@@ -57,7 +184,9 @@ def _parse_types(types_str: str) -> set[int]:
|
|
|
57
184
|
version=__version__, prog_name="codeecho", message="%(prog)s v%(version)s"
|
|
58
185
|
)
|
|
59
186
|
@click.argument(
|
|
60
|
-
"
|
|
187
|
+
"paths",
|
|
188
|
+
nargs=-1,
|
|
189
|
+
type=click.Path(exists=True, file_okay=True, dir_okay=True, path_type=Path),
|
|
61
190
|
)
|
|
62
191
|
@click.option(
|
|
63
192
|
"--types",
|
|
@@ -116,8 +245,35 @@ def _parse_types(types_str: str) -> set[int]:
|
|
|
116
245
|
metavar="PATTERN",
|
|
117
246
|
help="Glob pattern(s) to exclude from scanning (repeatable).",
|
|
118
247
|
)
|
|
248
|
+
@click.option(
|
|
249
|
+
"--target-list",
|
|
250
|
+
"target_list",
|
|
251
|
+
is_flag=True,
|
|
252
|
+
default=False,
|
|
253
|
+
help=(
|
|
254
|
+
"Treat PATH as a single existing file listing target paths (files "
|
|
255
|
+
"and/or directories), one per line, instead of individual PATH "
|
|
256
|
+
"arguments. Blank lines and lines starting with '#' are skipped."
|
|
257
|
+
),
|
|
258
|
+
)
|
|
259
|
+
@click.option(
|
|
260
|
+
"--basis",
|
|
261
|
+
"basis_path",
|
|
262
|
+
default=None,
|
|
263
|
+
type=click.Path(exists=True, dir_okay=False, path_type=Path),
|
|
264
|
+
metavar="FILE",
|
|
265
|
+
help=(
|
|
266
|
+
"File listing basis target paths, one per line, same format as "
|
|
267
|
+
"--target-list. Absolute file/directory entries are added to the scan "
|
|
268
|
+
"automatically and matched exactly; relative entries (e.g. a bare "
|
|
269
|
+
"filename) are matched by filename/suffix against any file discovered "
|
|
270
|
+
"in the scan. The report is filtered to only clone groups that touch "
|
|
271
|
+
"at least one basis file; groups duplicated purely among basis files "
|
|
272
|
+
"are flagged as such."
|
|
273
|
+
),
|
|
274
|
+
)
|
|
119
275
|
def main( # pylint: disable=too-many-arguments,too-many-positional-arguments,too-many-locals,invalid-name
|
|
120
|
-
|
|
276
|
+
paths: tuple[Path, ...],
|
|
121
277
|
types: str,
|
|
122
278
|
threshold: float,
|
|
123
279
|
output: str,
|
|
@@ -126,11 +282,15 @@ def main( # pylint: disable=too-many-arguments,too-many-positional-arguments,to
|
|
|
126
282
|
fmt: str,
|
|
127
283
|
min_tokens: int,
|
|
128
284
|
exclude: tuple[str, ...],
|
|
285
|
+
target_list: bool,
|
|
286
|
+
basis_path: Path | None,
|
|
129
287
|
) -> None:
|
|
130
|
-
"""Scan PATH for duplicate and near-duplicate code.
|
|
288
|
+
"""Scan one or more PATH(s) for duplicate and near-duplicate code.
|
|
131
289
|
|
|
132
|
-
Generates JSON and/or HTML reports,
|
|
133
|
-
from the embedded database.
|
|
290
|
+
Each PATH may be a file or a directory. Generates JSON and/or HTML reports,
|
|
291
|
+
then removes the intermediate session data from the embedded database.
|
|
292
|
+
|
|
293
|
+
:since: 1.0.0
|
|
134
294
|
"""
|
|
135
295
|
_console.print(
|
|
136
296
|
Panel(
|
|
@@ -139,56 +299,83 @@ def main( # pylint: disable=too-many-arguments,too-many-positional-arguments,to
|
|
|
139
299
|
)
|
|
140
300
|
)
|
|
141
301
|
|
|
302
|
+
paths = _resolve_target_paths(paths, target_list)
|
|
142
303
|
detect_types = _parse_types(types)
|
|
143
304
|
if output_dir is None:
|
|
144
305
|
output_dir = Path.cwd() / "reports"
|
|
306
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
307
|
+
|
|
308
|
+
basis_targets = _read_basis_targets(basis_path)
|
|
309
|
+
|
|
145
310
|
session_id = str(uuid.uuid4())
|
|
146
311
|
config = {
|
|
147
312
|
"types": types,
|
|
148
313
|
"threshold": threshold,
|
|
149
314
|
"min_tokens": min_tokens,
|
|
150
315
|
"exclude": list(exclude),
|
|
316
|
+
"basis": [str(p) for p in basis_targets],
|
|
151
317
|
}
|
|
152
|
-
|
|
153
|
-
|
|
318
|
+
resolved = [p.resolve() for p in paths]
|
|
319
|
+
for target in basis_targets:
|
|
320
|
+
if target.is_absolute() and target not in resolved:
|
|
321
|
+
resolved.append(target)
|
|
154
322
|
|
|
155
323
|
db_path = Path(get_db_path(str(db_dir) if db_dir else None))
|
|
156
324
|
with SessionDB(db_path=db_path) as session_db:
|
|
157
|
-
session_db.create_session(
|
|
325
|
+
session_db.create_session(
|
|
326
|
+
session_id, json.dumps([str(p) for p in resolved]), config
|
|
327
|
+
)
|
|
158
328
|
|
|
159
|
-
|
|
160
|
-
_console.print(f"[dim]Scanning:[/dim] [bold]{path.resolve()}[/bold]")
|
|
161
|
-
_ignore_file = IgnoreFile(Path(CONF_DIR) / ".ignore", base_dir=path.resolve())
|
|
162
|
-
files = scanner.scan(path, exclude_patterns=exclude, ignore_file=_ignore_file)
|
|
163
|
-
total_fragments = 0
|
|
329
|
+
files = _discover_files(resolved, exclude)
|
|
164
330
|
|
|
165
331
|
if not files:
|
|
166
332
|
_console.print("[yellow]No supported source files found.[/yellow]")
|
|
167
333
|
session_db.delete_session(session_id)
|
|
168
334
|
return
|
|
169
335
|
|
|
170
|
-
|
|
336
|
+
basis_files = None
|
|
337
|
+
if basis_targets:
|
|
338
|
+
basis_files = basis_module.resolve_basis_files(files, basis_targets)
|
|
339
|
+
if not basis_files:
|
|
340
|
+
_console.print(
|
|
341
|
+
"[yellow]Warning: none of the --basis paths matched any scanned files.[/yellow]"
|
|
342
|
+
)
|
|
343
|
+
|
|
171
344
|
total_fragments = _process_files(files, session_db, session_id, min_tokens)
|
|
172
345
|
|
|
173
|
-
# ── Phase 3: Clone detection ────────────────────────────────────────
|
|
174
346
|
cnt1, cnt2, cnt3 = _detect_clones(
|
|
175
347
|
session_db, session_id, detect_types, threshold
|
|
176
348
|
)
|
|
177
349
|
|
|
350
|
+
basis_counts = {1: 0, 2: 0, 3: 0}
|
|
351
|
+
if basis_files is not None:
|
|
352
|
+
groups = session_db.get_clone_groups(session_id)
|
|
353
|
+
groups_with_members = [
|
|
354
|
+
(g, session_db.get_fragments_for_group(g)) for g in groups
|
|
355
|
+
]
|
|
356
|
+
basis_counts = basis_module.count_basis_groups_by_type(
|
|
357
|
+
groups_with_members, basis_files
|
|
358
|
+
)
|
|
359
|
+
|
|
178
360
|
result = ScanResult(
|
|
179
361
|
session_id=session_id,
|
|
180
|
-
|
|
362
|
+
version=__version__,
|
|
363
|
+
scan_path=[str(p) for p in resolved],
|
|
181
364
|
files_scanned=len(files),
|
|
182
365
|
fragments_extracted=total_fragments,
|
|
183
366
|
type1_groups=cnt1,
|
|
184
367
|
type2_groups=cnt2,
|
|
185
368
|
type3_groups=cnt3,
|
|
369
|
+
basis_paths=[str(p) for p in basis_targets],
|
|
370
|
+
basis_type1_groups=basis_counts[1],
|
|
371
|
+
basis_type2_groups=basis_counts[2],
|
|
372
|
+
basis_type3_groups=basis_counts[3],
|
|
186
373
|
)
|
|
187
374
|
|
|
188
|
-
|
|
189
|
-
|
|
375
|
+
written = _write_reports(
|
|
376
|
+
session_db, result, output_dir, output, fmt, basis_files
|
|
377
|
+
)
|
|
190
378
|
|
|
191
|
-
# ── Phase 5: Clean up session ───────────────────────────────────────
|
|
192
379
|
session_db.delete_session(session_id)
|
|
193
380
|
|
|
194
381
|
# ── Summary table (printed after DB is closed) ──────────────────────────
|
|
@@ -201,46 +388,72 @@ def _detect_clones(
|
|
|
201
388
|
detect_types: set[int],
|
|
202
389
|
threshold: float,
|
|
203
390
|
) -> tuple[int, int, int]:
|
|
204
|
-
"""Run clone detection and return (type1_count, type2_count, type3_count).
|
|
391
|
+
"""Run clone detection and return (type1_count, type2_count, type3_count).
|
|
392
|
+
|
|
393
|
+
:since: 1.0.0
|
|
394
|
+
"""
|
|
205
395
|
_console.print("[dim]Detecting clones…[/dim]")
|
|
206
396
|
return detect(session_db, session_id, detect_types, threshold)
|
|
207
397
|
|
|
208
398
|
|
|
209
|
-
def _write_reports(
|
|
399
|
+
def _write_reports( # pylint: disable=too-many-arguments,too-many-positional-arguments
|
|
210
400
|
session_db: SessionDB,
|
|
211
401
|
result: ScanResult,
|
|
212
402
|
output_dir: Path,
|
|
213
403
|
output: str,
|
|
214
404
|
fmt: str,
|
|
405
|
+
basis_files: frozenset[str] | None = None,
|
|
215
406
|
) -> list[Path]:
|
|
216
|
-
"""Write the requested report formats and return a list of written paths.
|
|
407
|
+
"""Write the requested report formats and return a list of written paths.
|
|
408
|
+
|
|
409
|
+
:since: 1.0.0
|
|
410
|
+
"""
|
|
217
411
|
written: list[Path] = []
|
|
218
412
|
if fmt in ("json", "both"):
|
|
219
413
|
written.append(
|
|
220
|
-
json_reporter.write(
|
|
414
|
+
json_reporter.write(
|
|
415
|
+
session_db, result, output_dir / f"{output}.json", basis_files
|
|
416
|
+
)
|
|
221
417
|
)
|
|
222
418
|
if fmt in ("html", "both"):
|
|
223
419
|
written.append(
|
|
224
|
-
html_reporter.write(
|
|
420
|
+
html_reporter.write(
|
|
421
|
+
session_db, result, output_dir / f"{output}.html", basis_files
|
|
422
|
+
)
|
|
225
423
|
)
|
|
226
424
|
return written
|
|
227
425
|
|
|
228
426
|
|
|
229
427
|
def _print_summary(result: ScanResult, written: list[Path]) -> None:
|
|
230
|
-
"""Print the scan summary table and list of saved report paths.
|
|
428
|
+
"""Print the scan summary table and list of saved report paths.
|
|
429
|
+
|
|
430
|
+
:since: 1.0.0
|
|
431
|
+
"""
|
|
432
|
+
has_basis = bool(result.basis_paths)
|
|
433
|
+
|
|
434
|
+
def _value(count: int, basis_count: int) -> str:
|
|
435
|
+
return f"{count} ({basis_count})" if has_basis else str(count)
|
|
436
|
+
|
|
231
437
|
table = Table(title="Scan Summary", show_header=True, header_style="bold magenta")
|
|
232
438
|
table.add_column("Metric", style="dim", min_width=26)
|
|
233
439
|
table.add_column("Value", justify="right", style="bold")
|
|
234
440
|
table.add_row("Files scanned", str(result.files_scanned))
|
|
235
441
|
table.add_row("Fragments extracted", str(result.fragments_extracted))
|
|
236
|
-
table.add_row("[red]Type-1[/red] clone groups (exact)", str(result.type1_groups))
|
|
237
442
|
table.add_row(
|
|
238
|
-
"[
|
|
443
|
+
"[red]Type-1[/red] clone groups (exact)",
|
|
444
|
+
_value(result.type1_groups, result.basis_type1_groups),
|
|
239
445
|
)
|
|
240
446
|
table.add_row(
|
|
241
|
-
"[
|
|
447
|
+
"[yellow]Type-2[/yellow] clone groups (structural)",
|
|
448
|
+
_value(result.type2_groups, result.basis_type2_groups),
|
|
449
|
+
)
|
|
450
|
+
table.add_row(
|
|
451
|
+
"[green]Type-3[/green] clone groups (near-duplicate)",
|
|
452
|
+
_value(result.type3_groups, result.basis_type3_groups),
|
|
242
453
|
)
|
|
243
454
|
_console.print(table)
|
|
455
|
+
if has_basis:
|
|
456
|
+
_console.print("[dim]Value in parentheses = basis-touching groups.[/dim]")
|
|
244
457
|
_console.print("\n[bold]Reports saved:[/bold]")
|
|
245
458
|
for dest in written:
|
|
246
459
|
_console.print(f" [cyan]•[/cyan] {dest}")
|
|
@@ -256,6 +469,8 @@ def _process_files(
|
|
|
256
469
|
|
|
257
470
|
Returns:
|
|
258
471
|
Total number of fragments extracted.
|
|
472
|
+
|
|
473
|
+
:since: 1.0.0
|
|
259
474
|
"""
|
|
260
475
|
total = 0
|
|
261
476
|
with Progress(
|
codeecho/basis.py
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Basis-file resolution and clone-group filtering for ``--basis`` scans.
|
|
3
|
+
|
|
4
|
+
A "basis" is a user-supplied subset of the scanned files that acts as the reference
|
|
5
|
+
source for duplicate detection: clone groups are filtered down to only those that
|
|
6
|
+
touch at least one basis file, and groups whose members are *all* basis files are
|
|
7
|
+
flagged as ``basis_internal`` (duplication found purely among the basis files).
|
|
8
|
+
|
|
9
|
+
:author: Ron Webb
|
|
10
|
+
:since: 1.2.0
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from .models import CloneGroup, Fragment
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _matches_relative_suffix(file_path: Path, relative_target: Path) -> bool:
|
|
19
|
+
"""Return True when *file_path*'s trailing path segments equal *relative_target*.
|
|
20
|
+
|
|
21
|
+
Comparison is case-insensitive so a bare filename like ``IInfuser.gs`` matches
|
|
22
|
+
the file regardless of the casing baked into either path.
|
|
23
|
+
|
|
24
|
+
:since: 1.2.0
|
|
25
|
+
"""
|
|
26
|
+
rel_parts = relative_target.parts
|
|
27
|
+
file_parts = file_path.parts
|
|
28
|
+
if len(rel_parts) > len(file_parts):
|
|
29
|
+
return False
|
|
30
|
+
tail = file_parts[len(file_parts) - len(rel_parts) :]
|
|
31
|
+
return tuple(p.casefold() for p in tail) == tuple(p.casefold() for p in rel_parts)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def resolve_basis_files(
|
|
35
|
+
discovered: list[tuple[Path, str]], basis_targets: list[Path]
|
|
36
|
+
) -> frozenset[str]:
|
|
37
|
+
"""Return the resolved file paths from *discovered* that fall under *basis_targets*.
|
|
38
|
+
|
|
39
|
+
An absolute basis target matches a discovered file that equals it exactly, or that
|
|
40
|
+
lies inside it (when it's a directory). A relative basis target (e.g. a bare
|
|
41
|
+
filename like ``IInfuser.gs``, or a partial path like ``sub/File.gs``) instead
|
|
42
|
+
matches any discovered file whose trailing path segments equal it, wherever that
|
|
43
|
+
file lives within the scanned tree.
|
|
44
|
+
|
|
45
|
+
:param discovered: ``(path, language)`` pairs as returned by :func:`codeecho.scanner.scan`.
|
|
46
|
+
:param basis_targets: Basis file/directory paths read from the ``--basis`` file;
|
|
47
|
+
absolute entries are resolved, relative entries are kept as-is.
|
|
48
|
+
:returns: Frozen set of matching file paths (as strings, matching :attr:`Fragment.file_path`).
|
|
49
|
+
:since: 1.2.0
|
|
50
|
+
"""
|
|
51
|
+
absolute_targets = [p for p in basis_targets if p.is_absolute()]
|
|
52
|
+
relative_targets = [p for p in basis_targets if not p.is_absolute()]
|
|
53
|
+
basis_dirs = [p for p in absolute_targets if p.is_dir()]
|
|
54
|
+
basis_files = {p for p in absolute_targets if p.is_file()}
|
|
55
|
+
|
|
56
|
+
matched: set[str] = set()
|
|
57
|
+
for file_path, _ in discovered:
|
|
58
|
+
if file_path in basis_files or any(
|
|
59
|
+
file_path.is_relative_to(d) for d in basis_dirs
|
|
60
|
+
):
|
|
61
|
+
matched.add(str(file_path))
|
|
62
|
+
elif any(_matches_relative_suffix(file_path, rel) for rel in relative_targets):
|
|
63
|
+
matched.add(str(file_path))
|
|
64
|
+
return frozenset(matched)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def filter_groups_for_basis(
|
|
68
|
+
groups_with_members: list[tuple[CloneGroup, list[Fragment]]],
|
|
69
|
+
basis_files: frozenset[str],
|
|
70
|
+
) -> list[tuple[CloneGroup, list[Fragment], bool]]:
|
|
71
|
+
"""Keep only groups touching a basis file, flagging groups that are basis-only.
|
|
72
|
+
|
|
73
|
+
:param groups_with_members: ``(group, members)`` pairs for every detected clone group.
|
|
74
|
+
:param basis_files: Resolved basis file paths, as returned by :func:`resolve_basis_files`.
|
|
75
|
+
:returns: ``(group, members, basis_internal)`` triples for groups with >=1 basis member.
|
|
76
|
+
:since: 1.2.0
|
|
77
|
+
"""
|
|
78
|
+
result: list[tuple[CloneGroup, list[Fragment], bool]] = []
|
|
79
|
+
for group, members in groups_with_members:
|
|
80
|
+
member_is_basis = [m.file_path in basis_files for m in members]
|
|
81
|
+
if not any(member_is_basis):
|
|
82
|
+
continue
|
|
83
|
+
result.append((group, members, all(member_is_basis)))
|
|
84
|
+
return result
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def count_basis_groups_by_type(
|
|
88
|
+
groups_with_members: list[tuple[CloneGroup, list[Fragment]]],
|
|
89
|
+
basis_files: frozenset[str],
|
|
90
|
+
) -> dict[int, int]:
|
|
91
|
+
"""Count basis-touching clone groups per clone type (1, 2, 3).
|
|
92
|
+
|
|
93
|
+
:param groups_with_members: ``(group, members)`` pairs for every detected clone group.
|
|
94
|
+
:param basis_files: Resolved basis file paths, as returned by :func:`resolve_basis_files`.
|
|
95
|
+
:returns: Mapping of clone type to the number of basis-touching groups of that type.
|
|
96
|
+
:since: 1.2.0
|
|
97
|
+
"""
|
|
98
|
+
counts: dict[int, int] = {1: 0, 2: 0, 3: 0}
|
|
99
|
+
for group, _, _ in filter_groups_for_basis(groups_with_members, basis_files):
|
|
100
|
+
counts[group.clone_type] = counts.get(group.clone_type, 0) + 1
|
|
101
|
+
return counts
|
codeecho/config.ini
ADDED
codeecho/config.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""
|
|
2
|
+
User-configurable overrides for codeecho, read from ``config.ini``.
|
|
3
|
+
|
|
4
|
+
:author: Ron Webb
|
|
5
|
+
:since: 1.2.0
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import configparser
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from . import CONF_DIR
|
|
12
|
+
|
|
13
|
+
_DEFAULT_IGNORE_FILE = ".ignore"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class Config:
|
|
17
|
+
"""Reads user-configurable overrides from ``config.ini`` in *conf_dir*.
|
|
18
|
+
|
|
19
|
+
:author: Ron Webb
|
|
20
|
+
:since: 1.2.0
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
def __init__(self, conf_dir: str | None = None) -> None:
|
|
24
|
+
"""Load ``config.ini`` from *conf_dir* (defaults to ``CONF_DIR``).
|
|
25
|
+
|
|
26
|
+
:since: 1.2.0
|
|
27
|
+
"""
|
|
28
|
+
self._config = configparser.ConfigParser()
|
|
29
|
+
config_path = Path(conf_dir or CONF_DIR) / "config.ini"
|
|
30
|
+
self._config.read(config_path)
|
|
31
|
+
|
|
32
|
+
def get_ignore_file(self) -> str:
|
|
33
|
+
"""Return the configured ignore filename override.
|
|
34
|
+
|
|
35
|
+
Falls back to ``.ignore`` when the section/key is absent.
|
|
36
|
+
|
|
37
|
+
:since: 1.2.0
|
|
38
|
+
"""
|
|
39
|
+
return self._config.get(
|
|
40
|
+
"override", "ignore-file", fallback=_DEFAULT_IGNORE_FILE
|
|
41
|
+
)
|