codeecho 1.0.0__py3-none-any.whl → 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
codeecho/__init__.py CHANGED
@@ -1,22 +1,22 @@
1
- """
2
- codeecho - A developer tool that scans your codebase to detect and highlight ECHOES
3
- of duplicated or near-duplicated code so you can refactor toward cleaner, more
4
- maintainable designs.
5
- """
6
-
7
- from env_dir_bootstrap import EnvDirBootstrap
8
- from logenrich import setup_logger
9
-
10
- __version__ = "1.0.0"
11
-
12
- _bootstrapper = EnvDirBootstrap(
13
- env_var="CODEECHO_CONFIG_DIR",
14
- resources=["logging.ini", ".ignore"],
15
- package="codeecho",
16
- )
17
-
18
- _bootstrapper.setup()
19
-
20
- CONF_DIR = str(_bootstrapper.get_dir())
21
-
22
- setup_logger("codeecho", conf_dir=CONF_DIR)
1
+ """
2
+ codeecho - A developer tool that scans your codebase to detect and highlight ECHOES
3
+ of duplicated or near-duplicated code so you can refactor toward cleaner, more
4
+ maintainable designs.
5
+ """
6
+
7
+ from env_dir_bootstrap import EnvDirBootstrap
8
+ from logenrich import setup_logger
9
+
10
+ __version__ = "1.0.1"
11
+
12
+ _bootstrapper = EnvDirBootstrap(
13
+ env_var="CODEECHO_CONFIG_DIR",
14
+ resources=["logging.ini", ".ignore"],
15
+ package="codeecho",
16
+ )
17
+
18
+ _bootstrapper.setup()
19
+
20
+ CONF_DIR = str(_bootstrapper.get_dir())
21
+
22
+ setup_logger("codeecho", conf_dir=CONF_DIR)
codeecho/__main__.py CHANGED
@@ -1,289 +1,289 @@
1
- """
2
- CLI entry point for codeecho.
3
-
4
- Invoked via::
5
-
6
- poetry run python -m codeecho [OPTIONS] PATH
7
- """
8
-
9
- import uuid
10
- from pathlib import Path
11
-
12
- import click
13
- from braincraft import IgnoreFile
14
- from rich.console import Console
15
- from rich.panel import Panel
16
- from rich.progress import (
17
- BarColumn,
18
- MofNCompleteColumn,
19
- Progress,
20
- SpinnerColumn,
21
- TextColumn,
22
- )
23
- from rich.table import Table
24
-
25
- from codeecho import __version__, CONF_DIR
26
- from codeecho import extractor, fingerprint, parser as ts_parser, scanner
27
- from codeecho.db import SessionDB, get_db_path
28
- from codeecho.detector import detect
29
- from codeecho.models import ScanResult
30
- from codeecho.reporter import html_reporter, json_reporter
31
-
32
- _console = Console()
33
-
34
- _FORMAT_CHOICES = click.Choice(["json", "html", "both"])
35
-
36
-
37
- def _parse_types(types_str: str) -> set[int]:
38
- """Parse ``--types`` value into a set of integers."""
39
- if types_str.strip().lower() == "all":
40
- return {1, 2, 3}
41
- result: set[int] = set()
42
- for token in types_str.split(","):
43
- token = token.strip()
44
- if token.isdigit() and token in {"1", "2", "3"}:
45
- result.add(int(token))
46
- if not result:
47
- raise click.BadParameter(
48
- f"Invalid types value: {types_str!r}. Use 'all' or e.g. '1,2,3'."
49
- )
50
- return result
51
-
52
-
53
- @click.command(
54
- name="codeecho", context_settings={"help_option_names": ["-h", "--help"]}
55
- )
56
- @click.version_option(
57
- version=__version__, prog_name="codeecho", message="%(prog)s v%(version)s"
58
- )
59
- @click.argument(
60
- "path", type=click.Path(exists=True, file_okay=False, dir_okay=True, path_type=Path)
61
- )
62
- @click.option(
63
- "--types",
64
- default="all",
65
- show_default=True,
66
- metavar="TYPES",
67
- help="Clone types to detect: comma-separated (e.g. '1,2') or 'all'.",
68
- )
69
- @click.option(
70
- "--threshold",
71
- default=0.8,
72
- show_default=True,
73
- type=click.FloatRange(0.0, 1.0),
74
- help="Jaccard similarity threshold for Type-3 (near-duplicate) detection.",
75
- )
76
- @click.option(
77
- "--output",
78
- default="codeecho-output",
79
- show_default=True,
80
- metavar="NAME",
81
- help="Base name (without extension) for output file(s).",
82
- )
83
- @click.option(
84
- "--output-dir",
85
- default=None,
86
- show_default=False,
87
- type=click.Path(file_okay=False, path_type=Path),
88
- help="Directory where output file(s) will be written. [default: <cwd>/reports]",
89
- )
90
- @click.option(
91
- "--db-dir",
92
- "db_dir",
93
- default=None,
94
- show_default=False,
95
- type=click.Path(file_okay=False, path_type=Path),
96
- help="Directory for the SQLite scratch database. [default: ~/.codeecho]",
97
- )
98
- @click.option(
99
- "--format",
100
- "fmt",
101
- default="both",
102
- show_default=True,
103
- type=_FORMAT_CHOICES,
104
- help="Output format.",
105
- )
106
- @click.option(
107
- "--min-tokens",
108
- default=10,
109
- show_default=True,
110
- type=click.IntRange(1),
111
- help="Minimum token count for a fragment to be considered.",
112
- )
113
- @click.option(
114
- "--exclude",
115
- multiple=True,
116
- metavar="PATTERN",
117
- help="Glob pattern(s) to exclude from scanning (repeatable).",
118
- )
119
- def main( # pylint: disable=too-many-arguments,too-many-positional-arguments,too-many-locals,invalid-name
120
- path: Path,
121
- types: str,
122
- threshold: float,
123
- output: str,
124
- output_dir: Path | None,
125
- db_dir: Path | None,
126
- fmt: str,
127
- min_tokens: int,
128
- exclude: tuple[str, ...],
129
- ) -> None:
130
- """Scan PATH for duplicate and near-duplicate code.
131
-
132
- Generates JSON and/or HTML reports, then removes the intermediate session data
133
- from the embedded database.
134
- """
135
- _console.print(
136
- Panel(
137
- f"[bold cyan]codeecho[/bold cyan] [dim]v{__version__}[/dim] — Code Duplicate Scanner",
138
- border_style="dim",
139
- )
140
- )
141
-
142
- detect_types = _parse_types(types)
143
- if output_dir is None:
144
- output_dir = Path.cwd() / "reports"
145
- session_id = str(uuid.uuid4())
146
- config = {
147
- "types": types,
148
- "threshold": threshold,
149
- "min_tokens": min_tokens,
150
- "exclude": list(exclude),
151
- }
152
-
153
- output_dir.mkdir(parents=True, exist_ok=True)
154
-
155
- db_path = Path(get_db_path(str(db_dir) if db_dir else None))
156
- with SessionDB(db_path=db_path) as session_db:
157
- session_db.create_session(session_id, str(path.resolve()), config)
158
-
159
- # ── Phase 1: File discovery ─────────────────────────────────────────
160
- _console.print(f"[dim]Scanning:[/dim] [bold]{path.resolve()}[/bold]")
161
- _ignore_file = IgnoreFile(Path(CONF_DIR) / ".ignore", base_dir=path.resolve())
162
- files = scanner.scan(path, exclude_patterns=exclude, ignore_file=_ignore_file)
163
- total_fragments = 0
164
-
165
- if not files:
166
- _console.print("[yellow]No supported source files found.[/yellow]")
167
- session_db.delete_session(session_id)
168
- return
169
-
170
- # ── Phase 2: Parse → Extract → Hash ────────────────────────────────
171
- total_fragments = _process_files(files, session_db, session_id, min_tokens)
172
-
173
- # ── Phase 3: Clone detection ────────────────────────────────────────
174
- cnt1, cnt2, cnt3 = _detect_clones(
175
- session_db, session_id, detect_types, threshold
176
- )
177
-
178
- result = ScanResult(
179
- session_id=session_id,
180
- scan_path=str(path.resolve()),
181
- files_scanned=len(files),
182
- fragments_extracted=total_fragments,
183
- type1_groups=cnt1,
184
- type2_groups=cnt2,
185
- type3_groups=cnt3,
186
- )
187
-
188
- # ── Phase 4: Report generation ──────────────────────────────────────
189
- written = _write_reports(session_db, result, output_dir, output, fmt)
190
-
191
- # ── Phase 5: Clean up session ───────────────────────────────────────
192
- session_db.delete_session(session_id)
193
-
194
- # ── Summary table (printed after DB is closed) ──────────────────────────
195
- _print_summary(result, written)
196
-
197
-
198
- def _detect_clones(
199
- session_db: SessionDB,
200
- session_id: str,
201
- detect_types: set[int],
202
- threshold: float,
203
- ) -> tuple[int, int, int]:
204
- """Run clone detection and return (type1_count, type2_count, type3_count)."""
205
- _console.print("[dim]Detecting clones…[/dim]")
206
- return detect(session_db, session_id, detect_types, threshold)
207
-
208
-
209
- def _write_reports(
210
- session_db: SessionDB,
211
- result: ScanResult,
212
- output_dir: Path,
213
- output: str,
214
- fmt: str,
215
- ) -> list[Path]:
216
- """Write the requested report formats and return a list of written paths."""
217
- written: list[Path] = []
218
- if fmt in ("json", "both"):
219
- written.append(
220
- json_reporter.write(session_db, result, output_dir / f"{output}.json")
221
- )
222
- if fmt in ("html", "both"):
223
- written.append(
224
- html_reporter.write(session_db, result, output_dir / f"{output}.html")
225
- )
226
- return written
227
-
228
-
229
- def _print_summary(result: ScanResult, written: list[Path]) -> None:
230
- """Print the scan summary table and list of saved report paths."""
231
- table = Table(title="Scan Summary", show_header=True, header_style="bold magenta")
232
- table.add_column("Metric", style="dim", min_width=26)
233
- table.add_column("Value", justify="right", style="bold")
234
- table.add_row("Files scanned", str(result.files_scanned))
235
- table.add_row("Fragments extracted", str(result.fragments_extracted))
236
- table.add_row("[red]Type-1[/red] clone groups (exact)", str(result.type1_groups))
237
- table.add_row(
238
- "[yellow]Type-2[/yellow] clone groups (structural)", str(result.type2_groups)
239
- )
240
- table.add_row(
241
- "[green]Type-3[/green] clone groups (near-duplicate)", str(result.type3_groups)
242
- )
243
- _console.print(table)
244
- _console.print("\n[bold]Reports saved:[/bold]")
245
- for dest in written:
246
- _console.print(f" [cyan]•[/cyan] {dest}")
247
-
248
-
249
- def _process_files(
250
- files: list[tuple[Path, str]],
251
- session_db: SessionDB,
252
- session_id: str,
253
- min_tokens: int,
254
- ) -> int:
255
- """Parse, extract, and hash every file; bulk-insert fragments into *session_db*.
256
-
257
- Returns:
258
- Total number of fragments extracted.
259
- """
260
- total = 0
261
- with Progress(
262
- SpinnerColumn(),
263
- TextColumn("[progress.description]{task.description}"),
264
- BarColumn(),
265
- MofNCompleteColumn(),
266
- console=_console,
267
- transient=True,
268
- ) as progress:
269
- task = progress.add_task("Processing files…", total=len(files))
270
- for file_path, language in files:
271
- try:
272
- source_bytes = file_path.read_bytes()
273
- tree = ts_parser.parse(source_bytes, language)
274
- if tree is not None:
275
- frags = extractor.extract_fragments(
276
- tree, source_bytes, file_path, language, session_id, min_tokens
277
- )
278
- fingerprint.hash_all(frags)
279
- session_db.insert_many_fragments(frags)
280
- total += len(frags)
281
- except Exception as exc: # pylint: disable=broad-exception-caught
282
- _console.print(f"[red]Error processing {file_path.name}: {exc}[/red]")
283
- finally:
284
- progress.advance(task)
285
- return total
286
-
287
-
288
- if __name__ == "__main__":
289
- main() # pylint: disable=no-value-for-parameter
1
+ """
2
+ CLI entry point for codeecho.
3
+
4
+ Invoked via::
5
+
6
+ poetry run python -m codeecho [OPTIONS] PATH
7
+ """
8
+
9
+ import uuid
10
+ from pathlib import Path
11
+
12
+ import click
13
+ from braincraft import IgnoreFile
14
+ from rich.console import Console
15
+ from rich.panel import Panel
16
+ from rich.progress import (
17
+ BarColumn,
18
+ MofNCompleteColumn,
19
+ Progress,
20
+ SpinnerColumn,
21
+ TextColumn,
22
+ )
23
+ from rich.table import Table
24
+
25
+ from codeecho import __version__, CONF_DIR
26
+ from codeecho import extractor, fingerprint, parser as ts_parser, scanner
27
+ from codeecho.db import SessionDB, get_db_path
28
+ from codeecho.detector import detect
29
+ from codeecho.models import ScanResult
30
+ from codeecho.reporter import html_reporter, json_reporter
31
+
32
+ _console = Console()
33
+
34
+ _FORMAT_CHOICES = click.Choice(["json", "html", "both"])
35
+
36
+
37
+ def _parse_types(types_str: str) -> set[int]:
38
+ """Parse ``--types`` value into a set of integers."""
39
+ if types_str.strip().lower() == "all":
40
+ return {1, 2, 3}
41
+ result: set[int] = set()
42
+ for token in types_str.split(","):
43
+ token = token.strip()
44
+ if token.isdigit() and token in {"1", "2", "3"}:
45
+ result.add(int(token))
46
+ if not result:
47
+ raise click.BadParameter(
48
+ f"Invalid types value: {types_str!r}. Use 'all' or e.g. '1,2,3'."
49
+ )
50
+ return result
51
+
52
+
53
+ @click.command(
54
+ name="codeecho", context_settings={"help_option_names": ["-h", "--help"]}
55
+ )
56
+ @click.version_option(
57
+ version=__version__, prog_name="codeecho", message="%(prog)s v%(version)s"
58
+ )
59
+ @click.argument(
60
+ "path", type=click.Path(exists=True, file_okay=False, dir_okay=True, path_type=Path)
61
+ )
62
+ @click.option(
63
+ "--types",
64
+ default="all",
65
+ show_default=True,
66
+ metavar="TYPES",
67
+ help="Clone types to detect: comma-separated (e.g. '1,2') or 'all'.",
68
+ )
69
+ @click.option(
70
+ "--threshold",
71
+ default=0.8,
72
+ show_default=True,
73
+ type=click.FloatRange(0.0, 1.0),
74
+ help="Jaccard similarity threshold for Type-3 (near-duplicate) detection.",
75
+ )
76
+ @click.option(
77
+ "--output",
78
+ default="codeecho-output",
79
+ show_default=True,
80
+ metavar="NAME",
81
+ help="Base name (without extension) for output file(s).",
82
+ )
83
+ @click.option(
84
+ "--output-dir",
85
+ default=None,
86
+ show_default=False,
87
+ type=click.Path(file_okay=False, path_type=Path),
88
+ help="Directory where output file(s) will be written. [default: <cwd>/reports]",
89
+ )
90
+ @click.option(
91
+ "--db-dir",
92
+ "db_dir",
93
+ default=None,
94
+ show_default=False,
95
+ type=click.Path(file_okay=False, path_type=Path),
96
+ help="Directory for the SQLite scratch database. [default: ~/.codeecho]",
97
+ )
98
+ @click.option(
99
+ "--format",
100
+ "fmt",
101
+ default="both",
102
+ show_default=True,
103
+ type=_FORMAT_CHOICES,
104
+ help="Output format.",
105
+ )
106
+ @click.option(
107
+ "--min-tokens",
108
+ default=10,
109
+ show_default=True,
110
+ type=click.IntRange(1),
111
+ help="Minimum token count for a fragment to be considered.",
112
+ )
113
+ @click.option(
114
+ "--exclude",
115
+ multiple=True,
116
+ metavar="PATTERN",
117
+ help="Glob pattern(s) to exclude from scanning (repeatable).",
118
+ )
119
+ def main( # pylint: disable=too-many-arguments,too-many-positional-arguments,too-many-locals,invalid-name
120
+ path: Path,
121
+ types: str,
122
+ threshold: float,
123
+ output: str,
124
+ output_dir: Path | None,
125
+ db_dir: Path | None,
126
+ fmt: str,
127
+ min_tokens: int,
128
+ exclude: tuple[str, ...],
129
+ ) -> None:
130
+ """Scan PATH for duplicate and near-duplicate code.
131
+
132
+ Generates JSON and/or HTML reports, then removes the intermediate session data
133
+ from the embedded database.
134
+ """
135
+ _console.print(
136
+ Panel(
137
+ f"[bold cyan]codeecho[/bold cyan] [dim]v{__version__}[/dim] — Code Duplicate Scanner",
138
+ border_style="dim",
139
+ )
140
+ )
141
+
142
+ detect_types = _parse_types(types)
143
+ if output_dir is None:
144
+ output_dir = Path.cwd() / "reports"
145
+ session_id = str(uuid.uuid4())
146
+ config = {
147
+ "types": types,
148
+ "threshold": threshold,
149
+ "min_tokens": min_tokens,
150
+ "exclude": list(exclude),
151
+ }
152
+
153
+ output_dir.mkdir(parents=True, exist_ok=True)
154
+
155
+ db_path = Path(get_db_path(str(db_dir) if db_dir else None))
156
+ with SessionDB(db_path=db_path) as session_db:
157
+ session_db.create_session(session_id, str(path.resolve()), config)
158
+
159
+ # ── Phase 1: File discovery ─────────────────────────────────────────
160
+ _console.print(f"[dim]Scanning:[/dim] [bold]{path.resolve()}[/bold]")
161
+ _ignore_file = IgnoreFile(Path(CONF_DIR) / ".ignore", base_dir=path.resolve())
162
+ files = scanner.scan(path, exclude_patterns=exclude, ignore_file=_ignore_file)
163
+ total_fragments = 0
164
+
165
+ if not files:
166
+ _console.print("[yellow]No supported source files found.[/yellow]")
167
+ session_db.delete_session(session_id)
168
+ return
169
+
170
+ # ── Phase 2: Parse → Extract → Hash ────────────────────────────────
171
+ total_fragments = _process_files(files, session_db, session_id, min_tokens)
172
+
173
+ # ── Phase 3: Clone detection ────────────────────────────────────────
174
+ cnt1, cnt2, cnt3 = _detect_clones(
175
+ session_db, session_id, detect_types, threshold
176
+ )
177
+
178
+ result = ScanResult(
179
+ session_id=session_id,
180
+ scan_path=str(path.resolve()),
181
+ files_scanned=len(files),
182
+ fragments_extracted=total_fragments,
183
+ type1_groups=cnt1,
184
+ type2_groups=cnt2,
185
+ type3_groups=cnt3,
186
+ )
187
+
188
+ # ── Phase 4: Report generation ──────────────────────────────────────
189
+ written = _write_reports(session_db, result, output_dir, output, fmt)
190
+
191
+ # ── Phase 5: Clean up session ───────────────────────────────────────
192
+ session_db.delete_session(session_id)
193
+
194
+ # ── Summary table (printed after DB is closed) ──────────────────────────
195
+ _print_summary(result, written)
196
+
197
+
198
+ def _detect_clones(
199
+ session_db: SessionDB,
200
+ session_id: str,
201
+ detect_types: set[int],
202
+ threshold: float,
203
+ ) -> tuple[int, int, int]:
204
+ """Run clone detection and return (type1_count, type2_count, type3_count)."""
205
+ _console.print("[dim]Detecting clones…[/dim]")
206
+ return detect(session_db, session_id, detect_types, threshold)
207
+
208
+
209
+ def _write_reports(
210
+ session_db: SessionDB,
211
+ result: ScanResult,
212
+ output_dir: Path,
213
+ output: str,
214
+ fmt: str,
215
+ ) -> list[Path]:
216
+ """Write the requested report formats and return a list of written paths."""
217
+ written: list[Path] = []
218
+ if fmt in ("json", "both"):
219
+ written.append(
220
+ json_reporter.write(session_db, result, output_dir / f"{output}.json")
221
+ )
222
+ if fmt in ("html", "both"):
223
+ written.append(
224
+ html_reporter.write(session_db, result, output_dir / f"{output}.html")
225
+ )
226
+ return written
227
+
228
+
229
+ def _print_summary(result: ScanResult, written: list[Path]) -> None:
230
+ """Print the scan summary table and list of saved report paths."""
231
+ table = Table(title="Scan Summary", show_header=True, header_style="bold magenta")
232
+ table.add_column("Metric", style="dim", min_width=26)
233
+ table.add_column("Value", justify="right", style="bold")
234
+ table.add_row("Files scanned", str(result.files_scanned))
235
+ table.add_row("Fragments extracted", str(result.fragments_extracted))
236
+ table.add_row("[red]Type-1[/red] clone groups (exact)", str(result.type1_groups))
237
+ table.add_row(
238
+ "[yellow]Type-2[/yellow] clone groups (structural)", str(result.type2_groups)
239
+ )
240
+ table.add_row(
241
+ "[green]Type-3[/green] clone groups (near-duplicate)", str(result.type3_groups)
242
+ )
243
+ _console.print(table)
244
+ _console.print("\n[bold]Reports saved:[/bold]")
245
+ for dest in written:
246
+ _console.print(f" [cyan]•[/cyan] {dest}")
247
+
248
+
249
+ def _process_files(
250
+ files: list[tuple[Path, str]],
251
+ session_db: SessionDB,
252
+ session_id: str,
253
+ min_tokens: int,
254
+ ) -> int:
255
+ """Parse, extract, and hash every file; bulk-insert fragments into *session_db*.
256
+
257
+ Returns:
258
+ Total number of fragments extracted.
259
+ """
260
+ total = 0
261
+ with Progress(
262
+ SpinnerColumn(),
263
+ TextColumn("[progress.description]{task.description}"),
264
+ BarColumn(),
265
+ MofNCompleteColumn(),
266
+ console=_console,
267
+ transient=True,
268
+ ) as progress:
269
+ task = progress.add_task("Processing files…", total=len(files))
270
+ for file_path, language in files:
271
+ try:
272
+ source_bytes = file_path.read_bytes()
273
+ tree = ts_parser.parse(source_bytes, language)
274
+ if tree is not None:
275
+ frags = extractor.extract_fragments(
276
+ tree, source_bytes, file_path, language, session_id, min_tokens
277
+ )
278
+ fingerprint.hash_all(frags)
279
+ session_db.insert_many_fragments(frags)
280
+ total += len(frags)
281
+ except Exception as exc: # pylint: disable=broad-exception-caught
282
+ _console.print(f"[red]Error processing {file_path.name}: {exc}[/red]")
283
+ finally:
284
+ progress.advance(task)
285
+ return total
286
+
287
+
288
+ if __name__ == "__main__":
289
+ main() # pylint: disable=no-value-for-parameter