codevariability 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,350 @@
1
+ """Normalized Python ASTs and ordered tree edit similarity.
2
+
3
+ The implementation uses the Zhang-Shasha algorithm with unit insertion,
4
+ deletion, and relabeling costs. It deliberately depends only on the Python
5
+ standard library; AST identifiers, literal values, locations, comments, and
6
+ formatting do not become part of the normalized tree.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import ast
12
+ from array import array
13
+ from dataclasses import dataclass
14
+ from itertools import combinations
15
+ from pathlib import Path
16
+ from typing import Mapping
17
+
18
+ import pandas as pd
19
+
20
+ from .exceptions import AnalysisError
21
+ from .normalization import fenced_code_segments
22
+
23
+
24
+ @dataclass(frozen=True, slots=True)
25
+ class NormalizedAstNode:
26
+ """Parser-independent ordered tree node used by the TED implementation."""
27
+
28
+ label: str
29
+ children: tuple["NormalizedAstNode", ...] = ()
30
+
31
+
32
+ @dataclass(frozen=True, slots=True)
33
+ class CodeFragment:
34
+ code: str
35
+ start_line: int
36
+
37
+
38
+ _IGNORED_AST_FIELDS = {
39
+ "lineno",
40
+ "col_offset",
41
+ "end_lineno",
42
+ "end_col_offset",
43
+ "type_comment",
44
+ "type_ignores",
45
+ }
46
+
47
+ # Concrete names are deliberately ignored wherever CPython represents them as
48
+ # strings (Name.id, FunctionDef.name, keyword.arg, alias.name, Attribute.attr,
49
+ # pattern capture names, and similar fields). The following scalar fields are
50
+ # syntax-bearing rather than identifiers or source metadata.
51
+ _STRUCTURAL_SCALAR_FIELDS = ("is_async", "conversion", "level", "simple")
52
+
53
+
54
+ def extract_python_code_fragments(
55
+ source: str, filename: str = "<unknown>"
56
+ ) -> tuple[CodeFragment, ...]:
57
+ """Preserve .py source; parse closed fences only in Markdown/anonymous input."""
58
+ if Path(filename).suffix.lower() == ".py":
59
+ return (CodeFragment(source, 1),)
60
+ segments = (
61
+ fenced_code_segments(source)
62
+ if Path(filename).suffix.lower() in {".md", ".markdown"}
63
+ or filename == "<unknown>"
64
+ else ()
65
+ )
66
+ if segments:
67
+ if any(
68
+ (segment.language or "").casefold() not in {"", "py", "python"}
69
+ for segment in segments
70
+ ):
71
+ raise AnalysisError("A AST Python requer somente fragmentos Python.")
72
+ return tuple(
73
+ CodeFragment(segment.code, segment.start_line)
74
+ for segment in segments
75
+ if segment.code.strip()
76
+ )
77
+ return (CodeFragment(source, 1),)
78
+
79
+
80
+ def _constant_label(value: object) -> str:
81
+ if value is None:
82
+ category = "None"
83
+ elif value is Ellipsis:
84
+ category = "Ellipsis"
85
+ elif isinstance(value, bool):
86
+ category = "Boolean"
87
+ elif isinstance(value, (int, float, complex)):
88
+ category = "Number"
89
+ elif isinstance(value, str):
90
+ category = "String"
91
+ elif isinstance(value, bytes):
92
+ category = "Bytes"
93
+ else:
94
+ category = type(value).__name__
95
+ return f"Constant:{category}"
96
+
97
+
98
+ def _node_label(node: ast.AST) -> str:
99
+ if isinstance(node, ast.Constant):
100
+ return _constant_label(node.value)
101
+ label = type(node).__name__
102
+ if isinstance(node, ast.MatchSingleton):
103
+ return f"{label}:{_constant_label(node.value).partition(':')[2]}"
104
+ attributes = [
105
+ f"{field}={getattr(node, field)}"
106
+ for field in _STRUCTURAL_SCALAR_FIELDS
107
+ if field in getattr(node, "_fields", ())
108
+ ]
109
+ return ":".join((label, *attributes))
110
+
111
+
112
+ def normalize_ast(node: ast.AST | None) -> NormalizedAstNode | None:
113
+ """Convert a Python AST to an ordered tree without names, values, or metadata."""
114
+
115
+ if node is None:
116
+ return None
117
+ if not isinstance(node, ast.AST):
118
+ raise TypeError("normalize_ast() expects an ast.AST node or None")
119
+ # Explicit postorder avoids Python recursion limits on valid long expressions.
120
+ completed: dict[int, NormalizedAstNode] = {}
121
+ pending: list[tuple[ast.AST, bool]] = [(node, False)]
122
+ while pending:
123
+ current, visited = pending.pop()
124
+ children = [
125
+ child
126
+ for name, value in ast.iter_fields(current)
127
+ if name not in _IGNORED_AST_FIELDS
128
+ for child in (
129
+ [value]
130
+ if isinstance(value, ast.AST)
131
+ else value
132
+ if isinstance(value, list)
133
+ else []
134
+ )
135
+ if isinstance(child, ast.AST)
136
+ ]
137
+ if not visited:
138
+ pending.append((current, True))
139
+ pending.extend((child, False) for child in reversed(children))
140
+ else:
141
+ completed[id(current)] = NormalizedAstNode(
142
+ _node_label(current), tuple(completed[id(child)] for child in children)
143
+ )
144
+ return completed[id(node)]
145
+
146
+
147
+ def normalize_python_source(
148
+ source: str, filename: str = "<unknown>"
149
+ ) -> NormalizedAstNode:
150
+ """Parse all source fragments below a synthetic ordered root."""
151
+
152
+ fragments: list[NormalizedAstNode] = []
153
+ for index, fragment in enumerate(extract_python_code_fragments(source, filename)):
154
+ try:
155
+ parsed = ast.parse(fragment.code, filename=filename)
156
+ except SyntaxError as exc:
157
+ source_line = fragment.start_line + (exc.lineno or 1) - 1
158
+ raise AnalysisError(
159
+ f"{filename} (fragmento {index + 1}, linha de origem {source_line}): {exc.msg}."
160
+ ) from exc
161
+ except (ValueError, RecursionError, MemoryError) as exc:
162
+ raise AnalysisError(
163
+ f"{filename}: a fonte excede os limites do parser Python ou é inválida: {exc}."
164
+ ) from exc
165
+ if not parsed.body:
166
+ continue
167
+ normalized = normalize_ast(parsed)
168
+ assert normalized is not None
169
+ fragments.append(NormalizedAstNode("Fragment", (normalized,)))
170
+ return NormalizedAstNode("SyntheticProgram", tuple(fragments))
171
+
172
+
173
+ def count_ast_nodes(tree: NormalizedAstNode | None) -> int:
174
+ """Count normalized nodes iteratively, including the root."""
175
+
176
+ if tree is None:
177
+ return 0
178
+ count = 0
179
+ stack = [tree]
180
+ while stack:
181
+ current = stack.pop()
182
+ count += 1
183
+ stack.extend(current.children)
184
+ return count
185
+
186
+
187
+ def _trees_equal(left: NormalizedAstNode, right: NormalizedAstNode) -> bool:
188
+ pending = [(left, right)]
189
+ while pending:
190
+ left_node, right_node = pending.pop()
191
+ if left_node.label != right_node.label or len(left_node.children) != len(
192
+ right_node.children
193
+ ):
194
+ return False
195
+ pending.extend(zip(left_node.children, right_node.children, strict=True))
196
+ return True
197
+
198
+
199
+ def _postorder(
200
+ tree: NormalizedAstNode,
201
+ ) -> tuple[list[NormalizedAstNode], list[int], list[int]]:
202
+ # Index zero is unused by the one-based algorithm; its placeholder is typed.
203
+ nodes: list[NormalizedAstNode] = [tree]
204
+ leftmost = [0]
205
+ completed: list[tuple[int, int]] = []
206
+ stack: list[tuple[NormalizedAstNode, bool]] = [(tree, False)]
207
+ while stack:
208
+ node, visited = stack.pop()
209
+ if not visited:
210
+ stack.append((node, True))
211
+ stack.extend((child, False) for child in reversed(node.children))
212
+ continue
213
+ index = len(nodes)
214
+ descendant = completed[-len(node.children)][1] if node.children else index
215
+ if node.children:
216
+ del completed[-len(node.children) :]
217
+ nodes.append(node)
218
+ leftmost.append(descendant)
219
+ completed.append((index, descendant))
220
+
221
+ last_for_leftmost: dict[int, int] = {}
222
+ for index in range(1, len(nodes)):
223
+ last_for_leftmost[leftmost[index]] = index
224
+ return nodes, leftmost, sorted(last_for_leftmost.values())
225
+
226
+
227
+ def tree_edit_distance(
228
+ left: NormalizedAstNode | None,
229
+ right: NormalizedAstNode | None,
230
+ *,
231
+ max_cells: int | None = 2_000_000,
232
+ ) -> int:
233
+ """Return ordered Zhang-Shasha TED with unit node-operation costs."""
234
+
235
+ if max_cells is not None and (
236
+ not isinstance(max_cells, int) or isinstance(max_cells, bool) or max_cells < 1
237
+ ):
238
+ raise AnalysisError("max_ted_cells deve ser um inteiro positivo ou None.")
239
+ if left is None:
240
+ return count_ast_nodes(right)
241
+ if right is None:
242
+ return count_ast_nodes(left)
243
+ if _trees_equal(left, right):
244
+ return 0
245
+
246
+ left_nodes, leftmost_left, left_keyroots = _postorder(left)
247
+ right_nodes, leftmost_right, right_keyroots = _postorder(right)
248
+ right_count = len(right_nodes) - 1
249
+ cells = len(left_nodes) * len(right_nodes)
250
+ if max_cells is not None and cells > max_cells:
251
+ raise AnalysisError(
252
+ f"TED requer {cells} células; limite max_ted_cells={max_cells}. Selecione métricas rápidas ou aumente explicitamente o limite."
253
+ )
254
+ tree_distances = array("I", [0]) * cells
255
+
256
+ for left_root in left_keyroots:
257
+ left_start = leftmost_left[left_root]
258
+ rows = left_root - left_start + 2
259
+ for right_root in right_keyroots:
260
+ right_start = leftmost_right[right_root]
261
+ columns = right_root - right_start + 2
262
+ forest = array("I", [0]) * (rows * columns)
263
+ for row in range(1, rows):
264
+ forest[row * columns] = row
265
+ for column in range(1, columns):
266
+ forest[column] = column
267
+
268
+ for left_index in range(left_start, left_root + 1):
269
+ row = left_index - left_start + 1
270
+ for right_index in range(right_start, right_root + 1):
271
+ column = right_index - right_start + 1
272
+ deletion = forest[(row - 1) * columns + column] + 1
273
+ insertion = forest[row * columns + column - 1] + 1
274
+ if (
275
+ leftmost_left[left_index] == left_start
276
+ and leftmost_right[right_index] == right_start
277
+ ):
278
+ replacement = forest[(row - 1) * columns + column - 1] + (
279
+ left_nodes[left_index].label
280
+ != right_nodes[right_index].label
281
+ )
282
+ value = min(deletion, insertion, replacement)
283
+ tree_distances[left_index * (right_count + 1) + right_index] = (
284
+ value
285
+ )
286
+ else:
287
+ prefix_row = leftmost_left[left_index] - left_start
288
+ prefix_column = leftmost_right[right_index] - right_start
289
+ subtree = tree_distances[
290
+ left_index * (right_count + 1) + right_index
291
+ ]
292
+ value = min(
293
+ deletion,
294
+ insertion,
295
+ forest[prefix_row * columns + prefix_column] + subtree,
296
+ )
297
+ forest[row * columns + column] = value
298
+
299
+ return int(tree_distances[(len(left_nodes) - 1) * (right_count + 1) + right_count])
300
+
301
+
302
+ def tree_edit_similarity(
303
+ left: NormalizedAstNode | None,
304
+ right: NormalizedAstNode | None,
305
+ *,
306
+ max_cells: int | None = 2_000_000,
307
+ ) -> float:
308
+ """Return ``1 - TED / upper_bound`` using a proven unit-cost bound.
309
+
310
+ For two non-empty trees, deleting every non-root node, relabeling the root,
311
+ and inserting every target non-root node costs at most ``n + m - 1``.
312
+ For an empty/non-empty pair the exact upper bound is the non-empty size.
313
+ """
314
+
315
+ left_size = count_ast_nodes(left)
316
+ right_size = count_ast_nodes(right)
317
+ if left_size == 0 and right_size == 0:
318
+ return 1.0
319
+ if left_size == 0 or right_size == 0:
320
+ return 0.0
321
+ denominator = left_size + right_size - 1
322
+ distance = tree_edit_distance(left, right, max_cells=max_cells)
323
+ if distance > denominator:
324
+ raise RuntimeError("tree edit distance exceeded its normalization upper bound")
325
+ return 1.0 - distance / denominator
326
+
327
+
328
+ def ast_tree_edit_similarity(
329
+ sources: Mapping[str, str], *, max_cells: int | None = 2_000_000
330
+ ) -> pd.DataFrame:
331
+ """Calculate the public ``ast_tree_edit_similarity`` pairwise matrix."""
332
+
333
+ names = list(sources)
334
+ normalized = {
335
+ name: normalize_python_source(source, name) for name, source in sources.items()
336
+ }
337
+ # A synthetic container with no fragments is an empty program, not a
338
+ # syntax node that should contribute positive overlap with real code.
339
+ trees = {name: tree if tree.children else None for name, tree in normalized.items()}
340
+ matrix = pd.DataFrame(0.0, index=names, columns=names, dtype=float)
341
+ for name in names:
342
+ matrix.loc[name, name] = 1.0
343
+ for left, right in combinations(names, 2):
344
+ value = tree_edit_similarity(trees[left], trees[right], max_cells=max_cells)
345
+ matrix.loc[left, right] = matrix.loc[right, left] = value
346
+ return matrix
347
+
348
+
349
+ AST_TREE_EDIT_METRIC = "ast_tree_edit_similarity"
350
+ AST_TREE_EDIT_METRIC_ID = "ast_tree_edit_similarity_v2"
codevariability/cli.py ADDED
@@ -0,0 +1,75 @@
1
+ """Command-line entry point."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+
7
+ from . import __version__
8
+ from .analysis import analyze
9
+ from .exceptions import AnalysisError
10
+ from .metrics import METRICS
11
+
12
+
13
+ def _cell_limit(value: str) -> int | None:
14
+ if value == "none":
15
+ return None
16
+ try:
17
+ limit = int(value)
18
+ except ValueError as exc:
19
+ raise argparse.ArgumentTypeError("Use um inteiro positivo ou 'none'.") from exc
20
+ if limit < 1:
21
+ raise argparse.ArgumentTypeError("Use um inteiro positivo ou 'none'.")
22
+ return limit
23
+
24
+
25
+ def main() -> None:
26
+ parser = argparse.ArgumentParser(
27
+ prog="codevariability",
28
+ description="Analisa similaridade e variabilidade de código.",
29
+ )
30
+ parser.add_argument(
31
+ "--version", action="version", version=f"%(prog)s {__version__}"
32
+ )
33
+ subcommands = parser.add_subparsers(dest="command", required=True)
34
+ command = subcommands.add_parser("analyze", help="Analisa uma pasta de código.")
35
+ command.add_argument("directory")
36
+ command.add_argument(
37
+ "--metrics", nargs="+", choices=[*METRICS, "all"], default=["all"]
38
+ )
39
+ command.add_argument(
40
+ "--extensions",
41
+ nargs="+",
42
+ help="Filtra extensões no diretório, por exemplo: py java rs.",
43
+ )
44
+ command.add_argument("--output", help="Diretório para JSON/CSV ou arquivo .xlsx.")
45
+ command.add_argument(
46
+ "--max-ted-cells",
47
+ type=_cell_limit,
48
+ default=2_000_000,
49
+ help="Limite de células por par AST (padrão: 2000000); 'none' remove o limite.",
50
+ )
51
+ args = parser.parse_args()
52
+ if "all" in args.metrics and args.metrics != ["all"]:
53
+ parser.error("--metrics all não pode ser combinado com métricas individuais")
54
+ metrics = "all" if args.metrics == ["all"] else args.metrics
55
+ try:
56
+ result = analyze(
57
+ args.directory,
58
+ metrics=metrics,
59
+ extensions=args.extensions,
60
+ max_ted_cells=args.max_ted_cells,
61
+ )
62
+ if args.output:
63
+ result.to_excel(args.output) if args.output.lower().endswith(
64
+ ".xlsx"
65
+ ) else result.export(args.output)
66
+ except (AnalysisError, OSError) as exc:
67
+ parser.exit(1, f"codevariability: {exc}\n")
68
+ for metric, stats in result.statistics.iterrows():
69
+ print(
70
+ f"{metric}: similaridade média={stats['mean_similarity']!s}; variabilidade média={stats['mean_variability']!s}"
71
+ )
72
+
73
+
74
+ if __name__ == "__main__":
75
+ main()
@@ -0,0 +1,2 @@
1
+ class AnalysisError(Exception):
2
+ """Raised when input code cannot be analyzed."""