codevariability 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codevariability/__init__.py +23 -0
- codevariability/analysis.py +544 -0
- codevariability/ast_tree_edit.py +350 -0
- codevariability/cli.py +75 -0
- codevariability/exceptions.py +2 -0
- codevariability/group_comparison.py +979 -0
- codevariability/interop.py +94 -0
- codevariability/metrics.py +89 -0
- codevariability/normalization.py +257 -0
- codevariability/output.py +76 -0
- codevariability/spreadsheet.py +68 -0
- codevariability/validation.py +115 -0
- codevariability-0.2.0.dist-info/METADATA +171 -0
- codevariability-0.2.0.dist-info/RECORD +18 -0
- codevariability-0.2.0.dist-info/WHEEL +5 -0
- codevariability-0.2.0.dist-info/entry_points.txt +2 -0
- codevariability-0.2.0.dist-info/licenses/LICENSE +21 -0
- codevariability-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,350 @@
|
|
|
1
|
+
"""Normalized Python ASTs and ordered tree edit similarity.
|
|
2
|
+
|
|
3
|
+
The implementation uses the Zhang-Shasha algorithm with unit insertion,
|
|
4
|
+
deletion, and relabeling costs. It deliberately depends only on the Python
|
|
5
|
+
standard library; AST identifiers, literal values, locations, comments, and
|
|
6
|
+
formatting do not become part of the normalized tree.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import ast
|
|
12
|
+
from array import array
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from itertools import combinations
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Mapping
|
|
17
|
+
|
|
18
|
+
import pandas as pd
|
|
19
|
+
|
|
20
|
+
from .exceptions import AnalysisError
|
|
21
|
+
from .normalization import fenced_code_segments
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True, slots=True)
|
|
25
|
+
class NormalizedAstNode:
|
|
26
|
+
"""Parser-independent ordered tree node used by the TED implementation."""
|
|
27
|
+
|
|
28
|
+
label: str
|
|
29
|
+
children: tuple["NormalizedAstNode", ...] = ()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True, slots=True)
|
|
33
|
+
class CodeFragment:
|
|
34
|
+
code: str
|
|
35
|
+
start_line: int
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
_IGNORED_AST_FIELDS = {
|
|
39
|
+
"lineno",
|
|
40
|
+
"col_offset",
|
|
41
|
+
"end_lineno",
|
|
42
|
+
"end_col_offset",
|
|
43
|
+
"type_comment",
|
|
44
|
+
"type_ignores",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
# Concrete names are deliberately ignored wherever CPython represents them as
|
|
48
|
+
# strings (Name.id, FunctionDef.name, keyword.arg, alias.name, Attribute.attr,
|
|
49
|
+
# pattern capture names, and similar fields). The following scalar fields are
|
|
50
|
+
# syntax-bearing rather than identifiers or source metadata.
|
|
51
|
+
_STRUCTURAL_SCALAR_FIELDS = ("is_async", "conversion", "level", "simple")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def extract_python_code_fragments(
|
|
55
|
+
source: str, filename: str = "<unknown>"
|
|
56
|
+
) -> tuple[CodeFragment, ...]:
|
|
57
|
+
"""Preserve .py source; parse closed fences only in Markdown/anonymous input."""
|
|
58
|
+
if Path(filename).suffix.lower() == ".py":
|
|
59
|
+
return (CodeFragment(source, 1),)
|
|
60
|
+
segments = (
|
|
61
|
+
fenced_code_segments(source)
|
|
62
|
+
if Path(filename).suffix.lower() in {".md", ".markdown"}
|
|
63
|
+
or filename == "<unknown>"
|
|
64
|
+
else ()
|
|
65
|
+
)
|
|
66
|
+
if segments:
|
|
67
|
+
if any(
|
|
68
|
+
(segment.language or "").casefold() not in {"", "py", "python"}
|
|
69
|
+
for segment in segments
|
|
70
|
+
):
|
|
71
|
+
raise AnalysisError("A AST Python requer somente fragmentos Python.")
|
|
72
|
+
return tuple(
|
|
73
|
+
CodeFragment(segment.code, segment.start_line)
|
|
74
|
+
for segment in segments
|
|
75
|
+
if segment.code.strip()
|
|
76
|
+
)
|
|
77
|
+
return (CodeFragment(source, 1),)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _constant_label(value: object) -> str:
|
|
81
|
+
if value is None:
|
|
82
|
+
category = "None"
|
|
83
|
+
elif value is Ellipsis:
|
|
84
|
+
category = "Ellipsis"
|
|
85
|
+
elif isinstance(value, bool):
|
|
86
|
+
category = "Boolean"
|
|
87
|
+
elif isinstance(value, (int, float, complex)):
|
|
88
|
+
category = "Number"
|
|
89
|
+
elif isinstance(value, str):
|
|
90
|
+
category = "String"
|
|
91
|
+
elif isinstance(value, bytes):
|
|
92
|
+
category = "Bytes"
|
|
93
|
+
else:
|
|
94
|
+
category = type(value).__name__
|
|
95
|
+
return f"Constant:{category}"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _node_label(node: ast.AST) -> str:
|
|
99
|
+
if isinstance(node, ast.Constant):
|
|
100
|
+
return _constant_label(node.value)
|
|
101
|
+
label = type(node).__name__
|
|
102
|
+
if isinstance(node, ast.MatchSingleton):
|
|
103
|
+
return f"{label}:{_constant_label(node.value).partition(':')[2]}"
|
|
104
|
+
attributes = [
|
|
105
|
+
f"{field}={getattr(node, field)}"
|
|
106
|
+
for field in _STRUCTURAL_SCALAR_FIELDS
|
|
107
|
+
if field in getattr(node, "_fields", ())
|
|
108
|
+
]
|
|
109
|
+
return ":".join((label, *attributes))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def normalize_ast(node: ast.AST | None) -> NormalizedAstNode | None:
|
|
113
|
+
"""Convert a Python AST to an ordered tree without names, values, or metadata."""
|
|
114
|
+
|
|
115
|
+
if node is None:
|
|
116
|
+
return None
|
|
117
|
+
if not isinstance(node, ast.AST):
|
|
118
|
+
raise TypeError("normalize_ast() expects an ast.AST node or None")
|
|
119
|
+
# Explicit postorder avoids Python recursion limits on valid long expressions.
|
|
120
|
+
completed: dict[int, NormalizedAstNode] = {}
|
|
121
|
+
pending: list[tuple[ast.AST, bool]] = [(node, False)]
|
|
122
|
+
while pending:
|
|
123
|
+
current, visited = pending.pop()
|
|
124
|
+
children = [
|
|
125
|
+
child
|
|
126
|
+
for name, value in ast.iter_fields(current)
|
|
127
|
+
if name not in _IGNORED_AST_FIELDS
|
|
128
|
+
for child in (
|
|
129
|
+
[value]
|
|
130
|
+
if isinstance(value, ast.AST)
|
|
131
|
+
else value
|
|
132
|
+
if isinstance(value, list)
|
|
133
|
+
else []
|
|
134
|
+
)
|
|
135
|
+
if isinstance(child, ast.AST)
|
|
136
|
+
]
|
|
137
|
+
if not visited:
|
|
138
|
+
pending.append((current, True))
|
|
139
|
+
pending.extend((child, False) for child in reversed(children))
|
|
140
|
+
else:
|
|
141
|
+
completed[id(current)] = NormalizedAstNode(
|
|
142
|
+
_node_label(current), tuple(completed[id(child)] for child in children)
|
|
143
|
+
)
|
|
144
|
+
return completed[id(node)]
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def normalize_python_source(
|
|
148
|
+
source: str, filename: str = "<unknown>"
|
|
149
|
+
) -> NormalizedAstNode:
|
|
150
|
+
"""Parse all source fragments below a synthetic ordered root."""
|
|
151
|
+
|
|
152
|
+
fragments: list[NormalizedAstNode] = []
|
|
153
|
+
for index, fragment in enumerate(extract_python_code_fragments(source, filename)):
|
|
154
|
+
try:
|
|
155
|
+
parsed = ast.parse(fragment.code, filename=filename)
|
|
156
|
+
except SyntaxError as exc:
|
|
157
|
+
source_line = fragment.start_line + (exc.lineno or 1) - 1
|
|
158
|
+
raise AnalysisError(
|
|
159
|
+
f"{filename} (fragmento {index + 1}, linha de origem {source_line}): {exc.msg}."
|
|
160
|
+
) from exc
|
|
161
|
+
except (ValueError, RecursionError, MemoryError) as exc:
|
|
162
|
+
raise AnalysisError(
|
|
163
|
+
f"{filename}: a fonte excede os limites do parser Python ou é inválida: {exc}."
|
|
164
|
+
) from exc
|
|
165
|
+
if not parsed.body:
|
|
166
|
+
continue
|
|
167
|
+
normalized = normalize_ast(parsed)
|
|
168
|
+
assert normalized is not None
|
|
169
|
+
fragments.append(NormalizedAstNode("Fragment", (normalized,)))
|
|
170
|
+
return NormalizedAstNode("SyntheticProgram", tuple(fragments))
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def count_ast_nodes(tree: NormalizedAstNode | None) -> int:
|
|
174
|
+
"""Count normalized nodes iteratively, including the root."""
|
|
175
|
+
|
|
176
|
+
if tree is None:
|
|
177
|
+
return 0
|
|
178
|
+
count = 0
|
|
179
|
+
stack = [tree]
|
|
180
|
+
while stack:
|
|
181
|
+
current = stack.pop()
|
|
182
|
+
count += 1
|
|
183
|
+
stack.extend(current.children)
|
|
184
|
+
return count
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _trees_equal(left: NormalizedAstNode, right: NormalizedAstNode) -> bool:
|
|
188
|
+
pending = [(left, right)]
|
|
189
|
+
while pending:
|
|
190
|
+
left_node, right_node = pending.pop()
|
|
191
|
+
if left_node.label != right_node.label or len(left_node.children) != len(
|
|
192
|
+
right_node.children
|
|
193
|
+
):
|
|
194
|
+
return False
|
|
195
|
+
pending.extend(zip(left_node.children, right_node.children, strict=True))
|
|
196
|
+
return True
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _postorder(
|
|
200
|
+
tree: NormalizedAstNode,
|
|
201
|
+
) -> tuple[list[NormalizedAstNode], list[int], list[int]]:
|
|
202
|
+
# Index zero is unused by the one-based algorithm; its placeholder is typed.
|
|
203
|
+
nodes: list[NormalizedAstNode] = [tree]
|
|
204
|
+
leftmost = [0]
|
|
205
|
+
completed: list[tuple[int, int]] = []
|
|
206
|
+
stack: list[tuple[NormalizedAstNode, bool]] = [(tree, False)]
|
|
207
|
+
while stack:
|
|
208
|
+
node, visited = stack.pop()
|
|
209
|
+
if not visited:
|
|
210
|
+
stack.append((node, True))
|
|
211
|
+
stack.extend((child, False) for child in reversed(node.children))
|
|
212
|
+
continue
|
|
213
|
+
index = len(nodes)
|
|
214
|
+
descendant = completed[-len(node.children)][1] if node.children else index
|
|
215
|
+
if node.children:
|
|
216
|
+
del completed[-len(node.children) :]
|
|
217
|
+
nodes.append(node)
|
|
218
|
+
leftmost.append(descendant)
|
|
219
|
+
completed.append((index, descendant))
|
|
220
|
+
|
|
221
|
+
last_for_leftmost: dict[int, int] = {}
|
|
222
|
+
for index in range(1, len(nodes)):
|
|
223
|
+
last_for_leftmost[leftmost[index]] = index
|
|
224
|
+
return nodes, leftmost, sorted(last_for_leftmost.values())
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def tree_edit_distance(
|
|
228
|
+
left: NormalizedAstNode | None,
|
|
229
|
+
right: NormalizedAstNode | None,
|
|
230
|
+
*,
|
|
231
|
+
max_cells: int | None = 2_000_000,
|
|
232
|
+
) -> int:
|
|
233
|
+
"""Return ordered Zhang-Shasha TED with unit node-operation costs."""
|
|
234
|
+
|
|
235
|
+
if max_cells is not None and (
|
|
236
|
+
not isinstance(max_cells, int) or isinstance(max_cells, bool) or max_cells < 1
|
|
237
|
+
):
|
|
238
|
+
raise AnalysisError("max_ted_cells deve ser um inteiro positivo ou None.")
|
|
239
|
+
if left is None:
|
|
240
|
+
return count_ast_nodes(right)
|
|
241
|
+
if right is None:
|
|
242
|
+
return count_ast_nodes(left)
|
|
243
|
+
if _trees_equal(left, right):
|
|
244
|
+
return 0
|
|
245
|
+
|
|
246
|
+
left_nodes, leftmost_left, left_keyroots = _postorder(left)
|
|
247
|
+
right_nodes, leftmost_right, right_keyroots = _postorder(right)
|
|
248
|
+
right_count = len(right_nodes) - 1
|
|
249
|
+
cells = len(left_nodes) * len(right_nodes)
|
|
250
|
+
if max_cells is not None and cells > max_cells:
|
|
251
|
+
raise AnalysisError(
|
|
252
|
+
f"TED requer {cells} células; limite max_ted_cells={max_cells}. Selecione métricas rápidas ou aumente explicitamente o limite."
|
|
253
|
+
)
|
|
254
|
+
tree_distances = array("I", [0]) * cells
|
|
255
|
+
|
|
256
|
+
for left_root in left_keyroots:
|
|
257
|
+
left_start = leftmost_left[left_root]
|
|
258
|
+
rows = left_root - left_start + 2
|
|
259
|
+
for right_root in right_keyroots:
|
|
260
|
+
right_start = leftmost_right[right_root]
|
|
261
|
+
columns = right_root - right_start + 2
|
|
262
|
+
forest = array("I", [0]) * (rows * columns)
|
|
263
|
+
for row in range(1, rows):
|
|
264
|
+
forest[row * columns] = row
|
|
265
|
+
for column in range(1, columns):
|
|
266
|
+
forest[column] = column
|
|
267
|
+
|
|
268
|
+
for left_index in range(left_start, left_root + 1):
|
|
269
|
+
row = left_index - left_start + 1
|
|
270
|
+
for right_index in range(right_start, right_root + 1):
|
|
271
|
+
column = right_index - right_start + 1
|
|
272
|
+
deletion = forest[(row - 1) * columns + column] + 1
|
|
273
|
+
insertion = forest[row * columns + column - 1] + 1
|
|
274
|
+
if (
|
|
275
|
+
leftmost_left[left_index] == left_start
|
|
276
|
+
and leftmost_right[right_index] == right_start
|
|
277
|
+
):
|
|
278
|
+
replacement = forest[(row - 1) * columns + column - 1] + (
|
|
279
|
+
left_nodes[left_index].label
|
|
280
|
+
!= right_nodes[right_index].label
|
|
281
|
+
)
|
|
282
|
+
value = min(deletion, insertion, replacement)
|
|
283
|
+
tree_distances[left_index * (right_count + 1) + right_index] = (
|
|
284
|
+
value
|
|
285
|
+
)
|
|
286
|
+
else:
|
|
287
|
+
prefix_row = leftmost_left[left_index] - left_start
|
|
288
|
+
prefix_column = leftmost_right[right_index] - right_start
|
|
289
|
+
subtree = tree_distances[
|
|
290
|
+
left_index * (right_count + 1) + right_index
|
|
291
|
+
]
|
|
292
|
+
value = min(
|
|
293
|
+
deletion,
|
|
294
|
+
insertion,
|
|
295
|
+
forest[prefix_row * columns + prefix_column] + subtree,
|
|
296
|
+
)
|
|
297
|
+
forest[row * columns + column] = value
|
|
298
|
+
|
|
299
|
+
return int(tree_distances[(len(left_nodes) - 1) * (right_count + 1) + right_count])
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def tree_edit_similarity(
|
|
303
|
+
left: NormalizedAstNode | None,
|
|
304
|
+
right: NormalizedAstNode | None,
|
|
305
|
+
*,
|
|
306
|
+
max_cells: int | None = 2_000_000,
|
|
307
|
+
) -> float:
|
|
308
|
+
"""Return ``1 - TED / upper_bound`` using a proven unit-cost bound.
|
|
309
|
+
|
|
310
|
+
For two non-empty trees, deleting every non-root node, relabeling the root,
|
|
311
|
+
and inserting every target non-root node costs at most ``n + m - 1``.
|
|
312
|
+
For an empty/non-empty pair the exact upper bound is the non-empty size.
|
|
313
|
+
"""
|
|
314
|
+
|
|
315
|
+
left_size = count_ast_nodes(left)
|
|
316
|
+
right_size = count_ast_nodes(right)
|
|
317
|
+
if left_size == 0 and right_size == 0:
|
|
318
|
+
return 1.0
|
|
319
|
+
if left_size == 0 or right_size == 0:
|
|
320
|
+
return 0.0
|
|
321
|
+
denominator = left_size + right_size - 1
|
|
322
|
+
distance = tree_edit_distance(left, right, max_cells=max_cells)
|
|
323
|
+
if distance > denominator:
|
|
324
|
+
raise RuntimeError("tree edit distance exceeded its normalization upper bound")
|
|
325
|
+
return 1.0 - distance / denominator
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def ast_tree_edit_similarity(
|
|
329
|
+
sources: Mapping[str, str], *, max_cells: int | None = 2_000_000
|
|
330
|
+
) -> pd.DataFrame:
|
|
331
|
+
"""Calculate the public ``ast_tree_edit_similarity`` pairwise matrix."""
|
|
332
|
+
|
|
333
|
+
names = list(sources)
|
|
334
|
+
normalized = {
|
|
335
|
+
name: normalize_python_source(source, name) for name, source in sources.items()
|
|
336
|
+
}
|
|
337
|
+
# A synthetic container with no fragments is an empty program, not a
|
|
338
|
+
# syntax node that should contribute positive overlap with real code.
|
|
339
|
+
trees = {name: tree if tree.children else None for name, tree in normalized.items()}
|
|
340
|
+
matrix = pd.DataFrame(0.0, index=names, columns=names, dtype=float)
|
|
341
|
+
for name in names:
|
|
342
|
+
matrix.loc[name, name] = 1.0
|
|
343
|
+
for left, right in combinations(names, 2):
|
|
344
|
+
value = tree_edit_similarity(trees[left], trees[right], max_cells=max_cells)
|
|
345
|
+
matrix.loc[left, right] = matrix.loc[right, left] = value
|
|
346
|
+
return matrix
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
AST_TREE_EDIT_METRIC = "ast_tree_edit_similarity"
|
|
350
|
+
AST_TREE_EDIT_METRIC_ID = "ast_tree_edit_similarity_v2"
|
codevariability/cli.py
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""Command-line entry point."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
|
|
7
|
+
from . import __version__
|
|
8
|
+
from .analysis import analyze
|
|
9
|
+
from .exceptions import AnalysisError
|
|
10
|
+
from .metrics import METRICS
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _cell_limit(value: str) -> int | None:
|
|
14
|
+
if value == "none":
|
|
15
|
+
return None
|
|
16
|
+
try:
|
|
17
|
+
limit = int(value)
|
|
18
|
+
except ValueError as exc:
|
|
19
|
+
raise argparse.ArgumentTypeError("Use um inteiro positivo ou 'none'.") from exc
|
|
20
|
+
if limit < 1:
|
|
21
|
+
raise argparse.ArgumentTypeError("Use um inteiro positivo ou 'none'.")
|
|
22
|
+
return limit
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def main() -> None:
|
|
26
|
+
parser = argparse.ArgumentParser(
|
|
27
|
+
prog="codevariability",
|
|
28
|
+
description="Analisa similaridade e variabilidade de código.",
|
|
29
|
+
)
|
|
30
|
+
parser.add_argument(
|
|
31
|
+
"--version", action="version", version=f"%(prog)s {__version__}"
|
|
32
|
+
)
|
|
33
|
+
subcommands = parser.add_subparsers(dest="command", required=True)
|
|
34
|
+
command = subcommands.add_parser("analyze", help="Analisa uma pasta de código.")
|
|
35
|
+
command.add_argument("directory")
|
|
36
|
+
command.add_argument(
|
|
37
|
+
"--metrics", nargs="+", choices=[*METRICS, "all"], default=["all"]
|
|
38
|
+
)
|
|
39
|
+
command.add_argument(
|
|
40
|
+
"--extensions",
|
|
41
|
+
nargs="+",
|
|
42
|
+
help="Filtra extensões no diretório, por exemplo: py java rs.",
|
|
43
|
+
)
|
|
44
|
+
command.add_argument("--output", help="Diretório para JSON/CSV ou arquivo .xlsx.")
|
|
45
|
+
command.add_argument(
|
|
46
|
+
"--max-ted-cells",
|
|
47
|
+
type=_cell_limit,
|
|
48
|
+
default=2_000_000,
|
|
49
|
+
help="Limite de células por par AST (padrão: 2000000); 'none' remove o limite.",
|
|
50
|
+
)
|
|
51
|
+
args = parser.parse_args()
|
|
52
|
+
if "all" in args.metrics and args.metrics != ["all"]:
|
|
53
|
+
parser.error("--metrics all não pode ser combinado com métricas individuais")
|
|
54
|
+
metrics = "all" if args.metrics == ["all"] else args.metrics
|
|
55
|
+
try:
|
|
56
|
+
result = analyze(
|
|
57
|
+
args.directory,
|
|
58
|
+
metrics=metrics,
|
|
59
|
+
extensions=args.extensions,
|
|
60
|
+
max_ted_cells=args.max_ted_cells,
|
|
61
|
+
)
|
|
62
|
+
if args.output:
|
|
63
|
+
result.to_excel(args.output) if args.output.lower().endswith(
|
|
64
|
+
".xlsx"
|
|
65
|
+
) else result.export(args.output)
|
|
66
|
+
except (AnalysisError, OSError) as exc:
|
|
67
|
+
parser.exit(1, f"codevariability: {exc}\n")
|
|
68
|
+
for metric, stats in result.statistics.iterrows():
|
|
69
|
+
print(
|
|
70
|
+
f"{metric}: similaridade média={stats['mean_similarity']!s}; variabilidade média={stats['mean_variability']!s}"
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
if __name__ == "__main__":
|
|
75
|
+
main()
|