codevariability 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codevariability/__init__.py +23 -0
- codevariability/analysis.py +544 -0
- codevariability/ast_tree_edit.py +350 -0
- codevariability/cli.py +75 -0
- codevariability/exceptions.py +2 -0
- codevariability/group_comparison.py +979 -0
- codevariability/interop.py +94 -0
- codevariability/metrics.py +89 -0
- codevariability/normalization.py +257 -0
- codevariability/output.py +76 -0
- codevariability/spreadsheet.py +68 -0
- codevariability/validation.py +115 -0
- codevariability-0.2.0.dist-info/METADATA +171 -0
- codevariability-0.2.0.dist-info/RECORD +18 -0
- codevariability-0.2.0.dist-info/WHEEL +5 -0
- codevariability-0.2.0.dist-info/entry_points.txt +2 -0
- codevariability-0.2.0.dist-info/licenses/LICENSE +21 -0
- codevariability-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Public API for multidimensional source-code similarity analysis."""
|
|
2
|
+
|
|
3
|
+
from .analysis import AnalysisResult, CodeDataset, analyze
|
|
4
|
+
from .exceptions import AnalysisError
|
|
5
|
+
from .group_comparison import (
|
|
6
|
+
GroupComparisonResult,
|
|
7
|
+
MultiGroupComparisonResult,
|
|
8
|
+
compare_groups,
|
|
9
|
+
)
|
|
10
|
+
from .interop import load_matrix_json
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"AnalysisError",
|
|
14
|
+
"AnalysisResult",
|
|
15
|
+
"CodeDataset",
|
|
16
|
+
"GroupComparisonResult",
|
|
17
|
+
"MultiGroupComparisonResult",
|
|
18
|
+
"analyze",
|
|
19
|
+
"compare_groups",
|
|
20
|
+
"load_matrix_json",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
__version__ = "0.2.0"
|
|
@@ -0,0 +1,544 @@
|
|
|
1
|
+
"""Dataset loading, pairwise analysis, summaries, and result export."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
import platform
|
|
9
|
+
import re
|
|
10
|
+
from copy import deepcopy
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from datetime import datetime, timezone
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from types import MappingProxyType
|
|
15
|
+
from typing import Iterable, Mapping, Sequence
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import pandas as pd
|
|
19
|
+
import pygments
|
|
20
|
+
import rapidfuzz
|
|
21
|
+
import sklearn
|
|
22
|
+
|
|
23
|
+
from .exceptions import AnalysisError
|
|
24
|
+
from .metrics import CALCULATORS, METRIC_DIMENSIONS, METRIC_IDS, METRICS
|
|
25
|
+
from .normalization import CODE_TOKENIZATION, TEXT_NORMALIZATION, uses_python_code
|
|
26
|
+
from .output import excel_writer, frame_json, write_json
|
|
27
|
+
from .spreadsheet import safe_to_csv, safe_to_excel
|
|
28
|
+
from .validation import (
|
|
29
|
+
DIMENSION_SCORE_COLUMNS,
|
|
30
|
+
dimensions,
|
|
31
|
+
metric_name,
|
|
32
|
+
similarity_matrix,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
LEGACY_AST_METRIC = "ast_node_type_multiset_jaccard"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _dimension_groups(
|
|
39
|
+
metrics: Iterable[str], metric_dimensions: Mapping[str, object]
|
|
40
|
+
) -> dict[str, list[str]]:
|
|
41
|
+
"""Group aggregate metrics, preferring TED over the legacy AST baseline."""
|
|
42
|
+
metric_list = list(metrics)
|
|
43
|
+
groups: dict[str, list[str]] = {}
|
|
44
|
+
for metric in metric_list:
|
|
45
|
+
dimension = str(metric_dimensions.get(metric, metric))
|
|
46
|
+
if (
|
|
47
|
+
dimension == "structural_ast"
|
|
48
|
+
and "ast_tree_edit_similarity" in metric_list
|
|
49
|
+
and metric == LEGACY_AST_METRIC
|
|
50
|
+
):
|
|
51
|
+
continue
|
|
52
|
+
groups.setdefault(dimension, []).append(metric)
|
|
53
|
+
return groups
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _aggregation_metadata(
|
|
57
|
+
metrics: Iterable[str], metric_dimensions: Mapping[str, object]
|
|
58
|
+
) -> dict[str, object]:
|
|
59
|
+
return {
|
|
60
|
+
"method": "equal_weighted_available_dimensions_v1",
|
|
61
|
+
"dimension_metrics": _dimension_groups(metrics, metric_dimensions),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _natural_key(value: str) -> list[object]:
|
|
66
|
+
return [int(part) if part.isdecimal() else part.lower() for part in re.split(r"(\d+)", value)]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _metric_filename(metric: str) -> str:
|
|
70
|
+
"""Keep external metric names inside the export directory."""
|
|
71
|
+
if re.fullmatch(r"[A-Za-z0-9_-]{1,80}", metric):
|
|
72
|
+
return metric
|
|
73
|
+
label = re.sub(r"[^A-Za-z0-9_-]+", "_", metric).strip("_")[:60] or "metric"
|
|
74
|
+
digest = hashlib.sha256(metric.encode("utf-8")).hexdigest()[:12]
|
|
75
|
+
return f"{label}_{digest}"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _pair_values(matrix: pd.DataFrame) -> list[float]:
|
|
79
|
+
return [float(matrix.iloc[row, column]) for row in range(len(matrix)) for column in range(row + 1, len(matrix))]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _summary(matrix: pd.DataFrame) -> dict[str, float | int | None]:
|
|
83
|
+
values = pd.Series(_pair_values(matrix), dtype=float)
|
|
84
|
+
if values.empty:
|
|
85
|
+
return {"n_files": len(matrix), "n_pairs": 0, "mean_similarity": None, "median_similarity": None,
|
|
86
|
+
"std_similarity": None, "min_similarity": None, "max_similarity": None,
|
|
87
|
+
"q1_similarity": None, "q3_similarity": None, "mean_variability": None}
|
|
88
|
+
mean = float(values.mean())
|
|
89
|
+
return {"n_files": len(matrix), "n_pairs": len(values), "mean_similarity": mean,
|
|
90
|
+
"median_similarity": float(values.median()), "std_similarity": float(values.std(ddof=1)) if len(values) > 1 else 0.0,
|
|
91
|
+
"min_similarity": float(values.min()), "max_similarity": float(values.max()),
|
|
92
|
+
"q1_similarity": float(values.quantile(.25)), "q3_similarity": float(values.quantile(.75)),
|
|
93
|
+
"mean_variability": 1 - mean}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass(frozen=True)
|
|
97
|
+
class CodeDataset:
|
|
98
|
+
"""An immutable collection of UTF-8 documents identified by unique basenames."""
|
|
99
|
+
files: Mapping[str, Path]
|
|
100
|
+
|
|
101
|
+
def __post_init__(self) -> None:
|
|
102
|
+
if not isinstance(self.files, Mapping) or not self.files or any(not isinstance(name, str) or not name for name in self.files):
|
|
103
|
+
raise AnalysisError("O dataset deve conter arquivos com nomes textuais únicos e não vazios.")
|
|
104
|
+
try:
|
|
105
|
+
paths = {name: Path(path) for name, path in self.files.items()}
|
|
106
|
+
except (TypeError, ValueError) as exc:
|
|
107
|
+
raise AnalysisError("Os caminhos dos arquivos devem ser strings ou Paths válidos.") from exc
|
|
108
|
+
object.__setattr__(self, "files", MappingProxyType(paths))
|
|
109
|
+
|
|
110
|
+
@classmethod
|
|
111
|
+
def from_directory(cls, directory: str | Path, extensions: Iterable[str] | None = None) -> "CodeDataset":
|
|
112
|
+
try:
|
|
113
|
+
folder = Path(directory)
|
|
114
|
+
except (TypeError, ValueError) as exc:
|
|
115
|
+
raise AnalysisError("O diretório deve ser um caminho válido.") from exc
|
|
116
|
+
if not folder.exists():
|
|
117
|
+
raise AnalysisError(f"Diretório não encontrado: {folder}.")
|
|
118
|
+
if not folder.is_dir():
|
|
119
|
+
raise AnalysisError(f"O caminho informado não é um diretório: {folder}.")
|
|
120
|
+
if isinstance(extensions, str):
|
|
121
|
+
extensions = [extensions]
|
|
122
|
+
try:
|
|
123
|
+
allowed: set[str] | None = None
|
|
124
|
+
if extensions is not None:
|
|
125
|
+
allowed = set()
|
|
126
|
+
for extension in extensions:
|
|
127
|
+
if not isinstance(extension, str) or not extension.strip():
|
|
128
|
+
raise AnalysisError("Extensões devem ser strings não vazias.")
|
|
129
|
+
extension = extension.strip().lower()
|
|
130
|
+
allowed.add(extension if extension.startswith(".") else f".{extension}")
|
|
131
|
+
except TypeError as exc:
|
|
132
|
+
raise AnalysisError("extensions deve ser uma string ou coleção de strings.") from exc
|
|
133
|
+
try:
|
|
134
|
+
files = sorted(
|
|
135
|
+
(item for item in folder.iterdir() if item.is_file() and (allowed is None or item.suffix.lower() in allowed)),
|
|
136
|
+
key=lambda item: (_natural_key(item.name), item.name),
|
|
137
|
+
)
|
|
138
|
+
except OSError as exc:
|
|
139
|
+
raise AnalysisError(f"Não foi possível listar o diretório {folder}: {exc}.") from exc
|
|
140
|
+
if not files:
|
|
141
|
+
raise AnalysisError(f"Nenhum arquivo textual encontrado em {folder}.")
|
|
142
|
+
return cls({item.name: item for item in files})
|
|
143
|
+
|
|
144
|
+
@classmethod
|
|
145
|
+
def from_files(cls, files: Sequence[str | Path]) -> "CodeDataset":
|
|
146
|
+
if isinstance(files, (str, Path)):
|
|
147
|
+
files = [files]
|
|
148
|
+
try:
|
|
149
|
+
paths = [Path(item) for item in files]
|
|
150
|
+
except (TypeError, ValueError) as exc:
|
|
151
|
+
raise AnalysisError("Informe uma coleção de caminhos de arquivos válidos.") from exc
|
|
152
|
+
if not paths:
|
|
153
|
+
raise AnalysisError("Informe ao menos um arquivo textual.")
|
|
154
|
+
names = [item.name for item in paths]
|
|
155
|
+
if len(set(names)) != len(names):
|
|
156
|
+
raise AnalysisError("Os nomes-base dos arquivos devem ser únicos.")
|
|
157
|
+
invalid = [str(item) for item in paths if not item.is_file()]
|
|
158
|
+
if invalid:
|
|
159
|
+
raise AnalysisError(f"Arquivo(s) não encontrado(s) ou inválido(s): {', '.join(invalid)}.")
|
|
160
|
+
return cls(dict(zip(names, paths, strict=True)))
|
|
161
|
+
|
|
162
|
+
def analyze(self, metrics: str | Sequence[str] = "all", *, max_ted_cells: int | None = 2_000_000) -> "AnalysisResult":
|
|
163
|
+
if max_ted_cells is not None and (not isinstance(max_ted_cells, int) or isinstance(max_ted_cells, bool) or max_ted_cells < 1):
|
|
164
|
+
raise AnalysisError("max_ted_cells deve ser um inteiro positivo ou None.")
|
|
165
|
+
try:
|
|
166
|
+
raw_sources = {name: path.read_bytes() for name, path in self.files.items()}
|
|
167
|
+
sources = {name: raw.decode("utf-8-sig").replace("\r\n", "\n").replace("\r", "\n") for name, raw in raw_sources.items()}
|
|
168
|
+
except (OSError, UnicodeError) as exc:
|
|
169
|
+
raise AnalysisError(f"Não foi possível ler os arquivos textuais como UTF-8: {exc}.") from exc
|
|
170
|
+
|
|
171
|
+
def applicable(metric: str) -> bool:
|
|
172
|
+
if metric == "ast_tree_edit_similarity":
|
|
173
|
+
return all(uses_python_code(sources[name], name) for name in self.files)
|
|
174
|
+
return True
|
|
175
|
+
|
|
176
|
+
if isinstance(metrics, str):
|
|
177
|
+
selected = [metric for metric in METRICS if applicable(metric)] if metrics == "all" else [metrics]
|
|
178
|
+
else:
|
|
179
|
+
try:
|
|
180
|
+
selected = list(metrics)
|
|
181
|
+
except TypeError as exc:
|
|
182
|
+
raise AnalysisError("metrics deve ser uma string ou coleção de nomes de métricas.") from exc
|
|
183
|
+
if any(not isinstance(metric, str) or not metric for metric in selected):
|
|
184
|
+
raise AnalysisError("Os nomes das métricas devem ser strings não vazias.")
|
|
185
|
+
selected = list(dict.fromkeys(selected))
|
|
186
|
+
if not selected:
|
|
187
|
+
raise AnalysisError("Informe ao menos uma métrica.")
|
|
188
|
+
unknown = set(selected) - set(METRICS)
|
|
189
|
+
if unknown:
|
|
190
|
+
raise AnalysisError(f"Métrica(s) desconhecida(s): {', '.join(sorted(unknown))}.")
|
|
191
|
+
incompatible = [metric for metric in selected if not applicable(metric)]
|
|
192
|
+
if incompatible:
|
|
193
|
+
raise AnalysisError(
|
|
194
|
+
"Métrica(s) incompatível(is) com as entradas analisadas: "
|
|
195
|
+
f"{', '.join(incompatible)}."
|
|
196
|
+
)
|
|
197
|
+
matrices = {}
|
|
198
|
+
for metric in selected:
|
|
199
|
+
matrix = CALCULATORS[metric](sources, max_cells=max_ted_cells) if metric == "ast_tree_edit_similarity" else CALCULATORS[metric](sources)
|
|
200
|
+
# The exact TED normalization has a proven bound and must not rely
|
|
201
|
+
# on clipping to hide an invalid value. Other normalized metrics
|
|
202
|
+
# retain a floating-point guard.
|
|
203
|
+
if metric == "ast_tree_edit_similarity":
|
|
204
|
+
if matrix.isna().any().any() or ((matrix < 0) | (matrix > 1)).any().any():
|
|
205
|
+
raise AnalysisError("A similaridade TED calculada ficou fora do intervalo [0, 1].")
|
|
206
|
+
else:
|
|
207
|
+
matrix = matrix.clip(lower=0.0, upper=1.0)
|
|
208
|
+
# Set the diagonal explicitly to avoid floating-point drift.
|
|
209
|
+
for name in matrix.index:
|
|
210
|
+
matrix.loc[name, name] = 1.0
|
|
211
|
+
matrices[metric] = matrix
|
|
212
|
+
normalizations = {
|
|
213
|
+
metric: (
|
|
214
|
+
"python_normalized_ast_tree_v3" if metric == "ast_tree_edit_similarity"
|
|
215
|
+
else TEXT_NORMALIZATION if metric in {"cosine", "jaccard"}
|
|
216
|
+
else CODE_TOKENIZATION
|
|
217
|
+
)
|
|
218
|
+
for metric in selected
|
|
219
|
+
}
|
|
220
|
+
metric_dimensions = {metric: METRIC_DIMENSIONS[metric] for metric in selected}
|
|
221
|
+
return AnalysisResult(matrices=matrices, files=list(self.files), metadata={
|
|
222
|
+
"metrics": list(selected),
|
|
223
|
+
"normalization": next(iter(set(normalizations.values()))) if len(set(normalizations.values())) == 1 else "per_metric",
|
|
224
|
+
"metric_normalizations": normalizations,
|
|
225
|
+
"metric_ids": {metric: METRIC_IDS[metric] for metric in selected},
|
|
226
|
+
"metric_dimensions": metric_dimensions,
|
|
227
|
+
"overall_aggregation": _aggregation_metadata(selected, metric_dimensions),
|
|
228
|
+
"runtime_versions": {
|
|
229
|
+
"python": platform.python_version(),
|
|
230
|
+
"numpy": np.__version__,
|
|
231
|
+
"pandas": pd.__version__,
|
|
232
|
+
"pygments": pygments.__version__,
|
|
233
|
+
"rapidfuzz": rapidfuzz.__version__,
|
|
234
|
+
"scikit_learn": sklearn.__version__,
|
|
235
|
+
},
|
|
236
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
237
|
+
"input_sha256": {name: hashlib.sha256(raw).hexdigest() for name, raw in raw_sources.items()},
|
|
238
|
+
"max_ted_cells": max_ted_cells,
|
|
239
|
+
})
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@dataclass
|
|
243
|
+
class AnalysisResult:
|
|
244
|
+
matrices: Mapping[str, pd.DataFrame]
|
|
245
|
+
files: list[str]
|
|
246
|
+
metadata: dict[str, object] = field(default_factory=dict)
|
|
247
|
+
|
|
248
|
+
def __post_init__(self) -> None:
|
|
249
|
+
if not isinstance(self.files, (list, tuple)) or not self.files or any(not isinstance(name, str) or not name for name in self.files) or len(set(self.files)) != len(self.files):
|
|
250
|
+
raise AnalysisError("O resultado requer uma coleção não vazia de nomes de arquivos únicos.")
|
|
251
|
+
if not isinstance(self.matrices, Mapping) or not self.matrices:
|
|
252
|
+
raise AnalysisError("O resultado requer ao menos uma matriz de similaridade.")
|
|
253
|
+
if not isinstance(self.metadata, Mapping):
|
|
254
|
+
raise AnalysisError("metadata deve ser um mapeamento.")
|
|
255
|
+
self.files = list(self.files)
|
|
256
|
+
self.metadata = deepcopy(dict(self.metadata))
|
|
257
|
+
for name in self.matrices:
|
|
258
|
+
metric_name(name)
|
|
259
|
+
dimensions(list(self.matrices), self.metadata)
|
|
260
|
+
self.matrices = {name: similarity_matrix(matrix, self.files) for name, matrix in self.matrices.items()}
|
|
261
|
+
|
|
262
|
+
@property
|
|
263
|
+
def metrics(self) -> tuple[str, ...]:
|
|
264
|
+
"""Public metric names, in the order in which they were calculated."""
|
|
265
|
+
return tuple(self.matrices)
|
|
266
|
+
|
|
267
|
+
@property
|
|
268
|
+
def statistics(self) -> pd.DataFrame:
|
|
269
|
+
"""One row per metric, calculated from the unique pairs ``i < j``."""
|
|
270
|
+
result = pd.DataFrame.from_dict(
|
|
271
|
+
{metric: _summary(matrix) for metric, matrix in self.matrices.items()},
|
|
272
|
+
orient="index",
|
|
273
|
+
)
|
|
274
|
+
result.index.name = "metric"
|
|
275
|
+
return result
|
|
276
|
+
|
|
277
|
+
@staticmethod
|
|
278
|
+
def _metric_ranking(matrix: pd.DataFrame) -> pd.DataFrame:
|
|
279
|
+
if len(matrix) == 1:
|
|
280
|
+
values = pd.Series([1.0], index=matrix.index)
|
|
281
|
+
else:
|
|
282
|
+
diagonal = pd.Series(matrix.to_numpy().diagonal(), index=matrix.index)
|
|
283
|
+
values = (matrix.sum(axis=1) - diagonal) / (len(matrix) - 1)
|
|
284
|
+
result = pd.DataFrame({"file": values.index, "score": values.values})
|
|
285
|
+
result = result.sort_values(["score", "file"], ascending=[False, True], ignore_index=True)
|
|
286
|
+
result.insert(0, "rank", range(1, len(result) + 1))
|
|
287
|
+
return result
|
|
288
|
+
|
|
289
|
+
@property
|
|
290
|
+
def rankings_by_metric(self) -> dict[str, pd.DataFrame]:
|
|
291
|
+
"""Rank files by mean similarity to every other file, per metric."""
|
|
292
|
+
return {metric: self._metric_ranking(matrix) for metric, matrix in self.matrices.items()}
|
|
293
|
+
|
|
294
|
+
@staticmethod
|
|
295
|
+
def _row_dispersion(matrix: pd.DataFrame) -> pd.Series:
|
|
296
|
+
"""Sample standard deviation of each file's similarity to every other file.
|
|
297
|
+
|
|
298
|
+
A low value means the file is about equally similar to everyone; a
|
|
299
|
+
high value means its row average is driven disproportionately by a
|
|
300
|
+
few close (or a few distant) files rather than by broad centrality.
|
|
301
|
+
Returns 0.0 wherever fewer than two off-diagonal values exist, the
|
|
302
|
+
same convention used by :func:`_summary` for ``std_similarity``.
|
|
303
|
+
"""
|
|
304
|
+
if len(matrix) <= 2:
|
|
305
|
+
return pd.Series(0.0, index=matrix.index)
|
|
306
|
+
masked = matrix.to_numpy(dtype=float).copy()
|
|
307
|
+
np.fill_diagonal(masked, np.nan)
|
|
308
|
+
return pd.DataFrame(masked, index=matrix.index, columns=matrix.columns).std(axis=1, ddof=1)
|
|
309
|
+
|
|
310
|
+
@property
|
|
311
|
+
def representativeness(self) -> pd.DataFrame:
|
|
312
|
+
"""Per-file metric, dimension and overall representativeness scores.
|
|
313
|
+
|
|
314
|
+
Every metric score is the mean similarity from that file to all other
|
|
315
|
+
files. Metrics are first averaged within each methodological dimension;
|
|
316
|
+
``overall_score`` is then the unweighted mean of the available
|
|
317
|
+
dimension scores. When TED is present, the legacy AST node-frequency
|
|
318
|
+
metric remains visible but does not enter the structural score.
|
|
319
|
+
|
|
320
|
+
``overall_dispersion`` is the sample standard deviation of the same
|
|
321
|
+
file's row in the equally-weighted dimension-average matrix underlying
|
|
322
|
+
``overall_score`` (see :meth:`_row_dispersion`). It diagnoses whether a
|
|
323
|
+
high ``overall_score`` reflects broad similarity to the whole set or is
|
|
324
|
+
concentrated on a few close neighbours, which a mean alone cannot show.
|
|
325
|
+
"""
|
|
326
|
+
result = pd.DataFrame({"file": self.files})
|
|
327
|
+
for metric, ranking in self.rankings_by_metric.items():
|
|
328
|
+
scores = ranking.set_index("file")["score"]
|
|
329
|
+
result[metric] = result["file"].map(scores)
|
|
330
|
+
|
|
331
|
+
dimensions = self.metadata.get("metric_dimensions", {})
|
|
332
|
+
groups = _dimension_groups(
|
|
333
|
+
self.metrics, dimensions if isinstance(dimensions, Mapping) else {}
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
dimension_columns: list[str] = []
|
|
337
|
+
dimension_matrices: list[pd.DataFrame] = []
|
|
338
|
+
for dimension, metrics in groups.items():
|
|
339
|
+
column = DIMENSION_SCORE_COLUMNS.get(dimension, f"{dimension}_score")
|
|
340
|
+
result[column] = result[metrics].mean(axis=1)
|
|
341
|
+
dimension_columns.append(column)
|
|
342
|
+
combined = sum(self.matrices[metric] for metric in metrics) / len(metrics)
|
|
343
|
+
dimension_matrices.append(combined)
|
|
344
|
+
result["overall_score"] = result[dimension_columns].mean(axis=1)
|
|
345
|
+
if dimension_matrices:
|
|
346
|
+
overall_matrix = sum(dimension_matrices) / len(dimension_matrices)
|
|
347
|
+
dispersion = self._row_dispersion(overall_matrix)
|
|
348
|
+
result["overall_dispersion"] = result["file"].map(dispersion)
|
|
349
|
+
return result
|
|
350
|
+
|
|
351
|
+
@property
|
|
352
|
+
def ranking(self) -> pd.DataFrame:
|
|
353
|
+
"""Overall deterministic ranking, most representative first."""
|
|
354
|
+
result = self.representativeness[["file", "overall_score"]].rename(
|
|
355
|
+
columns={"overall_score": "score"}
|
|
356
|
+
)
|
|
357
|
+
result = result.sort_values(["score", "file"], ascending=[False, True], ignore_index=True)
|
|
358
|
+
result.insert(0, "rank", range(1, len(result) + 1))
|
|
359
|
+
return result
|
|
360
|
+
|
|
361
|
+
@property
|
|
362
|
+
def most_representative(self) -> dict[str, str | float | int]:
|
|
363
|
+
row = self.ranking.iloc[0]
|
|
364
|
+
return {"file": str(row["file"]), "score": float(row["score"]), "rank": int(row["rank"])}
|
|
365
|
+
|
|
366
|
+
@property
|
|
367
|
+
def most_distinct(self) -> dict[str, str | float | int]:
|
|
368
|
+
row = self.ranking.iloc[-1]
|
|
369
|
+
return {"file": str(row["file"]), "score": float(row["score"]), "rank": int(row["rank"])}
|
|
370
|
+
|
|
371
|
+
def medoid(self, metric: str) -> dict[str, str | float | int]:
|
|
372
|
+
self._matrix(metric)
|
|
373
|
+
row = self.rankings_by_metric[metric].iloc[0]
|
|
374
|
+
return {"file": str(row["file"]), "mean_similarity": float(row["score"]),
|
|
375
|
+
"mean_variability": 1 - float(row["score"]), "rank": int(row["rank"]), "metric": metric}
|
|
376
|
+
|
|
377
|
+
def composite(self, weights: Mapping[str, float] | str = "equal") -> pd.DataFrame:
|
|
378
|
+
"""Return an explicitly exploratory weighted composite similarity matrix."""
|
|
379
|
+
if not self.matrices:
|
|
380
|
+
raise AnalysisError("Não há matrizes para compor.")
|
|
381
|
+
if weights == "equal":
|
|
382
|
+
normalized = {metric: 1 / len(self.matrices) for metric in self.matrices}
|
|
383
|
+
elif isinstance(weights, Mapping):
|
|
384
|
+
unknown = set(weights) - set(self.matrices)
|
|
385
|
+
missing = set(self.matrices) - set(weights)
|
|
386
|
+
try:
|
|
387
|
+
numeric_weights = {metric: float(value) for metric, value in weights.items()}
|
|
388
|
+
total = sum(numeric_weights.values())
|
|
389
|
+
except (TypeError, ValueError, OverflowError) as exc:
|
|
390
|
+
raise AnalysisError("Pesos devem ser números finitos.") from exc
|
|
391
|
+
if unknown or missing or not math.isfinite(total) or total <= 0 or any(not math.isfinite(value) or value < 0 for value in numeric_weights.values()):
|
|
392
|
+
raise AnalysisError("Pesos devem ser finitos, não negativos, somar valor positivo e incluir exatamente as métricas analisadas.")
|
|
393
|
+
normalized = {metric: value / total for metric, value in numeric_weights.items()}
|
|
394
|
+
else:
|
|
395
|
+
raise AnalysisError("Use weights='equal' ou um dicionário de pesos.")
|
|
396
|
+
composite = sum((self.matrices[metric] * weight for metric, weight in normalized.items()))
|
|
397
|
+
composite.attrs["kind"] = "exploratory_composite_similarity"
|
|
398
|
+
composite.attrs["weights"] = normalized
|
|
399
|
+
return composite
|
|
400
|
+
|
|
401
|
+
def with_metric(self, metric: str, matrix: pd.DataFrame) -> "AnalysisResult":
|
|
402
|
+
"""Return a copy augmented by a normalized external metric matrix.
|
|
403
|
+
|
|
404
|
+
Language adapters use a readable public metric name, e.g.
|
|
405
|
+
``ast_node_type_multiset_jaccard``; their interchange metadata records
|
|
406
|
+
the versioned internal identifier.
|
|
407
|
+
"""
|
|
408
|
+
metric_name(metric)
|
|
409
|
+
if metric in self.matrices:
|
|
410
|
+
raise AnalysisError("A métrica externa deve ter um nome novo e não vazio.")
|
|
411
|
+
aligned = similarity_matrix(matrix, self.files)
|
|
412
|
+
metadata = dict(self.metadata)
|
|
413
|
+
metadata["metrics"] = [*self.metrics, metric]
|
|
414
|
+
for metadata_field, attribute in (
|
|
415
|
+
("metric_ids", "metric_id"),
|
|
416
|
+
("metric_normalizations", "normalization"),
|
|
417
|
+
("metric_dimensions", "dimension"),
|
|
418
|
+
):
|
|
419
|
+
existing = metadata.get(metadata_field, {})
|
|
420
|
+
if not isinstance(existing, Mapping):
|
|
421
|
+
raise AnalysisError(f"metadata.{metadata_field} deve ser um mapeamento.")
|
|
422
|
+
values = dict(existing)
|
|
423
|
+
value = matrix.attrs.get(attribute)
|
|
424
|
+
if value is not None:
|
|
425
|
+
if not isinstance(value, str) or not value.strip():
|
|
426
|
+
raise AnalysisError(f"O atributo '{attribute}' da métrica externa deve ser texto não vazio.")
|
|
427
|
+
values[metric] = value
|
|
428
|
+
metadata[metadata_field] = values
|
|
429
|
+
adapter_metadata = matrix.attrs.get("adapter_metadata")
|
|
430
|
+
if isinstance(adapter_metadata, dict):
|
|
431
|
+
existing_adapters = metadata.get("external_adapters", {})
|
|
432
|
+
if not isinstance(existing_adapters, Mapping):
|
|
433
|
+
raise AnalysisError("metadata.external_adapters deve ser um mapeamento.")
|
|
434
|
+
adapters = dict(existing_adapters)
|
|
435
|
+
adapters[metric] = adapter_metadata
|
|
436
|
+
metadata["external_adapters"] = adapters
|
|
437
|
+
dimensions = metadata.get("metric_dimensions", {})
|
|
438
|
+
metadata["overall_aggregation"] = _aggregation_metadata(
|
|
439
|
+
[*self.metrics, metric], dimensions if isinstance(dimensions, Mapping) else {}
|
|
440
|
+
)
|
|
441
|
+
normalizations = metadata.get("metric_normalizations", {})
|
|
442
|
+
if not isinstance(normalizations, Mapping):
|
|
443
|
+
raise AnalysisError("metadata.metric_normalizations deve ser um mapeamento.")
|
|
444
|
+
unique = set(normalizations.values())
|
|
445
|
+
metadata["normalization"] = next(iter(unique)) if len(unique) == 1 and len(normalizations) == len(self.metrics) + 1 else "per_metric"
|
|
446
|
+
return AnalysisResult(matrices={**self.matrices, metric: aligned}, files=list(self.files), metadata=metadata)
|
|
447
|
+
|
|
448
|
+
def export(self, directory: str | Path, formats: str | Sequence[str] = ("json", "csv")) -> None:
|
|
449
|
+
try:
|
|
450
|
+
requested = {formats} if isinstance(formats, str) else set(formats)
|
|
451
|
+
except TypeError as exc:
|
|
452
|
+
raise AnalysisError("formats deve ser uma string ou coleção de formatos.") from exc
|
|
453
|
+
if not requested:
|
|
454
|
+
raise AnalysisError("Informe ao menos um formato de exportação.")
|
|
455
|
+
if not requested <= {"json", "csv"}:
|
|
456
|
+
raise AnalysisError("Formatos suportados nesta versão: json, csv.")
|
|
457
|
+
output = Path(directory)
|
|
458
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
459
|
+
if "csv" in requested:
|
|
460
|
+
for metric, matrix in self.matrices.items():
|
|
461
|
+
safe_to_csv(matrix, output / f"{_metric_filename(metric)}_similarity_matrix.csv", index_label="file")
|
|
462
|
+
safe_to_csv(self.statistics, output / "statistics.csv", index_label="metric")
|
|
463
|
+
safe_to_csv(self.representativeness, output / "representativeness.csv", index=False)
|
|
464
|
+
safe_to_csv(self.ranking, output / "ranking.csv", index=False)
|
|
465
|
+
for metric, ranking in self.rankings_by_metric.items():
|
|
466
|
+
safe_to_csv(ranking, output / f"{_metric_filename(metric)}_ranking.csv", index=False)
|
|
467
|
+
if "json" in requested:
|
|
468
|
+
payload = {
|
|
469
|
+
"metadata": self.metadata,
|
|
470
|
+
"files": self.files,
|
|
471
|
+
"metrics": list(self.metrics),
|
|
472
|
+
"statistics": frame_json(self.statistics, "index"),
|
|
473
|
+
"representativeness": frame_json(self.representativeness, "records"),
|
|
474
|
+
"rankings_by_metric": {
|
|
475
|
+
metric: frame_json(ranking, "records")
|
|
476
|
+
for metric, ranking in self.rankings_by_metric.items()
|
|
477
|
+
},
|
|
478
|
+
"ranking": frame_json(self.ranking, "records"),
|
|
479
|
+
"most_representative": self.most_representative,
|
|
480
|
+
"most_distinct": self.most_distinct,
|
|
481
|
+
"matrices": {metric: matrix.to_dict() for metric, matrix in self.matrices.items()},
|
|
482
|
+
}
|
|
483
|
+
write_json(payload, output / "analysis.json")
|
|
484
|
+
|
|
485
|
+
def to_excel(self, path: str | Path) -> None:
|
|
486
|
+
"""Export the already calculated analysis to one Excel workbook."""
|
|
487
|
+
target = Path(path)
|
|
488
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
489
|
+
summary = pd.DataFrame({
|
|
490
|
+
"field": ["number_of_files", "metrics_used", "most_representative_file",
|
|
491
|
+
"most_distinct_file", "representative_overall_score"],
|
|
492
|
+
"value": [len(self.files), ", ".join(self.metrics), self.most_representative["file"],
|
|
493
|
+
self.most_distinct["file"], self.most_representative["score"]],
|
|
494
|
+
})
|
|
495
|
+
metadata = pd.DataFrame({
|
|
496
|
+
"field": list(self.metadata),
|
|
497
|
+
"value": [
|
|
498
|
+
json.dumps(value, ensure_ascii=False, sort_keys=True) if isinstance(value, (dict, list)) else value
|
|
499
|
+
for value in self.metadata.values()
|
|
500
|
+
],
|
|
501
|
+
})
|
|
502
|
+
used_sheets = {"summary", "metadata", "statistics", "representativeness", "ranking"}
|
|
503
|
+
try:
|
|
504
|
+
with excel_writer(target) as writer:
|
|
505
|
+
safe_to_excel(summary, writer, sheet_name="Summary", index=False)
|
|
506
|
+
safe_to_excel(metadata, writer, sheet_name="Metadata", index=False)
|
|
507
|
+
safe_to_excel(self.representativeness, writer, sheet_name="Representativeness", index=False)
|
|
508
|
+
safe_to_excel(self.ranking, writer, sheet_name="Ranking", index=False)
|
|
509
|
+
safe_to_excel(self.statistics, writer, sheet_name="Statistics")
|
|
510
|
+
for metric, matrix in self.matrices.items():
|
|
511
|
+
sheet = self._excel_sheet_name(metric, used_sheets)
|
|
512
|
+
safe_to_excel(matrix, writer, sheet_name=sheet, index_label="file")
|
|
513
|
+
except (ImportError, ModuleNotFoundError) as exc:
|
|
514
|
+
raise AnalysisError("A exportação Excel requer a dependência openpyxl.") from exc
|
|
515
|
+
|
|
516
|
+
@staticmethod
|
|
517
|
+
def _excel_sheet_name(metric: str, used: set[str]) -> str:
|
|
518
|
+
base = re.sub(r"[\\/*?:\[\]]", "_", metric)[:31] or "Metric"
|
|
519
|
+
candidate = base
|
|
520
|
+
suffix = 2
|
|
521
|
+
while candidate.lower() in used:
|
|
522
|
+
marker = f"_{suffix}"
|
|
523
|
+
candidate = f"{base[:31 - len(marker)]}{marker}"
|
|
524
|
+
suffix += 1
|
|
525
|
+
used.add(candidate.lower())
|
|
526
|
+
return candidate
|
|
527
|
+
|
|
528
|
+
def _matrix(self, metric: str) -> pd.DataFrame:
|
|
529
|
+
try:
|
|
530
|
+
return self.matrices[metric]
|
|
531
|
+
except KeyError as exc:
|
|
532
|
+
raise AnalysisError(f"A métrica '{metric}' não foi calculada.") from exc
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def analyze(
|
|
536
|
+
files_or_directory: str | Path | Sequence[str | Path],
|
|
537
|
+
metrics: str | Sequence[str] = "all",
|
|
538
|
+
*,
|
|
539
|
+
extensions: Iterable[str] | str | None = None,
|
|
540
|
+
max_ted_cells: int | None = 2_000_000,
|
|
541
|
+
) -> AnalysisResult:
|
|
542
|
+
"""Analyze code files from a directory or an explicit sequence of paths."""
|
|
543
|
+
dataset = CodeDataset.from_directory(files_or_directory, extensions=extensions) if isinstance(files_or_directory, (str, Path)) and Path(files_or_directory).is_dir() else CodeDataset.from_files([files_or_directory] if isinstance(files_or_directory, (str, Path)) else files_or_directory)
|
|
544
|
+
return dataset.analyze(metrics=metrics, max_ted_cells=max_ted_cells)
|