codevariability 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,23 @@
1
+ """Public API for multidimensional source-code similarity analysis."""
2
+
3
+ from .analysis import AnalysisResult, CodeDataset, analyze
4
+ from .exceptions import AnalysisError
5
+ from .group_comparison import (
6
+ GroupComparisonResult,
7
+ MultiGroupComparisonResult,
8
+ compare_groups,
9
+ )
10
+ from .interop import load_matrix_json
11
+
12
+ __all__ = [
13
+ "AnalysisError",
14
+ "AnalysisResult",
15
+ "CodeDataset",
16
+ "GroupComparisonResult",
17
+ "MultiGroupComparisonResult",
18
+ "analyze",
19
+ "compare_groups",
20
+ "load_matrix_json",
21
+ ]
22
+
23
+ __version__ = "0.2.0"
@@ -0,0 +1,544 @@
1
+ """Dataset loading, pairwise analysis, summaries, and result export."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import math
8
+ import platform
9
+ import re
10
+ from copy import deepcopy
11
+ from dataclasses import dataclass, field
12
+ from datetime import datetime, timezone
13
+ from pathlib import Path
14
+ from types import MappingProxyType
15
+ from typing import Iterable, Mapping, Sequence
16
+
17
+ import numpy as np
18
+ import pandas as pd
19
+ import pygments
20
+ import rapidfuzz
21
+ import sklearn
22
+
23
+ from .exceptions import AnalysisError
24
+ from .metrics import CALCULATORS, METRIC_DIMENSIONS, METRIC_IDS, METRICS
25
+ from .normalization import CODE_TOKENIZATION, TEXT_NORMALIZATION, uses_python_code
26
+ from .output import excel_writer, frame_json, write_json
27
+ from .spreadsheet import safe_to_csv, safe_to_excel
28
+ from .validation import (
29
+ DIMENSION_SCORE_COLUMNS,
30
+ dimensions,
31
+ metric_name,
32
+ similarity_matrix,
33
+ )
34
+
35
+ LEGACY_AST_METRIC = "ast_node_type_multiset_jaccard"
36
+
37
+
38
+ def _dimension_groups(
39
+ metrics: Iterable[str], metric_dimensions: Mapping[str, object]
40
+ ) -> dict[str, list[str]]:
41
+ """Group aggregate metrics, preferring TED over the legacy AST baseline."""
42
+ metric_list = list(metrics)
43
+ groups: dict[str, list[str]] = {}
44
+ for metric in metric_list:
45
+ dimension = str(metric_dimensions.get(metric, metric))
46
+ if (
47
+ dimension == "structural_ast"
48
+ and "ast_tree_edit_similarity" in metric_list
49
+ and metric == LEGACY_AST_METRIC
50
+ ):
51
+ continue
52
+ groups.setdefault(dimension, []).append(metric)
53
+ return groups
54
+
55
+
56
+ def _aggregation_metadata(
57
+ metrics: Iterable[str], metric_dimensions: Mapping[str, object]
58
+ ) -> dict[str, object]:
59
+ return {
60
+ "method": "equal_weighted_available_dimensions_v1",
61
+ "dimension_metrics": _dimension_groups(metrics, metric_dimensions),
62
+ }
63
+
64
+
65
+ def _natural_key(value: str) -> list[object]:
66
+ return [int(part) if part.isdecimal() else part.lower() for part in re.split(r"(\d+)", value)]
67
+
68
+
69
+ def _metric_filename(metric: str) -> str:
70
+ """Keep external metric names inside the export directory."""
71
+ if re.fullmatch(r"[A-Za-z0-9_-]{1,80}", metric):
72
+ return metric
73
+ label = re.sub(r"[^A-Za-z0-9_-]+", "_", metric).strip("_")[:60] or "metric"
74
+ digest = hashlib.sha256(metric.encode("utf-8")).hexdigest()[:12]
75
+ return f"{label}_{digest}"
76
+
77
+
78
+ def _pair_values(matrix: pd.DataFrame) -> list[float]:
79
+ return [float(matrix.iloc[row, column]) for row in range(len(matrix)) for column in range(row + 1, len(matrix))]
80
+
81
+
82
+ def _summary(matrix: pd.DataFrame) -> dict[str, float | int | None]:
83
+ values = pd.Series(_pair_values(matrix), dtype=float)
84
+ if values.empty:
85
+ return {"n_files": len(matrix), "n_pairs": 0, "mean_similarity": None, "median_similarity": None,
86
+ "std_similarity": None, "min_similarity": None, "max_similarity": None,
87
+ "q1_similarity": None, "q3_similarity": None, "mean_variability": None}
88
+ mean = float(values.mean())
89
+ return {"n_files": len(matrix), "n_pairs": len(values), "mean_similarity": mean,
90
+ "median_similarity": float(values.median()), "std_similarity": float(values.std(ddof=1)) if len(values) > 1 else 0.0,
91
+ "min_similarity": float(values.min()), "max_similarity": float(values.max()),
92
+ "q1_similarity": float(values.quantile(.25)), "q3_similarity": float(values.quantile(.75)),
93
+ "mean_variability": 1 - mean}
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class CodeDataset:
98
+ """An immutable collection of UTF-8 documents identified by unique basenames."""
99
+ files: Mapping[str, Path]
100
+
101
+ def __post_init__(self) -> None:
102
+ if not isinstance(self.files, Mapping) or not self.files or any(not isinstance(name, str) or not name for name in self.files):
103
+ raise AnalysisError("O dataset deve conter arquivos com nomes textuais únicos e não vazios.")
104
+ try:
105
+ paths = {name: Path(path) for name, path in self.files.items()}
106
+ except (TypeError, ValueError) as exc:
107
+ raise AnalysisError("Os caminhos dos arquivos devem ser strings ou Paths válidos.") from exc
108
+ object.__setattr__(self, "files", MappingProxyType(paths))
109
+
110
+ @classmethod
111
+ def from_directory(cls, directory: str | Path, extensions: Iterable[str] | None = None) -> "CodeDataset":
112
+ try:
113
+ folder = Path(directory)
114
+ except (TypeError, ValueError) as exc:
115
+ raise AnalysisError("O diretório deve ser um caminho válido.") from exc
116
+ if not folder.exists():
117
+ raise AnalysisError(f"Diretório não encontrado: {folder}.")
118
+ if not folder.is_dir():
119
+ raise AnalysisError(f"O caminho informado não é um diretório: {folder}.")
120
+ if isinstance(extensions, str):
121
+ extensions = [extensions]
122
+ try:
123
+ allowed: set[str] | None = None
124
+ if extensions is not None:
125
+ allowed = set()
126
+ for extension in extensions:
127
+ if not isinstance(extension, str) or not extension.strip():
128
+ raise AnalysisError("Extensões devem ser strings não vazias.")
129
+ extension = extension.strip().lower()
130
+ allowed.add(extension if extension.startswith(".") else f".{extension}")
131
+ except TypeError as exc:
132
+ raise AnalysisError("extensions deve ser uma string ou coleção de strings.") from exc
133
+ try:
134
+ files = sorted(
135
+ (item for item in folder.iterdir() if item.is_file() and (allowed is None or item.suffix.lower() in allowed)),
136
+ key=lambda item: (_natural_key(item.name), item.name),
137
+ )
138
+ except OSError as exc:
139
+ raise AnalysisError(f"Não foi possível listar o diretório {folder}: {exc}.") from exc
140
+ if not files:
141
+ raise AnalysisError(f"Nenhum arquivo textual encontrado em {folder}.")
142
+ return cls({item.name: item for item in files})
143
+
144
+ @classmethod
145
+ def from_files(cls, files: Sequence[str | Path]) -> "CodeDataset":
146
+ if isinstance(files, (str, Path)):
147
+ files = [files]
148
+ try:
149
+ paths = [Path(item) for item in files]
150
+ except (TypeError, ValueError) as exc:
151
+ raise AnalysisError("Informe uma coleção de caminhos de arquivos válidos.") from exc
152
+ if not paths:
153
+ raise AnalysisError("Informe ao menos um arquivo textual.")
154
+ names = [item.name for item in paths]
155
+ if len(set(names)) != len(names):
156
+ raise AnalysisError("Os nomes-base dos arquivos devem ser únicos.")
157
+ invalid = [str(item) for item in paths if not item.is_file()]
158
+ if invalid:
159
+ raise AnalysisError(f"Arquivo(s) não encontrado(s) ou inválido(s): {', '.join(invalid)}.")
160
+ return cls(dict(zip(names, paths, strict=True)))
161
+
162
+ def analyze(self, metrics: str | Sequence[str] = "all", *, max_ted_cells: int | None = 2_000_000) -> "AnalysisResult":
163
+ if max_ted_cells is not None and (not isinstance(max_ted_cells, int) or isinstance(max_ted_cells, bool) or max_ted_cells < 1):
164
+ raise AnalysisError("max_ted_cells deve ser um inteiro positivo ou None.")
165
+ try:
166
+ raw_sources = {name: path.read_bytes() for name, path in self.files.items()}
167
+ sources = {name: raw.decode("utf-8-sig").replace("\r\n", "\n").replace("\r", "\n") for name, raw in raw_sources.items()}
168
+ except (OSError, UnicodeError) as exc:
169
+ raise AnalysisError(f"Não foi possível ler os arquivos textuais como UTF-8: {exc}.") from exc
170
+
171
+ def applicable(metric: str) -> bool:
172
+ if metric == "ast_tree_edit_similarity":
173
+ return all(uses_python_code(sources[name], name) for name in self.files)
174
+ return True
175
+
176
+ if isinstance(metrics, str):
177
+ selected = [metric for metric in METRICS if applicable(metric)] if metrics == "all" else [metrics]
178
+ else:
179
+ try:
180
+ selected = list(metrics)
181
+ except TypeError as exc:
182
+ raise AnalysisError("metrics deve ser uma string ou coleção de nomes de métricas.") from exc
183
+ if any(not isinstance(metric, str) or not metric for metric in selected):
184
+ raise AnalysisError("Os nomes das métricas devem ser strings não vazias.")
185
+ selected = list(dict.fromkeys(selected))
186
+ if not selected:
187
+ raise AnalysisError("Informe ao menos uma métrica.")
188
+ unknown = set(selected) - set(METRICS)
189
+ if unknown:
190
+ raise AnalysisError(f"Métrica(s) desconhecida(s): {', '.join(sorted(unknown))}.")
191
+ incompatible = [metric for metric in selected if not applicable(metric)]
192
+ if incompatible:
193
+ raise AnalysisError(
194
+ "Métrica(s) incompatível(is) com as entradas analisadas: "
195
+ f"{', '.join(incompatible)}."
196
+ )
197
+ matrices = {}
198
+ for metric in selected:
199
+ matrix = CALCULATORS[metric](sources, max_cells=max_ted_cells) if metric == "ast_tree_edit_similarity" else CALCULATORS[metric](sources)
200
+ # The exact TED normalization has a proven bound and must not rely
201
+ # on clipping to hide an invalid value. Other normalized metrics
202
+ # retain a floating-point guard.
203
+ if metric == "ast_tree_edit_similarity":
204
+ if matrix.isna().any().any() or ((matrix < 0) | (matrix > 1)).any().any():
205
+ raise AnalysisError("A similaridade TED calculada ficou fora do intervalo [0, 1].")
206
+ else:
207
+ matrix = matrix.clip(lower=0.0, upper=1.0)
208
+ # Set the diagonal explicitly to avoid floating-point drift.
209
+ for name in matrix.index:
210
+ matrix.loc[name, name] = 1.0
211
+ matrices[metric] = matrix
212
+ normalizations = {
213
+ metric: (
214
+ "python_normalized_ast_tree_v3" if metric == "ast_tree_edit_similarity"
215
+ else TEXT_NORMALIZATION if metric in {"cosine", "jaccard"}
216
+ else CODE_TOKENIZATION
217
+ )
218
+ for metric in selected
219
+ }
220
+ metric_dimensions = {metric: METRIC_DIMENSIONS[metric] for metric in selected}
221
+ return AnalysisResult(matrices=matrices, files=list(self.files), metadata={
222
+ "metrics": list(selected),
223
+ "normalization": next(iter(set(normalizations.values()))) if len(set(normalizations.values())) == 1 else "per_metric",
224
+ "metric_normalizations": normalizations,
225
+ "metric_ids": {metric: METRIC_IDS[metric] for metric in selected},
226
+ "metric_dimensions": metric_dimensions,
227
+ "overall_aggregation": _aggregation_metadata(selected, metric_dimensions),
228
+ "runtime_versions": {
229
+ "python": platform.python_version(),
230
+ "numpy": np.__version__,
231
+ "pandas": pd.__version__,
232
+ "pygments": pygments.__version__,
233
+ "rapidfuzz": rapidfuzz.__version__,
234
+ "scikit_learn": sklearn.__version__,
235
+ },
236
+ "created_at": datetime.now(timezone.utc).isoformat(),
237
+ "input_sha256": {name: hashlib.sha256(raw).hexdigest() for name, raw in raw_sources.items()},
238
+ "max_ted_cells": max_ted_cells,
239
+ })
240
+
241
+
242
+ @dataclass
243
+ class AnalysisResult:
244
+ matrices: Mapping[str, pd.DataFrame]
245
+ files: list[str]
246
+ metadata: dict[str, object] = field(default_factory=dict)
247
+
248
+ def __post_init__(self) -> None:
249
+ if not isinstance(self.files, (list, tuple)) or not self.files or any(not isinstance(name, str) or not name for name in self.files) or len(set(self.files)) != len(self.files):
250
+ raise AnalysisError("O resultado requer uma coleção não vazia de nomes de arquivos únicos.")
251
+ if not isinstance(self.matrices, Mapping) or not self.matrices:
252
+ raise AnalysisError("O resultado requer ao menos uma matriz de similaridade.")
253
+ if not isinstance(self.metadata, Mapping):
254
+ raise AnalysisError("metadata deve ser um mapeamento.")
255
+ self.files = list(self.files)
256
+ self.metadata = deepcopy(dict(self.metadata))
257
+ for name in self.matrices:
258
+ metric_name(name)
259
+ dimensions(list(self.matrices), self.metadata)
260
+ self.matrices = {name: similarity_matrix(matrix, self.files) for name, matrix in self.matrices.items()}
261
+
262
+ @property
263
+ def metrics(self) -> tuple[str, ...]:
264
+ """Public metric names, in the order in which they were calculated."""
265
+ return tuple(self.matrices)
266
+
267
+ @property
268
+ def statistics(self) -> pd.DataFrame:
269
+ """One row per metric, calculated from the unique pairs ``i < j``."""
270
+ result = pd.DataFrame.from_dict(
271
+ {metric: _summary(matrix) for metric, matrix in self.matrices.items()},
272
+ orient="index",
273
+ )
274
+ result.index.name = "metric"
275
+ return result
276
+
277
+ @staticmethod
278
+ def _metric_ranking(matrix: pd.DataFrame) -> pd.DataFrame:
279
+ if len(matrix) == 1:
280
+ values = pd.Series([1.0], index=matrix.index)
281
+ else:
282
+ diagonal = pd.Series(matrix.to_numpy().diagonal(), index=matrix.index)
283
+ values = (matrix.sum(axis=1) - diagonal) / (len(matrix) - 1)
284
+ result = pd.DataFrame({"file": values.index, "score": values.values})
285
+ result = result.sort_values(["score", "file"], ascending=[False, True], ignore_index=True)
286
+ result.insert(0, "rank", range(1, len(result) + 1))
287
+ return result
288
+
289
+ @property
290
+ def rankings_by_metric(self) -> dict[str, pd.DataFrame]:
291
+ """Rank files by mean similarity to every other file, per metric."""
292
+ return {metric: self._metric_ranking(matrix) for metric, matrix in self.matrices.items()}
293
+
294
+ @staticmethod
295
+ def _row_dispersion(matrix: pd.DataFrame) -> pd.Series:
296
+ """Sample standard deviation of each file's similarity to every other file.
297
+
298
+ A low value means the file is about equally similar to everyone; a
299
+ high value means its row average is driven disproportionately by a
300
+ few close (or a few distant) files rather than by broad centrality.
301
+ Returns 0.0 wherever fewer than two off-diagonal values exist, the
302
+ same convention used by :func:`_summary` for ``std_similarity``.
303
+ """
304
+ if len(matrix) <= 2:
305
+ return pd.Series(0.0, index=matrix.index)
306
+ masked = matrix.to_numpy(dtype=float).copy()
307
+ np.fill_diagonal(masked, np.nan)
308
+ return pd.DataFrame(masked, index=matrix.index, columns=matrix.columns).std(axis=1, ddof=1)
309
+
310
+ @property
311
+ def representativeness(self) -> pd.DataFrame:
312
+ """Per-file metric, dimension and overall representativeness scores.
313
+
314
+ Every metric score is the mean similarity from that file to all other
315
+ files. Metrics are first averaged within each methodological dimension;
316
+ ``overall_score`` is then the unweighted mean of the available
317
+ dimension scores. When TED is present, the legacy AST node-frequency
318
+ metric remains visible but does not enter the structural score.
319
+
320
+ ``overall_dispersion`` is the sample standard deviation of the same
321
+ file's row in the equally-weighted dimension-average matrix underlying
322
+ ``overall_score`` (see :meth:`_row_dispersion`). It diagnoses whether a
323
+ high ``overall_score`` reflects broad similarity to the whole set or is
324
+ concentrated on a few close neighbours, which a mean alone cannot show.
325
+ """
326
+ result = pd.DataFrame({"file": self.files})
327
+ for metric, ranking in self.rankings_by_metric.items():
328
+ scores = ranking.set_index("file")["score"]
329
+ result[metric] = result["file"].map(scores)
330
+
331
+ dimensions = self.metadata.get("metric_dimensions", {})
332
+ groups = _dimension_groups(
333
+ self.metrics, dimensions if isinstance(dimensions, Mapping) else {}
334
+ )
335
+
336
+ dimension_columns: list[str] = []
337
+ dimension_matrices: list[pd.DataFrame] = []
338
+ for dimension, metrics in groups.items():
339
+ column = DIMENSION_SCORE_COLUMNS.get(dimension, f"{dimension}_score")
340
+ result[column] = result[metrics].mean(axis=1)
341
+ dimension_columns.append(column)
342
+ combined = sum(self.matrices[metric] for metric in metrics) / len(metrics)
343
+ dimension_matrices.append(combined)
344
+ result["overall_score"] = result[dimension_columns].mean(axis=1)
345
+ if dimension_matrices:
346
+ overall_matrix = sum(dimension_matrices) / len(dimension_matrices)
347
+ dispersion = self._row_dispersion(overall_matrix)
348
+ result["overall_dispersion"] = result["file"].map(dispersion)
349
+ return result
350
+
351
+ @property
352
+ def ranking(self) -> pd.DataFrame:
353
+ """Overall deterministic ranking, most representative first."""
354
+ result = self.representativeness[["file", "overall_score"]].rename(
355
+ columns={"overall_score": "score"}
356
+ )
357
+ result = result.sort_values(["score", "file"], ascending=[False, True], ignore_index=True)
358
+ result.insert(0, "rank", range(1, len(result) + 1))
359
+ return result
360
+
361
+ @property
362
+ def most_representative(self) -> dict[str, str | float | int]:
363
+ row = self.ranking.iloc[0]
364
+ return {"file": str(row["file"]), "score": float(row["score"]), "rank": int(row["rank"])}
365
+
366
+ @property
367
+ def most_distinct(self) -> dict[str, str | float | int]:
368
+ row = self.ranking.iloc[-1]
369
+ return {"file": str(row["file"]), "score": float(row["score"]), "rank": int(row["rank"])}
370
+
371
+ def medoid(self, metric: str) -> dict[str, str | float | int]:
372
+ self._matrix(metric)
373
+ row = self.rankings_by_metric[metric].iloc[0]
374
+ return {"file": str(row["file"]), "mean_similarity": float(row["score"]),
375
+ "mean_variability": 1 - float(row["score"]), "rank": int(row["rank"]), "metric": metric}
376
+
377
+ def composite(self, weights: Mapping[str, float] | str = "equal") -> pd.DataFrame:
378
+ """Return an explicitly exploratory weighted composite similarity matrix."""
379
+ if not self.matrices:
380
+ raise AnalysisError("Não há matrizes para compor.")
381
+ if weights == "equal":
382
+ normalized = {metric: 1 / len(self.matrices) for metric in self.matrices}
383
+ elif isinstance(weights, Mapping):
384
+ unknown = set(weights) - set(self.matrices)
385
+ missing = set(self.matrices) - set(weights)
386
+ try:
387
+ numeric_weights = {metric: float(value) for metric, value in weights.items()}
388
+ total = sum(numeric_weights.values())
389
+ except (TypeError, ValueError, OverflowError) as exc:
390
+ raise AnalysisError("Pesos devem ser números finitos.") from exc
391
+ if unknown or missing or not math.isfinite(total) or total <= 0 or any(not math.isfinite(value) or value < 0 for value in numeric_weights.values()):
392
+ raise AnalysisError("Pesos devem ser finitos, não negativos, somar valor positivo e incluir exatamente as métricas analisadas.")
393
+ normalized = {metric: value / total for metric, value in numeric_weights.items()}
394
+ else:
395
+ raise AnalysisError("Use weights='equal' ou um dicionário de pesos.")
396
+ composite = sum((self.matrices[metric] * weight for metric, weight in normalized.items()))
397
+ composite.attrs["kind"] = "exploratory_composite_similarity"
398
+ composite.attrs["weights"] = normalized
399
+ return composite
400
+
401
+ def with_metric(self, metric: str, matrix: pd.DataFrame) -> "AnalysisResult":
402
+ """Return a copy augmented by a normalized external metric matrix.
403
+
404
+ Language adapters use a readable public metric name, e.g.
405
+ ``ast_node_type_multiset_jaccard``; their interchange metadata records
406
+ the versioned internal identifier.
407
+ """
408
+ metric_name(metric)
409
+ if metric in self.matrices:
410
+ raise AnalysisError("A métrica externa deve ter um nome novo e não vazio.")
411
+ aligned = similarity_matrix(matrix, self.files)
412
+ metadata = dict(self.metadata)
413
+ metadata["metrics"] = [*self.metrics, metric]
414
+ for metadata_field, attribute in (
415
+ ("metric_ids", "metric_id"),
416
+ ("metric_normalizations", "normalization"),
417
+ ("metric_dimensions", "dimension"),
418
+ ):
419
+ existing = metadata.get(metadata_field, {})
420
+ if not isinstance(existing, Mapping):
421
+ raise AnalysisError(f"metadata.{metadata_field} deve ser um mapeamento.")
422
+ values = dict(existing)
423
+ value = matrix.attrs.get(attribute)
424
+ if value is not None:
425
+ if not isinstance(value, str) or not value.strip():
426
+ raise AnalysisError(f"O atributo '{attribute}' da métrica externa deve ser texto não vazio.")
427
+ values[metric] = value
428
+ metadata[metadata_field] = values
429
+ adapter_metadata = matrix.attrs.get("adapter_metadata")
430
+ if isinstance(adapter_metadata, dict):
431
+ existing_adapters = metadata.get("external_adapters", {})
432
+ if not isinstance(existing_adapters, Mapping):
433
+ raise AnalysisError("metadata.external_adapters deve ser um mapeamento.")
434
+ adapters = dict(existing_adapters)
435
+ adapters[metric] = adapter_metadata
436
+ metadata["external_adapters"] = adapters
437
+ dimensions = metadata.get("metric_dimensions", {})
438
+ metadata["overall_aggregation"] = _aggregation_metadata(
439
+ [*self.metrics, metric], dimensions if isinstance(dimensions, Mapping) else {}
440
+ )
441
+ normalizations = metadata.get("metric_normalizations", {})
442
+ if not isinstance(normalizations, Mapping):
443
+ raise AnalysisError("metadata.metric_normalizations deve ser um mapeamento.")
444
+ unique = set(normalizations.values())
445
+ metadata["normalization"] = next(iter(unique)) if len(unique) == 1 and len(normalizations) == len(self.metrics) + 1 else "per_metric"
446
+ return AnalysisResult(matrices={**self.matrices, metric: aligned}, files=list(self.files), metadata=metadata)
447
+
448
+ def export(self, directory: str | Path, formats: str | Sequence[str] = ("json", "csv")) -> None:
449
+ try:
450
+ requested = {formats} if isinstance(formats, str) else set(formats)
451
+ except TypeError as exc:
452
+ raise AnalysisError("formats deve ser uma string ou coleção de formatos.") from exc
453
+ if not requested:
454
+ raise AnalysisError("Informe ao menos um formato de exportação.")
455
+ if not requested <= {"json", "csv"}:
456
+ raise AnalysisError("Formatos suportados nesta versão: json, csv.")
457
+ output = Path(directory)
458
+ output.mkdir(parents=True, exist_ok=True)
459
+ if "csv" in requested:
460
+ for metric, matrix in self.matrices.items():
461
+ safe_to_csv(matrix, output / f"{_metric_filename(metric)}_similarity_matrix.csv", index_label="file")
462
+ safe_to_csv(self.statistics, output / "statistics.csv", index_label="metric")
463
+ safe_to_csv(self.representativeness, output / "representativeness.csv", index=False)
464
+ safe_to_csv(self.ranking, output / "ranking.csv", index=False)
465
+ for metric, ranking in self.rankings_by_metric.items():
466
+ safe_to_csv(ranking, output / f"{_metric_filename(metric)}_ranking.csv", index=False)
467
+ if "json" in requested:
468
+ payload = {
469
+ "metadata": self.metadata,
470
+ "files": self.files,
471
+ "metrics": list(self.metrics),
472
+ "statistics": frame_json(self.statistics, "index"),
473
+ "representativeness": frame_json(self.representativeness, "records"),
474
+ "rankings_by_metric": {
475
+ metric: frame_json(ranking, "records")
476
+ for metric, ranking in self.rankings_by_metric.items()
477
+ },
478
+ "ranking": frame_json(self.ranking, "records"),
479
+ "most_representative": self.most_representative,
480
+ "most_distinct": self.most_distinct,
481
+ "matrices": {metric: matrix.to_dict() for metric, matrix in self.matrices.items()},
482
+ }
483
+ write_json(payload, output / "analysis.json")
484
+
485
+ def to_excel(self, path: str | Path) -> None:
486
+ """Export the already calculated analysis to one Excel workbook."""
487
+ target = Path(path)
488
+ target.parent.mkdir(parents=True, exist_ok=True)
489
+ summary = pd.DataFrame({
490
+ "field": ["number_of_files", "metrics_used", "most_representative_file",
491
+ "most_distinct_file", "representative_overall_score"],
492
+ "value": [len(self.files), ", ".join(self.metrics), self.most_representative["file"],
493
+ self.most_distinct["file"], self.most_representative["score"]],
494
+ })
495
+ metadata = pd.DataFrame({
496
+ "field": list(self.metadata),
497
+ "value": [
498
+ json.dumps(value, ensure_ascii=False, sort_keys=True) if isinstance(value, (dict, list)) else value
499
+ for value in self.metadata.values()
500
+ ],
501
+ })
502
+ used_sheets = {"summary", "metadata", "statistics", "representativeness", "ranking"}
503
+ try:
504
+ with excel_writer(target) as writer:
505
+ safe_to_excel(summary, writer, sheet_name="Summary", index=False)
506
+ safe_to_excel(metadata, writer, sheet_name="Metadata", index=False)
507
+ safe_to_excel(self.representativeness, writer, sheet_name="Representativeness", index=False)
508
+ safe_to_excel(self.ranking, writer, sheet_name="Ranking", index=False)
509
+ safe_to_excel(self.statistics, writer, sheet_name="Statistics")
510
+ for metric, matrix in self.matrices.items():
511
+ sheet = self._excel_sheet_name(metric, used_sheets)
512
+ safe_to_excel(matrix, writer, sheet_name=sheet, index_label="file")
513
+ except (ImportError, ModuleNotFoundError) as exc:
514
+ raise AnalysisError("A exportação Excel requer a dependência openpyxl.") from exc
515
+
516
+ @staticmethod
517
+ def _excel_sheet_name(metric: str, used: set[str]) -> str:
518
+ base = re.sub(r"[\\/*?:\[\]]", "_", metric)[:31] or "Metric"
519
+ candidate = base
520
+ suffix = 2
521
+ while candidate.lower() in used:
522
+ marker = f"_{suffix}"
523
+ candidate = f"{base[:31 - len(marker)]}{marker}"
524
+ suffix += 1
525
+ used.add(candidate.lower())
526
+ return candidate
527
+
528
+ def _matrix(self, metric: str) -> pd.DataFrame:
529
+ try:
530
+ return self.matrices[metric]
531
+ except KeyError as exc:
532
+ raise AnalysisError(f"A métrica '{metric}' não foi calculada.") from exc
533
+
534
+
535
+ def analyze(
536
+ files_or_directory: str | Path | Sequence[str | Path],
537
+ metrics: str | Sequence[str] = "all",
538
+ *,
539
+ extensions: Iterable[str] | str | None = None,
540
+ max_ted_cells: int | None = 2_000_000,
541
+ ) -> AnalysisResult:
542
+ """Analyze code files from a directory or an explicit sequence of paths."""
543
+ dataset = CodeDataset.from_directory(files_or_directory, extensions=extensions) if isinstance(files_or_directory, (str, Path)) and Path(files_or_directory).is_dir() else CodeDataset.from_files([files_or_directory] if isinstance(files_or_directory, (str, Path)) else files_or_directory)
544
+ return dataset.analyze(metrics=metrics, max_ted_cells=max_ted_cells)