codevariability 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codevariability/__init__.py +23 -0
- codevariability/analysis.py +544 -0
- codevariability/ast_tree_edit.py +350 -0
- codevariability/cli.py +75 -0
- codevariability/exceptions.py +2 -0
- codevariability/group_comparison.py +979 -0
- codevariability/interop.py +94 -0
- codevariability/metrics.py +89 -0
- codevariability/normalization.py +257 -0
- codevariability/output.py +76 -0
- codevariability/spreadsheet.py +68 -0
- codevariability/validation.py +115 -0
- codevariability-0.2.0.dist-info/METADATA +171 -0
- codevariability-0.2.0.dist-info/RECORD +18 -0
- codevariability-0.2.0.dist-info/WHEEL +5 -0
- codevariability-0.2.0.dist-info/entry_points.txt +2 -0
- codevariability-0.2.0.dist-info/licenses/LICENSE +21 -0
- codevariability-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,979 @@
|
|
|
1
|
+
"""Permutation-based comparison of two or more independent code collections."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
import os
|
|
7
|
+
import random
|
|
8
|
+
import shutil
|
|
9
|
+
import signal
|
|
10
|
+
import subprocess
|
|
11
|
+
import sys
|
|
12
|
+
import tempfile
|
|
13
|
+
import threading
|
|
14
|
+
from collections import deque
|
|
15
|
+
from contextlib import suppress
|
|
16
|
+
from copy import deepcopy
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from itertools import combinations
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Callable, Mapping, Sequence
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
from .analysis import AnalysisResult, CodeDataset, _dimension_groups
|
|
25
|
+
from .exceptions import AnalysisError
|
|
26
|
+
from .interop import load_matrix_json
|
|
27
|
+
from .output import excel_writer, frame_json, write_json
|
|
28
|
+
from .spreadsheet import safe_to_csv, safe_to_excel
|
|
29
|
+
|
|
30
|
+
AST_TREE_EDIT_METRIC = "ast_tree_edit_similarity"
|
|
31
|
+
JAVASCRIPT_EXTENSIONS = {".js", ".jsx", ".ts", ".tsx", ".md", ".markdown"}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _dataset(value: str | Path | Sequence[str | Path]) -> CodeDataset:
|
|
35
|
+
if isinstance(value, (str, Path)) and Path(value).is_dir():
|
|
36
|
+
return CodeDataset.from_directory(value)
|
|
37
|
+
files = [value] if isinstance(value, (str, Path)) else value
|
|
38
|
+
return CodeDataset.from_files(files)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _mean(values: list[float]) -> float:
|
|
42
|
+
return sum(values) / len(values)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _within(matrix: pd.DataFrame, indices: Sequence[int]) -> tuple[float, list[float]]:
|
|
46
|
+
values = [float(matrix.iat[i, j]) for i, j in combinations(indices, 2)]
|
|
47
|
+
return _mean(values), values
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _between(matrix: pd.DataFrame, left: Sequence[int], right: Sequence[int]) -> float:
|
|
51
|
+
return _mean([float(matrix.iat[i, j]) for i in left for j in right])
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _group_statistics(
|
|
55
|
+
matrix: pd.DataFrame, left: Sequence[int], right: Sequence[int]
|
|
56
|
+
) -> tuple[float, float, float, float, float, list[float], list[float]]:
|
|
57
|
+
within_left, left_values = _within(matrix, left)
|
|
58
|
+
within_right, right_values = _within(matrix, right)
|
|
59
|
+
between = _between(matrix, left, right)
|
|
60
|
+
return (
|
|
61
|
+
within_left,
|
|
62
|
+
within_right,
|
|
63
|
+
between,
|
|
64
|
+
within_left - within_right,
|
|
65
|
+
(within_left + within_right) / 2 - between,
|
|
66
|
+
left_values,
|
|
67
|
+
right_values,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _array_group_statistics(
|
|
72
|
+
values: Any, left: Sequence[int], right: Sequence[int]
|
|
73
|
+
) -> tuple[float, float]:
|
|
74
|
+
"""Hot-path statistics over a NumPy-backed matrix without DataFrame indexing."""
|
|
75
|
+
within_left = _mean([float(values[i, j]) for i, j in combinations(left, 2)])
|
|
76
|
+
within_right = _mean([float(values[i, j]) for i, j in combinations(right, 2)])
|
|
77
|
+
between = _mean([float(values[i, j]) for i in left for j in right])
|
|
78
|
+
return within_left - within_right, (within_left + within_right) / 2 - between
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _cliffs_delta(left: Sequence[float], right: Sequence[float]) -> float:
|
|
82
|
+
comparisons = len(left) * len(right)
|
|
83
|
+
if comparisons == 0:
|
|
84
|
+
return 0.0
|
|
85
|
+
greater = sum(a > b for a in left for b in right)
|
|
86
|
+
lower = sum(a < b for a in left for b in right)
|
|
87
|
+
return (greater - lower) / comparisons
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _permutation_p_values(
|
|
91
|
+
matrix: pd.DataFrame,
|
|
92
|
+
size_left: int,
|
|
93
|
+
observed_homogeneity: float,
|
|
94
|
+
observed_separation: float,
|
|
95
|
+
permutations: int,
|
|
96
|
+
random_state: int,
|
|
97
|
+
) -> tuple[float, float]:
|
|
98
|
+
rng = random.Random(random_state)
|
|
99
|
+
values = matrix.to_numpy(dtype=float, copy=False)
|
|
100
|
+
population = list(range(len(matrix)))
|
|
101
|
+
extreme_homogeneity = 0
|
|
102
|
+
extreme_separation = 0
|
|
103
|
+
for _ in range(permutations):
|
|
104
|
+
left_set = set(rng.sample(population, size_left))
|
|
105
|
+
left = [index for index in population if index in left_set]
|
|
106
|
+
right = [index for index in population if index not in left_set]
|
|
107
|
+
homogeneity, separation = _array_group_statistics(values, left, right)
|
|
108
|
+
extreme_homogeneity += abs(homogeneity) >= abs(observed_homogeneity)
|
|
109
|
+
extreme_separation += abs(separation) >= abs(observed_separation)
|
|
110
|
+
denominator = permutations + 1
|
|
111
|
+
return (
|
|
112
|
+
(extreme_homogeneity + 1) / denominator,
|
|
113
|
+
(extreme_separation + 1) / denominator,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _holm(values: Mapping[str, float]) -> dict[str, float]:
|
|
118
|
+
ordered = sorted(values, key=lambda name: values[name])
|
|
119
|
+
adjusted: dict[str, float] = {}
|
|
120
|
+
previous = 0.0
|
|
121
|
+
total = len(ordered)
|
|
122
|
+
for position, name in enumerate(ordered):
|
|
123
|
+
candidate = min(1.0, (total - position) * values[name])
|
|
124
|
+
previous = max(previous, candidate)
|
|
125
|
+
adjusted[name] = previous
|
|
126
|
+
return adjusted
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _mean_matrix(matrices: Sequence[pd.DataFrame]) -> pd.DataFrame:
|
|
130
|
+
return sum(matrices[1:], matrices[0].copy()) / len(matrices)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _measure_matrices(analysis: AnalysisResult) -> dict[str, tuple[str, pd.DataFrame]]:
|
|
134
|
+
dimensions = analysis.metadata.get("metric_dimensions", {})
|
|
135
|
+
groups = _dimension_groups(
|
|
136
|
+
analysis.metrics, dimensions if isinstance(dimensions, Mapping) else {}
|
|
137
|
+
)
|
|
138
|
+
measures: dict[str, tuple[str, pd.DataFrame]] = {
|
|
139
|
+
metric: ("metric", matrix) for metric, matrix in analysis.matrices.items()
|
|
140
|
+
}
|
|
141
|
+
dimension_matrices: list[pd.DataFrame] = []
|
|
142
|
+
for dimension, dimension_metrics in groups.items():
|
|
143
|
+
matrix = _mean_matrix([analysis.matrices[name] for name in dimension_metrics])
|
|
144
|
+
measures[f"dimension:{dimension}"] = ("dimension", matrix)
|
|
145
|
+
dimension_matrices.append(matrix)
|
|
146
|
+
measures["overall"] = ("overall", _mean_matrix(dimension_matrices))
|
|
147
|
+
return measures
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _multi_group_statistics(
|
|
151
|
+
matrix: pd.DataFrame, groups: Mapping[str, Sequence[int]]
|
|
152
|
+
) -> tuple[dict[str, float], dict[tuple[str, str], float], float, float]:
|
|
153
|
+
within = {name: _within(matrix, indices)[0] for name, indices in groups.items()}
|
|
154
|
+
between = {
|
|
155
|
+
(left, right): _between(matrix, groups[left], groups[right])
|
|
156
|
+
for left, right in combinations(groups, 2)
|
|
157
|
+
}
|
|
158
|
+
mean_within = _mean(list(within.values()))
|
|
159
|
+
mean_between = _mean(list(between.values()))
|
|
160
|
+
homogeneity = sum((value - mean_within) ** 2 for value in within.values())
|
|
161
|
+
return within, between, homogeneity, mean_within - mean_between
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _global_permutation_p_values(
|
|
165
|
+
matrix: pd.DataFrame,
|
|
166
|
+
group_names: Sequence[str],
|
|
167
|
+
group_sizes: Sequence[int],
|
|
168
|
+
observed_homogeneity: float,
|
|
169
|
+
observed_separation: float,
|
|
170
|
+
permutations: int,
|
|
171
|
+
random_state: int,
|
|
172
|
+
) -> tuple[float, float]:
|
|
173
|
+
rng = random.Random(random_state)
|
|
174
|
+
values = matrix.to_numpy(dtype=float, copy=False)
|
|
175
|
+
population = list(range(len(matrix)))
|
|
176
|
+
extreme_homogeneity = 0
|
|
177
|
+
extreme_separation = 0
|
|
178
|
+
for _ in range(permutations):
|
|
179
|
+
shuffled = rng.sample(population, len(population))
|
|
180
|
+
offset = 0
|
|
181
|
+
groups: dict[str, list[int]] = {}
|
|
182
|
+
for name, size in zip(group_names, group_sizes, strict=True):
|
|
183
|
+
groups[name] = shuffled[offset:offset + size]
|
|
184
|
+
offset += size
|
|
185
|
+
within = {
|
|
186
|
+
name: _mean([
|
|
187
|
+
float(values[i, j]) for i, j in combinations(indices, 2)
|
|
188
|
+
])
|
|
189
|
+
for name, indices in groups.items()
|
|
190
|
+
}
|
|
191
|
+
between = [
|
|
192
|
+
_mean([
|
|
193
|
+
float(values[i, j])
|
|
194
|
+
for i in groups[left]
|
|
195
|
+
for j in groups[right]
|
|
196
|
+
])
|
|
197
|
+
for left, right in combinations(group_names, 2)
|
|
198
|
+
]
|
|
199
|
+
mean_within = _mean(list(within.values()))
|
|
200
|
+
homogeneity = sum((value - mean_within) ** 2 for value in within.values())
|
|
201
|
+
separation = mean_within - _mean(between)
|
|
202
|
+
extreme_homogeneity += homogeneity >= observed_homogeneity
|
|
203
|
+
extreme_separation += abs(separation) >= abs(observed_separation)
|
|
204
|
+
denominator = permutations + 1
|
|
205
|
+
return (
|
|
206
|
+
(extreme_homogeneity + 1) / denominator,
|
|
207
|
+
(extreme_separation + 1) / denominator,
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _javascript_adapter_command() -> list[str]:
|
|
212
|
+
installed = shutil.which("codevariability-js")
|
|
213
|
+
if installed:
|
|
214
|
+
return [installed]
|
|
215
|
+
raise AnalysisError(
|
|
216
|
+
"A TED JavaScript requer Node.js e o pacote codevariability-js instalado."
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _validate_options(permutations, random_state, include_ast, alpha, progress, ast_timeout, max_ted_cells):
|
|
221
|
+
if not isinstance(permutations, int) or isinstance(permutations, bool) or permutations < 1:
|
|
222
|
+
raise AnalysisError("permutations deve ser um inteiro positivo.")
|
|
223
|
+
if not isinstance(random_state, int) or isinstance(random_state, bool):
|
|
224
|
+
raise AnalysisError("random_state deve ser um inteiro.")
|
|
225
|
+
if not isinstance(include_ast, bool):
|
|
226
|
+
raise AnalysisError("include_ast deve ser True ou False.")
|
|
227
|
+
if not isinstance(alpha, (int, float)) or isinstance(alpha, bool) or not 0 < alpha < 1:
|
|
228
|
+
raise AnalysisError("alpha deve estar estritamente entre 0 e 1.")
|
|
229
|
+
if not isinstance(progress, bool) and not callable(progress):
|
|
230
|
+
raise AnalysisError("progress deve ser True, False ou uma função.")
|
|
231
|
+
try:
|
|
232
|
+
valid_timeout = isinstance(ast_timeout, (int, float)) and not isinstance(ast_timeout, bool) and math.isfinite(ast_timeout) and ast_timeout > 0
|
|
233
|
+
except OverflowError:
|
|
234
|
+
valid_timeout = False
|
|
235
|
+
if not valid_timeout:
|
|
236
|
+
raise AnalysisError("ast_timeout deve ser um número finito e positivo em segundos.")
|
|
237
|
+
if max_ted_cells is not None and (not isinstance(max_ted_cells, int) or isinstance(max_ted_cells, bool) or max_ted_cells < 1):
|
|
238
|
+
raise AnalysisError("max_ted_cells deve ser um inteiro positivo ou None.")
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _independent_files(datasets: Mapping[str, CodeDataset]) -> None:
|
|
242
|
+
seen: set[tuple[int, int]] = set()
|
|
243
|
+
try:
|
|
244
|
+
for dataset in datasets.values():
|
|
245
|
+
for path in dataset.files.values():
|
|
246
|
+
stat = path.stat()
|
|
247
|
+
identity = (stat.st_dev, stat.st_ino)
|
|
248
|
+
if identity in seen:
|
|
249
|
+
raise AnalysisError("Os grupos independentes não podem compartilhar arquivos físicos, inclusive hardlinks/symlinks.")
|
|
250
|
+
seen.add(identity)
|
|
251
|
+
except OSError as exc:
|
|
252
|
+
raise AnalysisError(f"Não foi possível verificar a identidade física dos arquivos: {exc}.") from exc
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _run_adapter(command: list[str], report, timeout: float) -> None:
|
|
256
|
+
"""Drain bounded stderr in a reader while the main thread enforces timeout."""
|
|
257
|
+
lines: deque[str] = deque(maxlen=64)
|
|
258
|
+
reader_errors: list[BaseException] = []
|
|
259
|
+
try:
|
|
260
|
+
process = subprocess.Popen(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE,
|
|
261
|
+
text=True, encoding="utf-8", errors="replace", start_new_session=os.name == "posix")
|
|
262
|
+
except OSError as exc:
|
|
263
|
+
raise AnalysisError(f"Não foi possível executar o adaptador AST JavaScript: {exc}.") from exc
|
|
264
|
+
assert process.stderr is not None
|
|
265
|
+
stderr = process.stderr
|
|
266
|
+
|
|
267
|
+
def read_stderr():
|
|
268
|
+
try:
|
|
269
|
+
while True:
|
|
270
|
+
# Bound a single line as well as the tail buffer.
|
|
271
|
+
line = stderr.readline(4096)
|
|
272
|
+
if not line:
|
|
273
|
+
break
|
|
274
|
+
message = line.strip()
|
|
275
|
+
if message:
|
|
276
|
+
lines.append(message)
|
|
277
|
+
if report is not None:
|
|
278
|
+
report(message)
|
|
279
|
+
except BaseException as exc:
|
|
280
|
+
reader_errors.append(exc)
|
|
281
|
+
|
|
282
|
+
reader = threading.Thread(target=read_stderr, daemon=True)
|
|
283
|
+
reader.start()
|
|
284
|
+
try:
|
|
285
|
+
try:
|
|
286
|
+
return_code = process.wait(timeout=timeout)
|
|
287
|
+
except subprocess.TimeoutExpired as exc:
|
|
288
|
+
raise AnalysisError(f"O adaptador AST JavaScript excedeu ast_timeout={timeout} segundos.") from exc
|
|
289
|
+
reader.join(timeout=1)
|
|
290
|
+
if reader_errors:
|
|
291
|
+
raise AnalysisError(f"Falha ao acompanhar o adaptador AST: {reader_errors[0]}.")
|
|
292
|
+
if return_code:
|
|
293
|
+
raise AnalysisError(f"Falha no adaptador AST JavaScript: {lines[-1] if lines else 'falha desconhecida'}")
|
|
294
|
+
finally:
|
|
295
|
+
# Also stop descendants holding the stderr pipe after the leader exits.
|
|
296
|
+
if os.name == "posix":
|
|
297
|
+
with suppress(ProcessLookupError):
|
|
298
|
+
os.killpg(process.pid, signal.SIGKILL)
|
|
299
|
+
elif process.poll() is None:
|
|
300
|
+
process.kill()
|
|
301
|
+
process.wait()
|
|
302
|
+
reader.join(timeout=1)
|
|
303
|
+
stderr.close()
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def _javascript_ast_matrix(
|
|
307
|
+
files: Mapping[str, Path],
|
|
308
|
+
report: Callable[[str], None] | None = None,
|
|
309
|
+
cache_dir: str | Path | None = None,
|
|
310
|
+
timeout: float = 120.0,
|
|
311
|
+
max_ted_cells: int | None = 2_000_000,
|
|
312
|
+
) -> tuple[str, pd.DataFrame]:
|
|
313
|
+
unsupported = [label for label, path in files.items() if path.suffix.lower() not in JAVASCRIPT_EXTENSIONS]
|
|
314
|
+
if unsupported:
|
|
315
|
+
raise AnalysisError(
|
|
316
|
+
"include_ast=True requer somente JavaScript, JSX, TypeScript ou Markdown; "
|
|
317
|
+
f"entradas incompatíveis: {', '.join(unsupported)}."
|
|
318
|
+
)
|
|
319
|
+
with tempfile.TemporaryDirectory(prefix="codevariability-js-ast-") as temporary:
|
|
320
|
+
folder = Path(temporary)
|
|
321
|
+
temporary_files: list[str] = []
|
|
322
|
+
temporary_names: list[str] = []
|
|
323
|
+
for index, path in enumerate(files.values()):
|
|
324
|
+
target = folder / f"{index:06d}{path.suffix.lower()}"
|
|
325
|
+
shutil.copyfile(path, target)
|
|
326
|
+
temporary_files.append(str(target))
|
|
327
|
+
temporary_names.append(target.name)
|
|
328
|
+
output = folder / "ast-tree-edit.json"
|
|
329
|
+
command = [
|
|
330
|
+
*_javascript_adapter_command(),
|
|
331
|
+
"ast",
|
|
332
|
+
*temporary_files,
|
|
333
|
+
"--metric",
|
|
334
|
+
AST_TREE_EDIT_METRIC,
|
|
335
|
+
"--output",
|
|
336
|
+
str(output),
|
|
337
|
+
]
|
|
338
|
+
if report is not None:
|
|
339
|
+
command.append("--progress")
|
|
340
|
+
if cache_dir is not None:
|
|
341
|
+
command.extend(["--cache-dir", str(Path(cache_dir).resolve())])
|
|
342
|
+
command.extend(["--max-cells", str(max_ted_cells) if max_ted_cells is not None else "none"])
|
|
343
|
+
_run_adapter(command, report, timeout)
|
|
344
|
+
metric, matrix = load_matrix_json(output)
|
|
345
|
+
if metric != AST_TREE_EDIT_METRIC or set(matrix.index) != set(temporary_names):
|
|
346
|
+
raise AnalysisError("O adaptador AST retornou uma métrica ou arquivos incompatíveis.")
|
|
347
|
+
labels = list(files)
|
|
348
|
+
rename = dict(zip(temporary_names, labels, strict=True))
|
|
349
|
+
attrs = deepcopy(matrix.attrs)
|
|
350
|
+
adapter_metadata = attrs.get("adapter_metadata", {})
|
|
351
|
+
hashes = adapter_metadata.get("input_sha256")
|
|
352
|
+
if isinstance(hashes, dict):
|
|
353
|
+
adapter_metadata["input_sha256"] = {rename.get(name, name): value for name, value in hashes.items()}
|
|
354
|
+
matrix = matrix.rename(index=rename, columns=rename).loc[labels, labels]
|
|
355
|
+
matrix.attrs.update(attrs)
|
|
356
|
+
return metric, matrix
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def _metadata_alpha(metadata: Mapping[str, object]) -> float:
|
|
360
|
+
value = metadata.get("alpha", 0.05)
|
|
361
|
+
if not isinstance(value, (int, float, str)):
|
|
362
|
+
raise AnalysisError("metadata.alpha deve ser um número.")
|
|
363
|
+
try:
|
|
364
|
+
return float(value)
|
|
365
|
+
except ValueError as exc:
|
|
366
|
+
raise AnalysisError("metadata.alpha deve ser um número.") from exc
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _within_columns(names: tuple[str, str], reserved: set[str]) -> tuple[str, str]:
|
|
370
|
+
"""Allocate presentation labels without mixing them with statistic keys."""
|
|
371
|
+
requested = [f"within_{name}" for name in names]
|
|
372
|
+
used = set(reserved)
|
|
373
|
+
columns = []
|
|
374
|
+
for candidate in requested:
|
|
375
|
+
column = candidate
|
|
376
|
+
while column in used or (column != candidate and column in requested):
|
|
377
|
+
column += "__group"
|
|
378
|
+
columns.append(column)
|
|
379
|
+
used.add(column)
|
|
380
|
+
return columns[0], columns[1]
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _check_adapter_snapshot(analysis: AnalysisResult, matrix: pd.DataFrame) -> None:
|
|
384
|
+
"""Require the external metric to describe the same bytes as base metrics."""
|
|
385
|
+
adapter = matrix.attrs.get("adapter_metadata", {})
|
|
386
|
+
if not isinstance(adapter, Mapping) or adapter.get("input_sha256") != analysis.metadata.get("input_sha256"):
|
|
387
|
+
raise AnalysisError("Os arquivos mudaram entre as etapas da análise ou o adaptador não confirmou os hashes de entrada.")
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
@dataclass
|
|
391
|
+
class GroupComparisonResult:
|
|
392
|
+
"""Results of an independent two-group comparison."""
|
|
393
|
+
|
|
394
|
+
summary: pd.DataFrame
|
|
395
|
+
group_names: tuple[str, str]
|
|
396
|
+
files: Mapping[str, list[str]]
|
|
397
|
+
metrics: tuple[str, ...]
|
|
398
|
+
metadata: dict[str, object]
|
|
399
|
+
|
|
400
|
+
@property
|
|
401
|
+
def within_group_columns(self) -> tuple[str, str]:
|
|
402
|
+
"""Columns for the two named groups, separate from statistic columns."""
|
|
403
|
+
columns = self.metadata.get("within_group_columns")
|
|
404
|
+
if isinstance(columns, (list, tuple)) and len(columns) == 2 and all(isinstance(item, str) for item in columns):
|
|
405
|
+
return columns[0], columns[1]
|
|
406
|
+
left, right = self.group_names
|
|
407
|
+
return f"within_{left}", f"within_{right}"
|
|
408
|
+
|
|
409
|
+
@property
|
|
410
|
+
def overview(self) -> pd.DataFrame:
|
|
411
|
+
"""Compact user-facing view; technical metric rows stay in ``summary``."""
|
|
412
|
+
left, right = self.group_names
|
|
413
|
+
left_column, right_column = self.within_group_columns
|
|
414
|
+
relevant = self.summary[self.summary["kind"].isin(["dimension", "overall"])]
|
|
415
|
+
labels = {
|
|
416
|
+
"dimension:textual": "Textual",
|
|
417
|
+
"dimension:syntactic_token_sequence": "Sintática",
|
|
418
|
+
"dimension:structural_ast": "Estrutural (AST TED)",
|
|
419
|
+
"overall": "Geral",
|
|
420
|
+
}
|
|
421
|
+
rows = []
|
|
422
|
+
alpha = _metadata_alpha(self.metadata)
|
|
423
|
+
for measure, row in relevant.iterrows():
|
|
424
|
+
difference = float(row["within_difference"])
|
|
425
|
+
rows.append({
|
|
426
|
+
"analysis": labels.get(measure, measure),
|
|
427
|
+
left_column: row[left_column],
|
|
428
|
+
right_column: row[right_column],
|
|
429
|
+
"between_groups": row["between_groups"],
|
|
430
|
+
"more_homogeneous": left if difference > 0 else right if difference < 0 else "tie",
|
|
431
|
+
"homogeneity_p_holm": row["p_value_homogeneity_holm"],
|
|
432
|
+
"separation": row["separation"],
|
|
433
|
+
"separation_p_holm": row["p_value_separation_holm"],
|
|
434
|
+
"evidence_of_separation": bool(
|
|
435
|
+
row["separation"] > 0 and row["p_value_separation_holm"] < alpha
|
|
436
|
+
),
|
|
437
|
+
})
|
|
438
|
+
return pd.DataFrame(rows).set_index("analysis")
|
|
439
|
+
|
|
440
|
+
@property
|
|
441
|
+
def interpretation(self) -> str:
|
|
442
|
+
"""Plain-language interpretation of the overall row."""
|
|
443
|
+
left, right = self.group_names
|
|
444
|
+
row = self.summary.loc["overall"]
|
|
445
|
+
alpha = _metadata_alpha(self.metadata)
|
|
446
|
+
difference = float(row["within_difference"])
|
|
447
|
+
if difference == 0:
|
|
448
|
+
homogeneity = f"{left} e {right} tiveram a mesma similaridade interna média"
|
|
449
|
+
else:
|
|
450
|
+
leader = left if difference > 0 else right
|
|
451
|
+
homogeneity = f"{leader} apresentou maior similaridade interna média"
|
|
452
|
+
homogeneity_evidence = (
|
|
453
|
+
"com evidência estatística"
|
|
454
|
+
if row["p_value_homogeneity_holm"] < alpha
|
|
455
|
+
else "sem evidência estatística suficiente"
|
|
456
|
+
)
|
|
457
|
+
separation_evidence = (
|
|
458
|
+
"Há evidência de separação entre os grupos"
|
|
459
|
+
if row["separation"] > 0 and row["p_value_separation_holm"] < alpha
|
|
460
|
+
else "Não há evidência suficiente de separação entre os grupos"
|
|
461
|
+
)
|
|
462
|
+
return (
|
|
463
|
+
f"{homogeneity}, {homogeneity_evidence} após correção de Holm "
|
|
464
|
+
f"(p={row['p_value_homogeneity_holm']:.4g}). {separation_evidence} "
|
|
465
|
+
f"(separação={row['separation']:.4f}; "
|
|
466
|
+
f"p={row['p_value_separation_holm']:.4g})."
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
def print_report(self) -> None:
|
|
470
|
+
"""Print only the compact overview and its cautious interpretation."""
|
|
471
|
+
print(self.overview.round(4))
|
|
472
|
+
print(f"\n{self.interpretation}")
|
|
473
|
+
|
|
474
|
+
@property
|
|
475
|
+
def by_metric(self) -> pd.DataFrame:
|
|
476
|
+
return self.summary[self.summary["kind"] == "metric"].copy()
|
|
477
|
+
|
|
478
|
+
@property
|
|
479
|
+
def by_dimension(self) -> pd.DataFrame:
|
|
480
|
+
return self.summary[self.summary["kind"].isin(["dimension", "overall"])].copy()
|
|
481
|
+
|
|
482
|
+
@property
|
|
483
|
+
def within_groups(self) -> pd.DataFrame:
|
|
484
|
+
return self.summary[[*self.within_group_columns, "within_difference"]].copy()
|
|
485
|
+
|
|
486
|
+
@property
|
|
487
|
+
def between_groups(self) -> pd.Series:
|
|
488
|
+
return self.summary["between_groups"].copy()
|
|
489
|
+
|
|
490
|
+
@property
|
|
491
|
+
def p_values(self) -> pd.DataFrame:
|
|
492
|
+
return self.summary[[
|
|
493
|
+
"p_value_homogeneity",
|
|
494
|
+
"p_value_homogeneity_holm",
|
|
495
|
+
"p_value_separation",
|
|
496
|
+
"p_value_separation_holm",
|
|
497
|
+
]].copy()
|
|
498
|
+
|
|
499
|
+
def export(self, directory: str | Path) -> None:
|
|
500
|
+
output = Path(directory)
|
|
501
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
502
|
+
safe_to_csv(self.summary, output / "group_comparison.csv", index_label="measure")
|
|
503
|
+
payload = {
|
|
504
|
+
"group_names": list(self.group_names),
|
|
505
|
+
"files": self.files,
|
|
506
|
+
"metrics": list(self.metrics),
|
|
507
|
+
"metadata": self.metadata,
|
|
508
|
+
"summary": frame_json(self.summary.reset_index(), "records"),
|
|
509
|
+
}
|
|
510
|
+
write_json(payload, output / "group_comparison.json")
|
|
511
|
+
|
|
512
|
+
def to_excel(self, path: str | Path) -> None:
|
|
513
|
+
target = Path(path)
|
|
514
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
515
|
+
with excel_writer(target) as writer:
|
|
516
|
+
safe_to_excel(self.overview, writer, sheet_name="Overview")
|
|
517
|
+
safe_to_excel(pd.DataFrame({"interpretation": [self.interpretation]}), writer,
|
|
518
|
+
sheet_name="Overview", startrow=len(self.overview) + 3, index=False
|
|
519
|
+
)
|
|
520
|
+
safe_to_excel(self.summary, writer, sheet_name="Technical details")
|
|
521
|
+
safe_to_excel(self.by_metric, writer, sheet_name="Metrics")
|
|
522
|
+
safe_to_excel(pd.DataFrame(
|
|
523
|
+
[(group, file) for group, files in self.files.items() for file in files],
|
|
524
|
+
columns=["group", "file"],
|
|
525
|
+
), writer, sheet_name="Files", index=False)
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
@dataclass
|
|
529
|
+
class MultiGroupComparisonResult:
|
|
530
|
+
"""Global and post-hoc results for two or more independent groups."""
|
|
531
|
+
|
|
532
|
+
global_test: pd.DataFrame
|
|
533
|
+
pairwise: pd.DataFrame
|
|
534
|
+
within_groups: pd.DataFrame
|
|
535
|
+
between_groups: pd.DataFrame
|
|
536
|
+
group_names: tuple[str, ...]
|
|
537
|
+
files: Mapping[str, list[str]]
|
|
538
|
+
metrics: tuple[str, ...]
|
|
539
|
+
metadata: dict[str, object]
|
|
540
|
+
|
|
541
|
+
@property
|
|
542
|
+
def overview(self) -> pd.DataFrame:
|
|
543
|
+
relevant = self.global_test[
|
|
544
|
+
self.global_test["kind"].isin(["dimension", "overall"])
|
|
545
|
+
]
|
|
546
|
+
labels = {
|
|
547
|
+
"dimension:textual": "Textual",
|
|
548
|
+
"dimension:syntactic_token_sequence": "Sintática",
|
|
549
|
+
"dimension:structural_ast": "Estrutural (AST TED)",
|
|
550
|
+
"overall": "Geral",
|
|
551
|
+
}
|
|
552
|
+
alpha = _metadata_alpha(self.metadata)
|
|
553
|
+
rows = []
|
|
554
|
+
for measure, row in relevant.iterrows():
|
|
555
|
+
rows.append({
|
|
556
|
+
"analysis": labels.get(measure, measure),
|
|
557
|
+
"homogeneity_p_holm": row["p_value_homogeneity_holm"],
|
|
558
|
+
"separation": row["separation"],
|
|
559
|
+
"separation_p_holm": row["p_value_separation_holm"],
|
|
560
|
+
"some_group_differs": bool(row["p_value_homogeneity_holm"] < alpha),
|
|
561
|
+
"evidence_of_separation": bool(
|
|
562
|
+
row["separation"] > 0 and row["p_value_separation_holm"] < alpha
|
|
563
|
+
),
|
|
564
|
+
})
|
|
565
|
+
return pd.DataFrame(rows).set_index("analysis")
|
|
566
|
+
|
|
567
|
+
@property
|
|
568
|
+
def pairwise_overview(self) -> pd.DataFrame:
|
|
569
|
+
overall = self.pairwise.xs("overall", level="measure").reset_index()
|
|
570
|
+
alpha = _metadata_alpha(self.metadata)
|
|
571
|
+
overall["pair"] = overall["group_a"] + " × " + overall["group_b"]
|
|
572
|
+
overall["more_homogeneous"] = overall.apply(
|
|
573
|
+
lambda row: row["group_a"] if row["within_difference"] > 0
|
|
574
|
+
else row["group_b"] if row["within_difference"] < 0 else "tie",
|
|
575
|
+
axis=1,
|
|
576
|
+
)
|
|
577
|
+
overall["evidence_of_separation"] = (
|
|
578
|
+
(overall["separation"] > 0)
|
|
579
|
+
& (overall["p_value_separation_holm"] < alpha)
|
|
580
|
+
)
|
|
581
|
+
return overall.set_index("pair")[[
|
|
582
|
+
"more_homogeneous",
|
|
583
|
+
"within_difference",
|
|
584
|
+
"separation",
|
|
585
|
+
"p_value_homogeneity_holm",
|
|
586
|
+
"p_value_separation_holm",
|
|
587
|
+
"evidence_of_separation",
|
|
588
|
+
]]
|
|
589
|
+
|
|
590
|
+
@property
|
|
591
|
+
def interpretation(self) -> str:
|
|
592
|
+
row = self.global_test.loc["overall"]
|
|
593
|
+
alpha = _metadata_alpha(self.metadata)
|
|
594
|
+
homogeneity = (
|
|
595
|
+
"Há evidência de que ao menos um grupo possui homogeneidade diferente"
|
|
596
|
+
if row["p_value_homogeneity_holm"] < alpha
|
|
597
|
+
else "Não há evidência suficiente de diferença global de homogeneidade"
|
|
598
|
+
)
|
|
599
|
+
separation = (
|
|
600
|
+
"há evidência de separação global entre os grupos"
|
|
601
|
+
if row["separation"] > 0 and row["p_value_separation_holm"] < alpha
|
|
602
|
+
else "não há evidência suficiente de separação global entre os grupos"
|
|
603
|
+
)
|
|
604
|
+
return (
|
|
605
|
+
f"{homogeneity} (p={row['p_value_homogeneity_holm']:.4g}); {separation} "
|
|
606
|
+
f"(separação={row['separation']:.4f}; "
|
|
607
|
+
f"p={row['p_value_separation_holm']:.4g}). "
|
|
608
|
+
"As comparações pareadas são pós-hoc e devem ser interpretadas em conjunto."
|
|
609
|
+
)
|
|
610
|
+
|
|
611
|
+
def print_report(self) -> None:
|
|
612
|
+
print("Teste global:")
|
|
613
|
+
print(self.overview.round(4))
|
|
614
|
+
print("\nComparações pareadas (resultado geral):")
|
|
615
|
+
print(self.pairwise_overview.round(4))
|
|
616
|
+
print(f"\n{self.interpretation}")
|
|
617
|
+
|
|
618
|
+
def export(self, directory: str | Path) -> None:
|
|
619
|
+
output = Path(directory)
|
|
620
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
621
|
+
safe_to_csv(self.global_test, output / "global_test.csv", index_label="measure")
|
|
622
|
+
safe_to_csv(self.pairwise, output / "pairwise.csv")
|
|
623
|
+
safe_to_csv(self.within_groups, output / "within_groups.csv", index_label="group")
|
|
624
|
+
safe_to_csv(self.between_groups, output / "between_groups.csv")
|
|
625
|
+
payload = {
|
|
626
|
+
"group_names": list(self.group_names),
|
|
627
|
+
"files": self.files,
|
|
628
|
+
"metrics": list(self.metrics),
|
|
629
|
+
"metadata": self.metadata,
|
|
630
|
+
"global_test": frame_json(self.global_test.reset_index(), "records"),
|
|
631
|
+
"pairwise": frame_json(self.pairwise.reset_index(), "records"),
|
|
632
|
+
}
|
|
633
|
+
write_json(payload, output / "group_comparison.json")
|
|
634
|
+
|
|
635
|
+
def to_excel(self, path: str | Path) -> None:
|
|
636
|
+
target = Path(path)
|
|
637
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
638
|
+
with excel_writer(target) as writer:
|
|
639
|
+
safe_to_excel(self.overview, writer, sheet_name="Overview")
|
|
640
|
+
safe_to_excel(self.pairwise_overview,
|
|
641
|
+
writer, sheet_name="Overview", startrow=len(self.overview) + 3
|
|
642
|
+
)
|
|
643
|
+
safe_to_excel(pd.DataFrame({"interpretation": [self.interpretation]}),
|
|
644
|
+
writer,
|
|
645
|
+
sheet_name="Overview",
|
|
646
|
+
startrow=len(self.overview) + len(self.pairwise_overview) + 7,
|
|
647
|
+
index=False,
|
|
648
|
+
)
|
|
649
|
+
safe_to_excel(self.global_test, writer, sheet_name="Global tests")
|
|
650
|
+
safe_to_excel(self.pairwise, writer, sheet_name="Pairwise")
|
|
651
|
+
safe_to_excel(self.within_groups, writer, sheet_name="Within groups")
|
|
652
|
+
safe_to_excel(self.between_groups, writer, sheet_name="Between groups")
|
|
653
|
+
safe_to_excel(pd.DataFrame(
|
|
654
|
+
[(group, file) for group, files in self.files.items() for file in files],
|
|
655
|
+
columns=["group", "file"],
|
|
656
|
+
), writer, sheet_name="Files", index=False)
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def _compare_multiple_groups(
|
|
660
|
+
group_inputs: Mapping[str, str | Path | Sequence[str | Path]],
|
|
661
|
+
*,
|
|
662
|
+
metrics: str | Sequence[str],
|
|
663
|
+
permutations: int,
|
|
664
|
+
random_state: int,
|
|
665
|
+
include_ast: bool,
|
|
666
|
+
alpha: float,
|
|
667
|
+
progress: bool | Callable[[str], None],
|
|
668
|
+
cache_dir: str | Path | None,
|
|
669
|
+
ast_timeout: float,
|
|
670
|
+
max_ted_cells: int | None,
|
|
671
|
+
) -> MultiGroupComparisonResult:
|
|
672
|
+
names = tuple(group_inputs)
|
|
673
|
+
if len(names) < 2 or any(not isinstance(name, str) or not name for name in names):
|
|
674
|
+
raise AnalysisError("Informe ao menos dois grupos com nomes não vazios.")
|
|
675
|
+
_validate_options(permutations, random_state, include_ast, alpha, progress, ast_timeout, max_ted_cells)
|
|
676
|
+
report = (
|
|
677
|
+
(lambda message: print(f"[codevariability] {message}", file=sys.stderr, flush=True))
|
|
678
|
+
if progress is True else progress if callable(progress) else None
|
|
679
|
+
)
|
|
680
|
+
if report is not None:
|
|
681
|
+
report(f"Carregando e validando {len(names)} grupos...")
|
|
682
|
+
|
|
683
|
+
datasets = {name: _dataset(value) for name, value in group_inputs.items()}
|
|
684
|
+
if any(len(dataset.files) < 2 for dataset in datasets.values()):
|
|
685
|
+
raise AnalysisError("Cada grupo deve conter pelo menos dois arquivos.")
|
|
686
|
+
_independent_files(datasets)
|
|
687
|
+
|
|
688
|
+
combined: dict[str, Path] = {}
|
|
689
|
+
public_files: dict[str, list[str]] = {}
|
|
690
|
+
index_groups: dict[str, list[int]] = {}
|
|
691
|
+
for name, dataset in datasets.items():
|
|
692
|
+
public_files[name] = list(dataset.files)
|
|
693
|
+
index_groups[name] = []
|
|
694
|
+
for file_name, path_value in dataset.files.items():
|
|
695
|
+
index_groups[name].append(len(combined))
|
|
696
|
+
combined[f"{name}/{file_name}"] = path_value
|
|
697
|
+
if report is not None:
|
|
698
|
+
report(f"Calculando métricas base para {len(combined)} arquivos...")
|
|
699
|
+
analysis = CodeDataset(combined).analyze(metrics, max_ted_cells=max_ted_cells)
|
|
700
|
+
javascript_ast_added = False
|
|
701
|
+
javascript_ast_cache: object = None
|
|
702
|
+
if include_ast and AST_TREE_EDIT_METRIC not in analysis.metrics:
|
|
703
|
+
if report is not None:
|
|
704
|
+
pairs = len(combined) * (len(combined) - 1) // 2
|
|
705
|
+
report(f"Calculando AST TED JavaScript: {pairs} pares únicos...")
|
|
706
|
+
metric, matrix = _javascript_ast_matrix(combined, report, cache_dir, ast_timeout, max_ted_cells)
|
|
707
|
+
_check_adapter_snapshot(analysis, matrix)
|
|
708
|
+
javascript_ast_cache = matrix.attrs.get("adapter_metadata", {}).get("cache")
|
|
709
|
+
analysis = analysis.with_metric(metric, matrix)
|
|
710
|
+
javascript_ast_added = True
|
|
711
|
+
measures = _measure_matrices(analysis)
|
|
712
|
+
|
|
713
|
+
global_rows: dict[str, dict[str, object]] = {}
|
|
714
|
+
within_values: dict[str, dict[str, float]] = {name: {} for name in names}
|
|
715
|
+
between_values: dict[tuple[str, str], dict[str, float]] = {
|
|
716
|
+
pair: {} for pair in combinations(names, 2)
|
|
717
|
+
}
|
|
718
|
+
pairwise_rows: dict[tuple[str, str, str], dict[str, object]] = {}
|
|
719
|
+
group_sizes = [len(index_groups[name]) for name in names]
|
|
720
|
+
total_steps = len(measures) * (1 + len(between_values))
|
|
721
|
+
step = 0
|
|
722
|
+
for measure, (kind, matrix) in measures.items():
|
|
723
|
+
step += 1
|
|
724
|
+
if report is not None:
|
|
725
|
+
report(f"Teste global para {measure} ({step}/{total_steps})...")
|
|
726
|
+
within, between, homogeneity, separation = _multi_group_statistics(matrix, index_groups)
|
|
727
|
+
p_homogeneity, p_separation = _global_permutation_p_values(
|
|
728
|
+
matrix, names, group_sizes, homogeneity, separation, permutations, random_state
|
|
729
|
+
)
|
|
730
|
+
global_rows[measure] = {
|
|
731
|
+
"kind": kind,
|
|
732
|
+
"mean_within": _mean(list(within.values())),
|
|
733
|
+
"mean_between": _mean(list(between.values())),
|
|
734
|
+
"homogeneity_statistic": homogeneity,
|
|
735
|
+
"separation": separation,
|
|
736
|
+
"p_value_homogeneity": p_homogeneity,
|
|
737
|
+
"p_value_separation": p_separation,
|
|
738
|
+
}
|
|
739
|
+
for name, value in within.items():
|
|
740
|
+
within_values[name][measure] = value
|
|
741
|
+
for pair, value in between.items():
|
|
742
|
+
between_values[pair][measure] = value
|
|
743
|
+
|
|
744
|
+
for left, right in combinations(names, 2):
|
|
745
|
+
step += 1
|
|
746
|
+
if report is not None:
|
|
747
|
+
report(f"Pós-teste {left} × {right}, {measure} ({step}/{total_steps})...")
|
|
748
|
+
stats = _group_statistics(matrix, index_groups[left], index_groups[right])
|
|
749
|
+
within_left, within_right, between_pair, difference, pair_separation, left_pairs, right_pairs = stats
|
|
750
|
+
pair_matrix_indices = [*index_groups[left], *index_groups[right]]
|
|
751
|
+
pair_matrix = matrix.iloc[pair_matrix_indices, pair_matrix_indices]
|
|
752
|
+
p_pair_h, p_pair_s = _permutation_p_values(
|
|
753
|
+
pair_matrix,
|
|
754
|
+
len(index_groups[left]),
|
|
755
|
+
difference,
|
|
756
|
+
pair_separation,
|
|
757
|
+
permutations,
|
|
758
|
+
random_state,
|
|
759
|
+
)
|
|
760
|
+
pairwise_rows[(left, right, measure)] = {
|
|
761
|
+
"kind": kind,
|
|
762
|
+
"within_group_a": within_left,
|
|
763
|
+
"within_group_b": within_right,
|
|
764
|
+
"between_groups": between_pair,
|
|
765
|
+
"within_difference": difference,
|
|
766
|
+
"separation": pair_separation,
|
|
767
|
+
"cliffs_delta_within": _cliffs_delta(left_pairs, right_pairs),
|
|
768
|
+
"p_value_homogeneity": p_pair_h,
|
|
769
|
+
"p_value_separation": p_pair_s,
|
|
770
|
+
}
|
|
771
|
+
|
|
772
|
+
global_test = pd.DataFrame.from_dict(global_rows, orient="index")
|
|
773
|
+
global_test.index.name = "measure"
|
|
774
|
+
for source, target in (
|
|
775
|
+
("p_value_homogeneity", "p_value_homogeneity_holm"),
|
|
776
|
+
("p_value_separation", "p_value_separation_holm"),
|
|
777
|
+
):
|
|
778
|
+
global_test[target] = pd.Series(_holm(global_test[source].to_dict()))
|
|
779
|
+
|
|
780
|
+
pairwise = pd.DataFrame.from_dict(pairwise_rows, orient="index")
|
|
781
|
+
pairwise.index = pd.MultiIndex.from_tuples(
|
|
782
|
+
pairwise.index, names=["group_a", "group_b", "measure"]
|
|
783
|
+
)
|
|
784
|
+
for source, target in (
|
|
785
|
+
("p_value_homogeneity", "p_value_homogeneity_holm"),
|
|
786
|
+
("p_value_separation", "p_value_separation_holm"),
|
|
787
|
+
):
|
|
788
|
+
raw = {key: float(value) for key, value in pairwise[source].items()}
|
|
789
|
+
pairwise[target] = pd.Series(_holm(raw))
|
|
790
|
+
|
|
791
|
+
dimensions = analysis.metadata.get("metric_dimensions", {})
|
|
792
|
+
result = MultiGroupComparisonResult(
|
|
793
|
+
global_test=global_test,
|
|
794
|
+
pairwise=pairwise,
|
|
795
|
+
within_groups=pd.DataFrame.from_dict(within_values, orient="index"),
|
|
796
|
+
between_groups=pd.DataFrame.from_dict(between_values, orient="index").rename_axis(
|
|
797
|
+
index=["group_a", "group_b"]
|
|
798
|
+
),
|
|
799
|
+
group_names=names,
|
|
800
|
+
files=public_files,
|
|
801
|
+
metrics=analysis.metrics,
|
|
802
|
+
metadata={
|
|
803
|
+
**deepcopy(analysis.metadata),
|
|
804
|
+
"ast_timeout": ast_timeout,
|
|
805
|
+
"method": "multigroup_file_label_permutation_v1",
|
|
806
|
+
"alternative_homogeneity": "upper-tail squared dispersion of within-group means",
|
|
807
|
+
"alternative_separation": "two-sided",
|
|
808
|
+
"group_weighting": "equal",
|
|
809
|
+
"permutations": permutations,
|
|
810
|
+
"random_state": random_state,
|
|
811
|
+
"alpha": float(alpha),
|
|
812
|
+
"global_p_value_correction": "Holm across all reported measures, separately by hypothesis family",
|
|
813
|
+
"pairwise_p_value_correction": "Holm across every pair x measure, separately by hypothesis family",
|
|
814
|
+
"metric_ids": analysis.metadata.get("metric_ids", {}),
|
|
815
|
+
"metric_dimensions": dimensions,
|
|
816
|
+
"overall_aggregation": analysis.metadata.get("overall_aggregation", {}),
|
|
817
|
+
"javascript_ast_included": javascript_ast_added,
|
|
818
|
+
"javascript_ast_cache": javascript_ast_cache,
|
|
819
|
+
"cache_dir": str(Path(cache_dir).resolve()) if cache_dir is not None else None,
|
|
820
|
+
},
|
|
821
|
+
)
|
|
822
|
+
if report is not None:
|
|
823
|
+
report("Comparação multigrupo concluída.")
|
|
824
|
+
return result
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
def compare_groups(
|
|
828
|
+
group_a: Mapping[str, str | Path | Sequence[str | Path]] | str | Path | Sequence[str | Path],
|
|
829
|
+
group_b: str | Path | Sequence[str | Path] | None = None,
|
|
830
|
+
*,
|
|
831
|
+
metrics: str | Sequence[str] = "all",
|
|
832
|
+
group_names: tuple[str, str] = ("group_a", "group_b"),
|
|
833
|
+
permutations: int = 10_000,
|
|
834
|
+
random_state: int = 42,
|
|
835
|
+
include_ast: bool = False,
|
|
836
|
+
alpha: float = 0.05,
|
|
837
|
+
progress: bool | Callable[[str], None] = False,
|
|
838
|
+
cache_dir: str | Path | None = None,
|
|
839
|
+
ast_timeout: float = 120.0,
|
|
840
|
+
max_ted_cells: int | None = 2_000_000,
|
|
841
|
+
) -> GroupComparisonResult | MultiGroupComparisonResult:
|
|
842
|
+
"""Compare internal homogeneity and separation of independent groups.
|
|
843
|
+
|
|
844
|
+
P-values come from permutation of file-level group labels while preserving
|
|
845
|
+
the original group sizes. Pairwise matrix cells are never permuted as if
|
|
846
|
+
they were independent observations. Pass two inputs for a direct comparison
|
|
847
|
+
or an ordered mapping of names to inputs for a global multigroup analysis.
|
|
848
|
+
"""
|
|
849
|
+
if isinstance(group_a, Mapping):
|
|
850
|
+
if group_b is not None:
|
|
851
|
+
raise AnalysisError("Não informe group_b ao usar o formato multigrupo.")
|
|
852
|
+
return _compare_multiple_groups(
|
|
853
|
+
group_a,
|
|
854
|
+
metrics=metrics,
|
|
855
|
+
permutations=permutations,
|
|
856
|
+
random_state=random_state,
|
|
857
|
+
include_ast=include_ast,
|
|
858
|
+
alpha=alpha,
|
|
859
|
+
progress=progress,
|
|
860
|
+
cache_dir=cache_dir,
|
|
861
|
+
ast_timeout=ast_timeout,
|
|
862
|
+
max_ted_cells=max_ted_cells,
|
|
863
|
+
)
|
|
864
|
+
if group_b is None:
|
|
865
|
+
raise AnalysisError("Informe o segundo grupo ou um mapeamento com dois ou mais grupos.")
|
|
866
|
+
if not isinstance(group_names, (tuple, list)) or len(group_names) != 2 or any(not isinstance(name, str) or not name.strip() for name in group_names) or group_names[0] == group_names[1]:
|
|
867
|
+
raise AnalysisError("Informe dois nomes de grupo distintos e não vazios.")
|
|
868
|
+
_validate_options(permutations, random_state, include_ast, alpha, progress, ast_timeout, max_ted_cells)
|
|
869
|
+
report: Callable[[str], None] | None
|
|
870
|
+
if progress is True:
|
|
871
|
+
def report(message: str) -> None:
|
|
872
|
+
print(f"[codevariability] {message}", file=sys.stderr, flush=True)
|
|
873
|
+
elif callable(progress):
|
|
874
|
+
report = progress
|
|
875
|
+
else:
|
|
876
|
+
report = None
|
|
877
|
+
|
|
878
|
+
if report is not None:
|
|
879
|
+
report("Carregando e validando os dois grupos...")
|
|
880
|
+
|
|
881
|
+
datasets = (_dataset(group_a), _dataset(group_b))
|
|
882
|
+
if any(len(dataset.files) < 2 for dataset in datasets):
|
|
883
|
+
raise AnalysisError("Cada grupo deve conter pelo menos dois arquivos.")
|
|
884
|
+
_independent_files(dict(zip(group_names, datasets, strict=True)))
|
|
885
|
+
|
|
886
|
+
combined: dict[str, Path] = {}
|
|
887
|
+
public_files: dict[str, list[str]] = {}
|
|
888
|
+
for group, dataset in zip(group_names, datasets, strict=True):
|
|
889
|
+
public_files[group] = list(dataset.files)
|
|
890
|
+
for name, path in dataset.files.items():
|
|
891
|
+
combined[f"{group}/{name}"] = path
|
|
892
|
+
if report is not None:
|
|
893
|
+
report(f"Calculando métricas base para {len(combined)} arquivos...")
|
|
894
|
+
analysis = CodeDataset(combined).analyze(metrics, max_ted_cells=max_ted_cells)
|
|
895
|
+
javascript_ast_added = False
|
|
896
|
+
javascript_ast_cache: object = None
|
|
897
|
+
if include_ast and AST_TREE_EDIT_METRIC not in analysis.metrics:
|
|
898
|
+
if report is not None:
|
|
899
|
+
pairs = len(combined) * (len(combined) - 1) // 2
|
|
900
|
+
report(f"Calculando AST TED JavaScript: {pairs} pares únicos...")
|
|
901
|
+
metric, matrix = _javascript_ast_matrix(combined, report, cache_dir, ast_timeout, max_ted_cells)
|
|
902
|
+
_check_adapter_snapshot(analysis, matrix)
|
|
903
|
+
javascript_ast_cache = matrix.attrs.get("adapter_metadata", {}).get("cache")
|
|
904
|
+
analysis = analysis.with_metric(metric, matrix)
|
|
905
|
+
javascript_ast_added = True
|
|
906
|
+
|
|
907
|
+
dimensions = analysis.metadata.get("metric_dimensions", {})
|
|
908
|
+
measures = _measure_matrices(analysis)
|
|
909
|
+
|
|
910
|
+
size_left = len(datasets[0].files)
|
|
911
|
+
left_indices = list(range(size_left))
|
|
912
|
+
right_indices = list(range(size_left, len(combined)))
|
|
913
|
+
rows: dict[str, dict[str, object]] = {}
|
|
914
|
+
for measure_index, (measure, (kind, matrix)) in enumerate(measures.items(), start=1):
|
|
915
|
+
if report is not None:
|
|
916
|
+
report(
|
|
917
|
+
f"Permutações para {measure} ({measure_index}/{len(measures)}; "
|
|
918
|
+
f"{permutations} repetições)..."
|
|
919
|
+
)
|
|
920
|
+
within_left, within_right, between, difference, separation, left_values, right_values = _group_statistics(
|
|
921
|
+
matrix, left_indices, right_indices
|
|
922
|
+
)
|
|
923
|
+
p_homogeneity, p_separation = _permutation_p_values(
|
|
924
|
+
matrix,
|
|
925
|
+
size_left,
|
|
926
|
+
difference,
|
|
927
|
+
separation,
|
|
928
|
+
permutations,
|
|
929
|
+
random_state,
|
|
930
|
+
)
|
|
931
|
+
rows[measure] = {
|
|
932
|
+
"kind": kind,
|
|
933
|
+
"within_group_a": within_left,
|
|
934
|
+
"within_group_b": within_right,
|
|
935
|
+
"between_groups": between,
|
|
936
|
+
"within_difference": difference,
|
|
937
|
+
"separation": separation,
|
|
938
|
+
"cliffs_delta_within": _cliffs_delta(left_values, right_values),
|
|
939
|
+
"p_value_homogeneity": p_homogeneity,
|
|
940
|
+
"p_value_separation": p_separation,
|
|
941
|
+
}
|
|
942
|
+
summary = pd.DataFrame.from_dict(rows, orient="index")
|
|
943
|
+
names = (group_names[0], group_names[1])
|
|
944
|
+
within_columns = _within_columns(names, set(summary.columns) - {"within_group_a", "within_group_b"})
|
|
945
|
+
summary = summary.rename(columns=dict(zip(("within_group_a", "within_group_b"), within_columns, strict=True)))
|
|
946
|
+
summary.index.name = "measure"
|
|
947
|
+
for source, target in (
|
|
948
|
+
("p_value_homogeneity", "p_value_homogeneity_holm"),
|
|
949
|
+
("p_value_separation", "p_value_separation_holm"),
|
|
950
|
+
):
|
|
951
|
+
adjusted = _holm(summary[source].to_dict())
|
|
952
|
+
summary[target] = pd.Series(adjusted)
|
|
953
|
+
|
|
954
|
+
result = GroupComparisonResult(
|
|
955
|
+
summary=summary,
|
|
956
|
+
group_names=names,
|
|
957
|
+
files=public_files,
|
|
958
|
+
metrics=analysis.metrics,
|
|
959
|
+
metadata={
|
|
960
|
+
**deepcopy(analysis.metadata),
|
|
961
|
+
"within_group_columns": list(within_columns),
|
|
962
|
+
"ast_timeout": ast_timeout,
|
|
963
|
+
"method": "file_label_permutation_v1",
|
|
964
|
+
"alternative": "two-sided",
|
|
965
|
+
"permutations": permutations,
|
|
966
|
+
"random_state": random_state,
|
|
967
|
+
"alpha": float(alpha),
|
|
968
|
+
"javascript_ast_included": javascript_ast_added,
|
|
969
|
+
"javascript_ast_cache": javascript_ast_cache,
|
|
970
|
+
"cache_dir": str(Path(cache_dir).resolve()) if cache_dir is not None else None,
|
|
971
|
+
"p_value_correction": "Holm separately by hypothesis family",
|
|
972
|
+
"metric_ids": analysis.metadata.get("metric_ids", {}),
|
|
973
|
+
"metric_dimensions": dimensions,
|
|
974
|
+
"overall_aggregation": analysis.metadata.get("overall_aggregation", {}),
|
|
975
|
+
},
|
|
976
|
+
)
|
|
977
|
+
if report is not None:
|
|
978
|
+
report("Comparação concluída.")
|
|
979
|
+
return result
|