codevariability 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,979 @@
1
+ """Permutation-based comparison of two or more independent code collections."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ import os
7
+ import random
8
+ import shutil
9
+ import signal
10
+ import subprocess
11
+ import sys
12
+ import tempfile
13
+ import threading
14
+ from collections import deque
15
+ from contextlib import suppress
16
+ from copy import deepcopy
17
+ from dataclasses import dataclass
18
+ from itertools import combinations
19
+ from pathlib import Path
20
+ from typing import Any, Callable, Mapping, Sequence
21
+
22
+ import pandas as pd
23
+
24
+ from .analysis import AnalysisResult, CodeDataset, _dimension_groups
25
+ from .exceptions import AnalysisError
26
+ from .interop import load_matrix_json
27
+ from .output import excel_writer, frame_json, write_json
28
+ from .spreadsheet import safe_to_csv, safe_to_excel
29
+
30
+ AST_TREE_EDIT_METRIC = "ast_tree_edit_similarity"
31
+ JAVASCRIPT_EXTENSIONS = {".js", ".jsx", ".ts", ".tsx", ".md", ".markdown"}
32
+
33
+
34
+ def _dataset(value: str | Path | Sequence[str | Path]) -> CodeDataset:
35
+ if isinstance(value, (str, Path)) and Path(value).is_dir():
36
+ return CodeDataset.from_directory(value)
37
+ files = [value] if isinstance(value, (str, Path)) else value
38
+ return CodeDataset.from_files(files)
39
+
40
+
41
+ def _mean(values: list[float]) -> float:
42
+ return sum(values) / len(values)
43
+
44
+
45
+ def _within(matrix: pd.DataFrame, indices: Sequence[int]) -> tuple[float, list[float]]:
46
+ values = [float(matrix.iat[i, j]) for i, j in combinations(indices, 2)]
47
+ return _mean(values), values
48
+
49
+
50
+ def _between(matrix: pd.DataFrame, left: Sequence[int], right: Sequence[int]) -> float:
51
+ return _mean([float(matrix.iat[i, j]) for i in left for j in right])
52
+
53
+
54
+ def _group_statistics(
55
+ matrix: pd.DataFrame, left: Sequence[int], right: Sequence[int]
56
+ ) -> tuple[float, float, float, float, float, list[float], list[float]]:
57
+ within_left, left_values = _within(matrix, left)
58
+ within_right, right_values = _within(matrix, right)
59
+ between = _between(matrix, left, right)
60
+ return (
61
+ within_left,
62
+ within_right,
63
+ between,
64
+ within_left - within_right,
65
+ (within_left + within_right) / 2 - between,
66
+ left_values,
67
+ right_values,
68
+ )
69
+
70
+
71
+ def _array_group_statistics(
72
+ values: Any, left: Sequence[int], right: Sequence[int]
73
+ ) -> tuple[float, float]:
74
+ """Hot-path statistics over a NumPy-backed matrix without DataFrame indexing."""
75
+ within_left = _mean([float(values[i, j]) for i, j in combinations(left, 2)])
76
+ within_right = _mean([float(values[i, j]) for i, j in combinations(right, 2)])
77
+ between = _mean([float(values[i, j]) for i in left for j in right])
78
+ return within_left - within_right, (within_left + within_right) / 2 - between
79
+
80
+
81
+ def _cliffs_delta(left: Sequence[float], right: Sequence[float]) -> float:
82
+ comparisons = len(left) * len(right)
83
+ if comparisons == 0:
84
+ return 0.0
85
+ greater = sum(a > b for a in left for b in right)
86
+ lower = sum(a < b for a in left for b in right)
87
+ return (greater - lower) / comparisons
88
+
89
+
90
+ def _permutation_p_values(
91
+ matrix: pd.DataFrame,
92
+ size_left: int,
93
+ observed_homogeneity: float,
94
+ observed_separation: float,
95
+ permutations: int,
96
+ random_state: int,
97
+ ) -> tuple[float, float]:
98
+ rng = random.Random(random_state)
99
+ values = matrix.to_numpy(dtype=float, copy=False)
100
+ population = list(range(len(matrix)))
101
+ extreme_homogeneity = 0
102
+ extreme_separation = 0
103
+ for _ in range(permutations):
104
+ left_set = set(rng.sample(population, size_left))
105
+ left = [index for index in population if index in left_set]
106
+ right = [index for index in population if index not in left_set]
107
+ homogeneity, separation = _array_group_statistics(values, left, right)
108
+ extreme_homogeneity += abs(homogeneity) >= abs(observed_homogeneity)
109
+ extreme_separation += abs(separation) >= abs(observed_separation)
110
+ denominator = permutations + 1
111
+ return (
112
+ (extreme_homogeneity + 1) / denominator,
113
+ (extreme_separation + 1) / denominator,
114
+ )
115
+
116
+
117
+ def _holm(values: Mapping[str, float]) -> dict[str, float]:
118
+ ordered = sorted(values, key=lambda name: values[name])
119
+ adjusted: dict[str, float] = {}
120
+ previous = 0.0
121
+ total = len(ordered)
122
+ for position, name in enumerate(ordered):
123
+ candidate = min(1.0, (total - position) * values[name])
124
+ previous = max(previous, candidate)
125
+ adjusted[name] = previous
126
+ return adjusted
127
+
128
+
129
+ def _mean_matrix(matrices: Sequence[pd.DataFrame]) -> pd.DataFrame:
130
+ return sum(matrices[1:], matrices[0].copy()) / len(matrices)
131
+
132
+
133
+ def _measure_matrices(analysis: AnalysisResult) -> dict[str, tuple[str, pd.DataFrame]]:
134
+ dimensions = analysis.metadata.get("metric_dimensions", {})
135
+ groups = _dimension_groups(
136
+ analysis.metrics, dimensions if isinstance(dimensions, Mapping) else {}
137
+ )
138
+ measures: dict[str, tuple[str, pd.DataFrame]] = {
139
+ metric: ("metric", matrix) for metric, matrix in analysis.matrices.items()
140
+ }
141
+ dimension_matrices: list[pd.DataFrame] = []
142
+ for dimension, dimension_metrics in groups.items():
143
+ matrix = _mean_matrix([analysis.matrices[name] for name in dimension_metrics])
144
+ measures[f"dimension:{dimension}"] = ("dimension", matrix)
145
+ dimension_matrices.append(matrix)
146
+ measures["overall"] = ("overall", _mean_matrix(dimension_matrices))
147
+ return measures
148
+
149
+
150
+ def _multi_group_statistics(
151
+ matrix: pd.DataFrame, groups: Mapping[str, Sequence[int]]
152
+ ) -> tuple[dict[str, float], dict[tuple[str, str], float], float, float]:
153
+ within = {name: _within(matrix, indices)[0] for name, indices in groups.items()}
154
+ between = {
155
+ (left, right): _between(matrix, groups[left], groups[right])
156
+ for left, right in combinations(groups, 2)
157
+ }
158
+ mean_within = _mean(list(within.values()))
159
+ mean_between = _mean(list(between.values()))
160
+ homogeneity = sum((value - mean_within) ** 2 for value in within.values())
161
+ return within, between, homogeneity, mean_within - mean_between
162
+
163
+
164
+ def _global_permutation_p_values(
165
+ matrix: pd.DataFrame,
166
+ group_names: Sequence[str],
167
+ group_sizes: Sequence[int],
168
+ observed_homogeneity: float,
169
+ observed_separation: float,
170
+ permutations: int,
171
+ random_state: int,
172
+ ) -> tuple[float, float]:
173
+ rng = random.Random(random_state)
174
+ values = matrix.to_numpy(dtype=float, copy=False)
175
+ population = list(range(len(matrix)))
176
+ extreme_homogeneity = 0
177
+ extreme_separation = 0
178
+ for _ in range(permutations):
179
+ shuffled = rng.sample(population, len(population))
180
+ offset = 0
181
+ groups: dict[str, list[int]] = {}
182
+ for name, size in zip(group_names, group_sizes, strict=True):
183
+ groups[name] = shuffled[offset:offset + size]
184
+ offset += size
185
+ within = {
186
+ name: _mean([
187
+ float(values[i, j]) for i, j in combinations(indices, 2)
188
+ ])
189
+ for name, indices in groups.items()
190
+ }
191
+ between = [
192
+ _mean([
193
+ float(values[i, j])
194
+ for i in groups[left]
195
+ for j in groups[right]
196
+ ])
197
+ for left, right in combinations(group_names, 2)
198
+ ]
199
+ mean_within = _mean(list(within.values()))
200
+ homogeneity = sum((value - mean_within) ** 2 for value in within.values())
201
+ separation = mean_within - _mean(between)
202
+ extreme_homogeneity += homogeneity >= observed_homogeneity
203
+ extreme_separation += abs(separation) >= abs(observed_separation)
204
+ denominator = permutations + 1
205
+ return (
206
+ (extreme_homogeneity + 1) / denominator,
207
+ (extreme_separation + 1) / denominator,
208
+ )
209
+
210
+
211
+ def _javascript_adapter_command() -> list[str]:
212
+ installed = shutil.which("codevariability-js")
213
+ if installed:
214
+ return [installed]
215
+ raise AnalysisError(
216
+ "A TED JavaScript requer Node.js e o pacote codevariability-js instalado."
217
+ )
218
+
219
+
220
+ def _validate_options(permutations, random_state, include_ast, alpha, progress, ast_timeout, max_ted_cells):
221
+ if not isinstance(permutations, int) or isinstance(permutations, bool) or permutations < 1:
222
+ raise AnalysisError("permutations deve ser um inteiro positivo.")
223
+ if not isinstance(random_state, int) or isinstance(random_state, bool):
224
+ raise AnalysisError("random_state deve ser um inteiro.")
225
+ if not isinstance(include_ast, bool):
226
+ raise AnalysisError("include_ast deve ser True ou False.")
227
+ if not isinstance(alpha, (int, float)) or isinstance(alpha, bool) or not 0 < alpha < 1:
228
+ raise AnalysisError("alpha deve estar estritamente entre 0 e 1.")
229
+ if not isinstance(progress, bool) and not callable(progress):
230
+ raise AnalysisError("progress deve ser True, False ou uma função.")
231
+ try:
232
+ valid_timeout = isinstance(ast_timeout, (int, float)) and not isinstance(ast_timeout, bool) and math.isfinite(ast_timeout) and ast_timeout > 0
233
+ except OverflowError:
234
+ valid_timeout = False
235
+ if not valid_timeout:
236
+ raise AnalysisError("ast_timeout deve ser um número finito e positivo em segundos.")
237
+ if max_ted_cells is not None and (not isinstance(max_ted_cells, int) or isinstance(max_ted_cells, bool) or max_ted_cells < 1):
238
+ raise AnalysisError("max_ted_cells deve ser um inteiro positivo ou None.")
239
+
240
+
241
+ def _independent_files(datasets: Mapping[str, CodeDataset]) -> None:
242
+ seen: set[tuple[int, int]] = set()
243
+ try:
244
+ for dataset in datasets.values():
245
+ for path in dataset.files.values():
246
+ stat = path.stat()
247
+ identity = (stat.st_dev, stat.st_ino)
248
+ if identity in seen:
249
+ raise AnalysisError("Os grupos independentes não podem compartilhar arquivos físicos, inclusive hardlinks/symlinks.")
250
+ seen.add(identity)
251
+ except OSError as exc:
252
+ raise AnalysisError(f"Não foi possível verificar a identidade física dos arquivos: {exc}.") from exc
253
+
254
+
255
+ def _run_adapter(command: list[str], report, timeout: float) -> None:
256
+ """Drain bounded stderr in a reader while the main thread enforces timeout."""
257
+ lines: deque[str] = deque(maxlen=64)
258
+ reader_errors: list[BaseException] = []
259
+ try:
260
+ process = subprocess.Popen(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE,
261
+ text=True, encoding="utf-8", errors="replace", start_new_session=os.name == "posix")
262
+ except OSError as exc:
263
+ raise AnalysisError(f"Não foi possível executar o adaptador AST JavaScript: {exc}.") from exc
264
+ assert process.stderr is not None
265
+ stderr = process.stderr
266
+
267
+ def read_stderr():
268
+ try:
269
+ while True:
270
+ # Bound a single line as well as the tail buffer.
271
+ line = stderr.readline(4096)
272
+ if not line:
273
+ break
274
+ message = line.strip()
275
+ if message:
276
+ lines.append(message)
277
+ if report is not None:
278
+ report(message)
279
+ except BaseException as exc:
280
+ reader_errors.append(exc)
281
+
282
+ reader = threading.Thread(target=read_stderr, daemon=True)
283
+ reader.start()
284
+ try:
285
+ try:
286
+ return_code = process.wait(timeout=timeout)
287
+ except subprocess.TimeoutExpired as exc:
288
+ raise AnalysisError(f"O adaptador AST JavaScript excedeu ast_timeout={timeout} segundos.") from exc
289
+ reader.join(timeout=1)
290
+ if reader_errors:
291
+ raise AnalysisError(f"Falha ao acompanhar o adaptador AST: {reader_errors[0]}.")
292
+ if return_code:
293
+ raise AnalysisError(f"Falha no adaptador AST JavaScript: {lines[-1] if lines else 'falha desconhecida'}")
294
+ finally:
295
+ # Also stop descendants holding the stderr pipe after the leader exits.
296
+ if os.name == "posix":
297
+ with suppress(ProcessLookupError):
298
+ os.killpg(process.pid, signal.SIGKILL)
299
+ elif process.poll() is None:
300
+ process.kill()
301
+ process.wait()
302
+ reader.join(timeout=1)
303
+ stderr.close()
304
+
305
+
306
+ def _javascript_ast_matrix(
307
+ files: Mapping[str, Path],
308
+ report: Callable[[str], None] | None = None,
309
+ cache_dir: str | Path | None = None,
310
+ timeout: float = 120.0,
311
+ max_ted_cells: int | None = 2_000_000,
312
+ ) -> tuple[str, pd.DataFrame]:
313
+ unsupported = [label for label, path in files.items() if path.suffix.lower() not in JAVASCRIPT_EXTENSIONS]
314
+ if unsupported:
315
+ raise AnalysisError(
316
+ "include_ast=True requer somente JavaScript, JSX, TypeScript ou Markdown; "
317
+ f"entradas incompatíveis: {', '.join(unsupported)}."
318
+ )
319
+ with tempfile.TemporaryDirectory(prefix="codevariability-js-ast-") as temporary:
320
+ folder = Path(temporary)
321
+ temporary_files: list[str] = []
322
+ temporary_names: list[str] = []
323
+ for index, path in enumerate(files.values()):
324
+ target = folder / f"{index:06d}{path.suffix.lower()}"
325
+ shutil.copyfile(path, target)
326
+ temporary_files.append(str(target))
327
+ temporary_names.append(target.name)
328
+ output = folder / "ast-tree-edit.json"
329
+ command = [
330
+ *_javascript_adapter_command(),
331
+ "ast",
332
+ *temporary_files,
333
+ "--metric",
334
+ AST_TREE_EDIT_METRIC,
335
+ "--output",
336
+ str(output),
337
+ ]
338
+ if report is not None:
339
+ command.append("--progress")
340
+ if cache_dir is not None:
341
+ command.extend(["--cache-dir", str(Path(cache_dir).resolve())])
342
+ command.extend(["--max-cells", str(max_ted_cells) if max_ted_cells is not None else "none"])
343
+ _run_adapter(command, report, timeout)
344
+ metric, matrix = load_matrix_json(output)
345
+ if metric != AST_TREE_EDIT_METRIC or set(matrix.index) != set(temporary_names):
346
+ raise AnalysisError("O adaptador AST retornou uma métrica ou arquivos incompatíveis.")
347
+ labels = list(files)
348
+ rename = dict(zip(temporary_names, labels, strict=True))
349
+ attrs = deepcopy(matrix.attrs)
350
+ adapter_metadata = attrs.get("adapter_metadata", {})
351
+ hashes = adapter_metadata.get("input_sha256")
352
+ if isinstance(hashes, dict):
353
+ adapter_metadata["input_sha256"] = {rename.get(name, name): value for name, value in hashes.items()}
354
+ matrix = matrix.rename(index=rename, columns=rename).loc[labels, labels]
355
+ matrix.attrs.update(attrs)
356
+ return metric, matrix
357
+
358
+
359
+ def _metadata_alpha(metadata: Mapping[str, object]) -> float:
360
+ value = metadata.get("alpha", 0.05)
361
+ if not isinstance(value, (int, float, str)):
362
+ raise AnalysisError("metadata.alpha deve ser um número.")
363
+ try:
364
+ return float(value)
365
+ except ValueError as exc:
366
+ raise AnalysisError("metadata.alpha deve ser um número.") from exc
367
+
368
+
369
+ def _within_columns(names: tuple[str, str], reserved: set[str]) -> tuple[str, str]:
370
+ """Allocate presentation labels without mixing them with statistic keys."""
371
+ requested = [f"within_{name}" for name in names]
372
+ used = set(reserved)
373
+ columns = []
374
+ for candidate in requested:
375
+ column = candidate
376
+ while column in used or (column != candidate and column in requested):
377
+ column += "__group"
378
+ columns.append(column)
379
+ used.add(column)
380
+ return columns[0], columns[1]
381
+
382
+
383
+ def _check_adapter_snapshot(analysis: AnalysisResult, matrix: pd.DataFrame) -> None:
384
+ """Require the external metric to describe the same bytes as base metrics."""
385
+ adapter = matrix.attrs.get("adapter_metadata", {})
386
+ if not isinstance(adapter, Mapping) or adapter.get("input_sha256") != analysis.metadata.get("input_sha256"):
387
+ raise AnalysisError("Os arquivos mudaram entre as etapas da análise ou o adaptador não confirmou os hashes de entrada.")
388
+
389
+
390
+ @dataclass
391
+ class GroupComparisonResult:
392
+ """Results of an independent two-group comparison."""
393
+
394
+ summary: pd.DataFrame
395
+ group_names: tuple[str, str]
396
+ files: Mapping[str, list[str]]
397
+ metrics: tuple[str, ...]
398
+ metadata: dict[str, object]
399
+
400
+ @property
401
+ def within_group_columns(self) -> tuple[str, str]:
402
+ """Columns for the two named groups, separate from statistic columns."""
403
+ columns = self.metadata.get("within_group_columns")
404
+ if isinstance(columns, (list, tuple)) and len(columns) == 2 and all(isinstance(item, str) for item in columns):
405
+ return columns[0], columns[1]
406
+ left, right = self.group_names
407
+ return f"within_{left}", f"within_{right}"
408
+
409
+ @property
410
+ def overview(self) -> pd.DataFrame:
411
+ """Compact user-facing view; technical metric rows stay in ``summary``."""
412
+ left, right = self.group_names
413
+ left_column, right_column = self.within_group_columns
414
+ relevant = self.summary[self.summary["kind"].isin(["dimension", "overall"])]
415
+ labels = {
416
+ "dimension:textual": "Textual",
417
+ "dimension:syntactic_token_sequence": "Sintática",
418
+ "dimension:structural_ast": "Estrutural (AST TED)",
419
+ "overall": "Geral",
420
+ }
421
+ rows = []
422
+ alpha = _metadata_alpha(self.metadata)
423
+ for measure, row in relevant.iterrows():
424
+ difference = float(row["within_difference"])
425
+ rows.append({
426
+ "analysis": labels.get(measure, measure),
427
+ left_column: row[left_column],
428
+ right_column: row[right_column],
429
+ "between_groups": row["between_groups"],
430
+ "more_homogeneous": left if difference > 0 else right if difference < 0 else "tie",
431
+ "homogeneity_p_holm": row["p_value_homogeneity_holm"],
432
+ "separation": row["separation"],
433
+ "separation_p_holm": row["p_value_separation_holm"],
434
+ "evidence_of_separation": bool(
435
+ row["separation"] > 0 and row["p_value_separation_holm"] < alpha
436
+ ),
437
+ })
438
+ return pd.DataFrame(rows).set_index("analysis")
439
+
440
+ @property
441
+ def interpretation(self) -> str:
442
+ """Plain-language interpretation of the overall row."""
443
+ left, right = self.group_names
444
+ row = self.summary.loc["overall"]
445
+ alpha = _metadata_alpha(self.metadata)
446
+ difference = float(row["within_difference"])
447
+ if difference == 0:
448
+ homogeneity = f"{left} e {right} tiveram a mesma similaridade interna média"
449
+ else:
450
+ leader = left if difference > 0 else right
451
+ homogeneity = f"{leader} apresentou maior similaridade interna média"
452
+ homogeneity_evidence = (
453
+ "com evidência estatística"
454
+ if row["p_value_homogeneity_holm"] < alpha
455
+ else "sem evidência estatística suficiente"
456
+ )
457
+ separation_evidence = (
458
+ "Há evidência de separação entre os grupos"
459
+ if row["separation"] > 0 and row["p_value_separation_holm"] < alpha
460
+ else "Não há evidência suficiente de separação entre os grupos"
461
+ )
462
+ return (
463
+ f"{homogeneity}, {homogeneity_evidence} após correção de Holm "
464
+ f"(p={row['p_value_homogeneity_holm']:.4g}). {separation_evidence} "
465
+ f"(separação={row['separation']:.4f}; "
466
+ f"p={row['p_value_separation_holm']:.4g})."
467
+ )
468
+
469
+ def print_report(self) -> None:
470
+ """Print only the compact overview and its cautious interpretation."""
471
+ print(self.overview.round(4))
472
+ print(f"\n{self.interpretation}")
473
+
474
+ @property
475
+ def by_metric(self) -> pd.DataFrame:
476
+ return self.summary[self.summary["kind"] == "metric"].copy()
477
+
478
+ @property
479
+ def by_dimension(self) -> pd.DataFrame:
480
+ return self.summary[self.summary["kind"].isin(["dimension", "overall"])].copy()
481
+
482
+ @property
483
+ def within_groups(self) -> pd.DataFrame:
484
+ return self.summary[[*self.within_group_columns, "within_difference"]].copy()
485
+
486
+ @property
487
+ def between_groups(self) -> pd.Series:
488
+ return self.summary["between_groups"].copy()
489
+
490
+ @property
491
+ def p_values(self) -> pd.DataFrame:
492
+ return self.summary[[
493
+ "p_value_homogeneity",
494
+ "p_value_homogeneity_holm",
495
+ "p_value_separation",
496
+ "p_value_separation_holm",
497
+ ]].copy()
498
+
499
+ def export(self, directory: str | Path) -> None:
500
+ output = Path(directory)
501
+ output.mkdir(parents=True, exist_ok=True)
502
+ safe_to_csv(self.summary, output / "group_comparison.csv", index_label="measure")
503
+ payload = {
504
+ "group_names": list(self.group_names),
505
+ "files": self.files,
506
+ "metrics": list(self.metrics),
507
+ "metadata": self.metadata,
508
+ "summary": frame_json(self.summary.reset_index(), "records"),
509
+ }
510
+ write_json(payload, output / "group_comparison.json")
511
+
512
+ def to_excel(self, path: str | Path) -> None:
513
+ target = Path(path)
514
+ target.parent.mkdir(parents=True, exist_ok=True)
515
+ with excel_writer(target) as writer:
516
+ safe_to_excel(self.overview, writer, sheet_name="Overview")
517
+ safe_to_excel(pd.DataFrame({"interpretation": [self.interpretation]}), writer,
518
+ sheet_name="Overview", startrow=len(self.overview) + 3, index=False
519
+ )
520
+ safe_to_excel(self.summary, writer, sheet_name="Technical details")
521
+ safe_to_excel(self.by_metric, writer, sheet_name="Metrics")
522
+ safe_to_excel(pd.DataFrame(
523
+ [(group, file) for group, files in self.files.items() for file in files],
524
+ columns=["group", "file"],
525
+ ), writer, sheet_name="Files", index=False)
526
+
527
+
528
+ @dataclass
529
+ class MultiGroupComparisonResult:
530
+ """Global and post-hoc results for two or more independent groups."""
531
+
532
+ global_test: pd.DataFrame
533
+ pairwise: pd.DataFrame
534
+ within_groups: pd.DataFrame
535
+ between_groups: pd.DataFrame
536
+ group_names: tuple[str, ...]
537
+ files: Mapping[str, list[str]]
538
+ metrics: tuple[str, ...]
539
+ metadata: dict[str, object]
540
+
541
+ @property
542
+ def overview(self) -> pd.DataFrame:
543
+ relevant = self.global_test[
544
+ self.global_test["kind"].isin(["dimension", "overall"])
545
+ ]
546
+ labels = {
547
+ "dimension:textual": "Textual",
548
+ "dimension:syntactic_token_sequence": "Sintática",
549
+ "dimension:structural_ast": "Estrutural (AST TED)",
550
+ "overall": "Geral",
551
+ }
552
+ alpha = _metadata_alpha(self.metadata)
553
+ rows = []
554
+ for measure, row in relevant.iterrows():
555
+ rows.append({
556
+ "analysis": labels.get(measure, measure),
557
+ "homogeneity_p_holm": row["p_value_homogeneity_holm"],
558
+ "separation": row["separation"],
559
+ "separation_p_holm": row["p_value_separation_holm"],
560
+ "some_group_differs": bool(row["p_value_homogeneity_holm"] < alpha),
561
+ "evidence_of_separation": bool(
562
+ row["separation"] > 0 and row["p_value_separation_holm"] < alpha
563
+ ),
564
+ })
565
+ return pd.DataFrame(rows).set_index("analysis")
566
+
567
+ @property
568
+ def pairwise_overview(self) -> pd.DataFrame:
569
+ overall = self.pairwise.xs("overall", level="measure").reset_index()
570
+ alpha = _metadata_alpha(self.metadata)
571
+ overall["pair"] = overall["group_a"] + " × " + overall["group_b"]
572
+ overall["more_homogeneous"] = overall.apply(
573
+ lambda row: row["group_a"] if row["within_difference"] > 0
574
+ else row["group_b"] if row["within_difference"] < 0 else "tie",
575
+ axis=1,
576
+ )
577
+ overall["evidence_of_separation"] = (
578
+ (overall["separation"] > 0)
579
+ & (overall["p_value_separation_holm"] < alpha)
580
+ )
581
+ return overall.set_index("pair")[[
582
+ "more_homogeneous",
583
+ "within_difference",
584
+ "separation",
585
+ "p_value_homogeneity_holm",
586
+ "p_value_separation_holm",
587
+ "evidence_of_separation",
588
+ ]]
589
+
590
+ @property
591
+ def interpretation(self) -> str:
592
+ row = self.global_test.loc["overall"]
593
+ alpha = _metadata_alpha(self.metadata)
594
+ homogeneity = (
595
+ "Há evidência de que ao menos um grupo possui homogeneidade diferente"
596
+ if row["p_value_homogeneity_holm"] < alpha
597
+ else "Não há evidência suficiente de diferença global de homogeneidade"
598
+ )
599
+ separation = (
600
+ "há evidência de separação global entre os grupos"
601
+ if row["separation"] > 0 and row["p_value_separation_holm"] < alpha
602
+ else "não há evidência suficiente de separação global entre os grupos"
603
+ )
604
+ return (
605
+ f"{homogeneity} (p={row['p_value_homogeneity_holm']:.4g}); {separation} "
606
+ f"(separação={row['separation']:.4f}; "
607
+ f"p={row['p_value_separation_holm']:.4g}). "
608
+ "As comparações pareadas são pós-hoc e devem ser interpretadas em conjunto."
609
+ )
610
+
611
+ def print_report(self) -> None:
612
+ print("Teste global:")
613
+ print(self.overview.round(4))
614
+ print("\nComparações pareadas (resultado geral):")
615
+ print(self.pairwise_overview.round(4))
616
+ print(f"\n{self.interpretation}")
617
+
618
+ def export(self, directory: str | Path) -> None:
619
+ output = Path(directory)
620
+ output.mkdir(parents=True, exist_ok=True)
621
+ safe_to_csv(self.global_test, output / "global_test.csv", index_label="measure")
622
+ safe_to_csv(self.pairwise, output / "pairwise.csv")
623
+ safe_to_csv(self.within_groups, output / "within_groups.csv", index_label="group")
624
+ safe_to_csv(self.between_groups, output / "between_groups.csv")
625
+ payload = {
626
+ "group_names": list(self.group_names),
627
+ "files": self.files,
628
+ "metrics": list(self.metrics),
629
+ "metadata": self.metadata,
630
+ "global_test": frame_json(self.global_test.reset_index(), "records"),
631
+ "pairwise": frame_json(self.pairwise.reset_index(), "records"),
632
+ }
633
+ write_json(payload, output / "group_comparison.json")
634
+
635
+ def to_excel(self, path: str | Path) -> None:
636
+ target = Path(path)
637
+ target.parent.mkdir(parents=True, exist_ok=True)
638
+ with excel_writer(target) as writer:
639
+ safe_to_excel(self.overview, writer, sheet_name="Overview")
640
+ safe_to_excel(self.pairwise_overview,
641
+ writer, sheet_name="Overview", startrow=len(self.overview) + 3
642
+ )
643
+ safe_to_excel(pd.DataFrame({"interpretation": [self.interpretation]}),
644
+ writer,
645
+ sheet_name="Overview",
646
+ startrow=len(self.overview) + len(self.pairwise_overview) + 7,
647
+ index=False,
648
+ )
649
+ safe_to_excel(self.global_test, writer, sheet_name="Global tests")
650
+ safe_to_excel(self.pairwise, writer, sheet_name="Pairwise")
651
+ safe_to_excel(self.within_groups, writer, sheet_name="Within groups")
652
+ safe_to_excel(self.between_groups, writer, sheet_name="Between groups")
653
+ safe_to_excel(pd.DataFrame(
654
+ [(group, file) for group, files in self.files.items() for file in files],
655
+ columns=["group", "file"],
656
+ ), writer, sheet_name="Files", index=False)
657
+
658
+
659
+ def _compare_multiple_groups(
660
+ group_inputs: Mapping[str, str | Path | Sequence[str | Path]],
661
+ *,
662
+ metrics: str | Sequence[str],
663
+ permutations: int,
664
+ random_state: int,
665
+ include_ast: bool,
666
+ alpha: float,
667
+ progress: bool | Callable[[str], None],
668
+ cache_dir: str | Path | None,
669
+ ast_timeout: float,
670
+ max_ted_cells: int | None,
671
+ ) -> MultiGroupComparisonResult:
672
+ names = tuple(group_inputs)
673
+ if len(names) < 2 or any(not isinstance(name, str) or not name for name in names):
674
+ raise AnalysisError("Informe ao menos dois grupos com nomes não vazios.")
675
+ _validate_options(permutations, random_state, include_ast, alpha, progress, ast_timeout, max_ted_cells)
676
+ report = (
677
+ (lambda message: print(f"[codevariability] {message}", file=sys.stderr, flush=True))
678
+ if progress is True else progress if callable(progress) else None
679
+ )
680
+ if report is not None:
681
+ report(f"Carregando e validando {len(names)} grupos...")
682
+
683
+ datasets = {name: _dataset(value) for name, value in group_inputs.items()}
684
+ if any(len(dataset.files) < 2 for dataset in datasets.values()):
685
+ raise AnalysisError("Cada grupo deve conter pelo menos dois arquivos.")
686
+ _independent_files(datasets)
687
+
688
+ combined: dict[str, Path] = {}
689
+ public_files: dict[str, list[str]] = {}
690
+ index_groups: dict[str, list[int]] = {}
691
+ for name, dataset in datasets.items():
692
+ public_files[name] = list(dataset.files)
693
+ index_groups[name] = []
694
+ for file_name, path_value in dataset.files.items():
695
+ index_groups[name].append(len(combined))
696
+ combined[f"{name}/{file_name}"] = path_value
697
+ if report is not None:
698
+ report(f"Calculando métricas base para {len(combined)} arquivos...")
699
+ analysis = CodeDataset(combined).analyze(metrics, max_ted_cells=max_ted_cells)
700
+ javascript_ast_added = False
701
+ javascript_ast_cache: object = None
702
+ if include_ast and AST_TREE_EDIT_METRIC not in analysis.metrics:
703
+ if report is not None:
704
+ pairs = len(combined) * (len(combined) - 1) // 2
705
+ report(f"Calculando AST TED JavaScript: {pairs} pares únicos...")
706
+ metric, matrix = _javascript_ast_matrix(combined, report, cache_dir, ast_timeout, max_ted_cells)
707
+ _check_adapter_snapshot(analysis, matrix)
708
+ javascript_ast_cache = matrix.attrs.get("adapter_metadata", {}).get("cache")
709
+ analysis = analysis.with_metric(metric, matrix)
710
+ javascript_ast_added = True
711
+ measures = _measure_matrices(analysis)
712
+
713
+ global_rows: dict[str, dict[str, object]] = {}
714
+ within_values: dict[str, dict[str, float]] = {name: {} for name in names}
715
+ between_values: dict[tuple[str, str], dict[str, float]] = {
716
+ pair: {} for pair in combinations(names, 2)
717
+ }
718
+ pairwise_rows: dict[tuple[str, str, str], dict[str, object]] = {}
719
+ group_sizes = [len(index_groups[name]) for name in names]
720
+ total_steps = len(measures) * (1 + len(between_values))
721
+ step = 0
722
+ for measure, (kind, matrix) in measures.items():
723
+ step += 1
724
+ if report is not None:
725
+ report(f"Teste global para {measure} ({step}/{total_steps})...")
726
+ within, between, homogeneity, separation = _multi_group_statistics(matrix, index_groups)
727
+ p_homogeneity, p_separation = _global_permutation_p_values(
728
+ matrix, names, group_sizes, homogeneity, separation, permutations, random_state
729
+ )
730
+ global_rows[measure] = {
731
+ "kind": kind,
732
+ "mean_within": _mean(list(within.values())),
733
+ "mean_between": _mean(list(between.values())),
734
+ "homogeneity_statistic": homogeneity,
735
+ "separation": separation,
736
+ "p_value_homogeneity": p_homogeneity,
737
+ "p_value_separation": p_separation,
738
+ }
739
+ for name, value in within.items():
740
+ within_values[name][measure] = value
741
+ for pair, value in between.items():
742
+ between_values[pair][measure] = value
743
+
744
+ for left, right in combinations(names, 2):
745
+ step += 1
746
+ if report is not None:
747
+ report(f"Pós-teste {left} × {right}, {measure} ({step}/{total_steps})...")
748
+ stats = _group_statistics(matrix, index_groups[left], index_groups[right])
749
+ within_left, within_right, between_pair, difference, pair_separation, left_pairs, right_pairs = stats
750
+ pair_matrix_indices = [*index_groups[left], *index_groups[right]]
751
+ pair_matrix = matrix.iloc[pair_matrix_indices, pair_matrix_indices]
752
+ p_pair_h, p_pair_s = _permutation_p_values(
753
+ pair_matrix,
754
+ len(index_groups[left]),
755
+ difference,
756
+ pair_separation,
757
+ permutations,
758
+ random_state,
759
+ )
760
+ pairwise_rows[(left, right, measure)] = {
761
+ "kind": kind,
762
+ "within_group_a": within_left,
763
+ "within_group_b": within_right,
764
+ "between_groups": between_pair,
765
+ "within_difference": difference,
766
+ "separation": pair_separation,
767
+ "cliffs_delta_within": _cliffs_delta(left_pairs, right_pairs),
768
+ "p_value_homogeneity": p_pair_h,
769
+ "p_value_separation": p_pair_s,
770
+ }
771
+
772
+ global_test = pd.DataFrame.from_dict(global_rows, orient="index")
773
+ global_test.index.name = "measure"
774
+ for source, target in (
775
+ ("p_value_homogeneity", "p_value_homogeneity_holm"),
776
+ ("p_value_separation", "p_value_separation_holm"),
777
+ ):
778
+ global_test[target] = pd.Series(_holm(global_test[source].to_dict()))
779
+
780
+ pairwise = pd.DataFrame.from_dict(pairwise_rows, orient="index")
781
+ pairwise.index = pd.MultiIndex.from_tuples(
782
+ pairwise.index, names=["group_a", "group_b", "measure"]
783
+ )
784
+ for source, target in (
785
+ ("p_value_homogeneity", "p_value_homogeneity_holm"),
786
+ ("p_value_separation", "p_value_separation_holm"),
787
+ ):
788
+ raw = {key: float(value) for key, value in pairwise[source].items()}
789
+ pairwise[target] = pd.Series(_holm(raw))
790
+
791
+ dimensions = analysis.metadata.get("metric_dimensions", {})
792
+ result = MultiGroupComparisonResult(
793
+ global_test=global_test,
794
+ pairwise=pairwise,
795
+ within_groups=pd.DataFrame.from_dict(within_values, orient="index"),
796
+ between_groups=pd.DataFrame.from_dict(between_values, orient="index").rename_axis(
797
+ index=["group_a", "group_b"]
798
+ ),
799
+ group_names=names,
800
+ files=public_files,
801
+ metrics=analysis.metrics,
802
+ metadata={
803
+ **deepcopy(analysis.metadata),
804
+ "ast_timeout": ast_timeout,
805
+ "method": "multigroup_file_label_permutation_v1",
806
+ "alternative_homogeneity": "upper-tail squared dispersion of within-group means",
807
+ "alternative_separation": "two-sided",
808
+ "group_weighting": "equal",
809
+ "permutations": permutations,
810
+ "random_state": random_state,
811
+ "alpha": float(alpha),
812
+ "global_p_value_correction": "Holm across all reported measures, separately by hypothesis family",
813
+ "pairwise_p_value_correction": "Holm across every pair x measure, separately by hypothesis family",
814
+ "metric_ids": analysis.metadata.get("metric_ids", {}),
815
+ "metric_dimensions": dimensions,
816
+ "overall_aggregation": analysis.metadata.get("overall_aggregation", {}),
817
+ "javascript_ast_included": javascript_ast_added,
818
+ "javascript_ast_cache": javascript_ast_cache,
819
+ "cache_dir": str(Path(cache_dir).resolve()) if cache_dir is not None else None,
820
+ },
821
+ )
822
+ if report is not None:
823
+ report("Comparação multigrupo concluída.")
824
+ return result
825
+
826
+
827
+ def compare_groups(
828
+ group_a: Mapping[str, str | Path | Sequence[str | Path]] | str | Path | Sequence[str | Path],
829
+ group_b: str | Path | Sequence[str | Path] | None = None,
830
+ *,
831
+ metrics: str | Sequence[str] = "all",
832
+ group_names: tuple[str, str] = ("group_a", "group_b"),
833
+ permutations: int = 10_000,
834
+ random_state: int = 42,
835
+ include_ast: bool = False,
836
+ alpha: float = 0.05,
837
+ progress: bool | Callable[[str], None] = False,
838
+ cache_dir: str | Path | None = None,
839
+ ast_timeout: float = 120.0,
840
+ max_ted_cells: int | None = 2_000_000,
841
+ ) -> GroupComparisonResult | MultiGroupComparisonResult:
842
+ """Compare internal homogeneity and separation of independent groups.
843
+
844
+ P-values come from permutation of file-level group labels while preserving
845
+ the original group sizes. Pairwise matrix cells are never permuted as if
846
+ they were independent observations. Pass two inputs for a direct comparison
847
+ or an ordered mapping of names to inputs for a global multigroup analysis.
848
+ """
849
+ if isinstance(group_a, Mapping):
850
+ if group_b is not None:
851
+ raise AnalysisError("Não informe group_b ao usar o formato multigrupo.")
852
+ return _compare_multiple_groups(
853
+ group_a,
854
+ metrics=metrics,
855
+ permutations=permutations,
856
+ random_state=random_state,
857
+ include_ast=include_ast,
858
+ alpha=alpha,
859
+ progress=progress,
860
+ cache_dir=cache_dir,
861
+ ast_timeout=ast_timeout,
862
+ max_ted_cells=max_ted_cells,
863
+ )
864
+ if group_b is None:
865
+ raise AnalysisError("Informe o segundo grupo ou um mapeamento com dois ou mais grupos.")
866
+ if not isinstance(group_names, (tuple, list)) or len(group_names) != 2 or any(not isinstance(name, str) or not name.strip() for name in group_names) or group_names[0] == group_names[1]:
867
+ raise AnalysisError("Informe dois nomes de grupo distintos e não vazios.")
868
+ _validate_options(permutations, random_state, include_ast, alpha, progress, ast_timeout, max_ted_cells)
869
+ report: Callable[[str], None] | None
870
+ if progress is True:
871
+ def report(message: str) -> None:
872
+ print(f"[codevariability] {message}", file=sys.stderr, flush=True)
873
+ elif callable(progress):
874
+ report = progress
875
+ else:
876
+ report = None
877
+
878
+ if report is not None:
879
+ report("Carregando e validando os dois grupos...")
880
+
881
+ datasets = (_dataset(group_a), _dataset(group_b))
882
+ if any(len(dataset.files) < 2 for dataset in datasets):
883
+ raise AnalysisError("Cada grupo deve conter pelo menos dois arquivos.")
884
+ _independent_files(dict(zip(group_names, datasets, strict=True)))
885
+
886
+ combined: dict[str, Path] = {}
887
+ public_files: dict[str, list[str]] = {}
888
+ for group, dataset in zip(group_names, datasets, strict=True):
889
+ public_files[group] = list(dataset.files)
890
+ for name, path in dataset.files.items():
891
+ combined[f"{group}/{name}"] = path
892
+ if report is not None:
893
+ report(f"Calculando métricas base para {len(combined)} arquivos...")
894
+ analysis = CodeDataset(combined).analyze(metrics, max_ted_cells=max_ted_cells)
895
+ javascript_ast_added = False
896
+ javascript_ast_cache: object = None
897
+ if include_ast and AST_TREE_EDIT_METRIC not in analysis.metrics:
898
+ if report is not None:
899
+ pairs = len(combined) * (len(combined) - 1) // 2
900
+ report(f"Calculando AST TED JavaScript: {pairs} pares únicos...")
901
+ metric, matrix = _javascript_ast_matrix(combined, report, cache_dir, ast_timeout, max_ted_cells)
902
+ _check_adapter_snapshot(analysis, matrix)
903
+ javascript_ast_cache = matrix.attrs.get("adapter_metadata", {}).get("cache")
904
+ analysis = analysis.with_metric(metric, matrix)
905
+ javascript_ast_added = True
906
+
907
+ dimensions = analysis.metadata.get("metric_dimensions", {})
908
+ measures = _measure_matrices(analysis)
909
+
910
+ size_left = len(datasets[0].files)
911
+ left_indices = list(range(size_left))
912
+ right_indices = list(range(size_left, len(combined)))
913
+ rows: dict[str, dict[str, object]] = {}
914
+ for measure_index, (measure, (kind, matrix)) in enumerate(measures.items(), start=1):
915
+ if report is not None:
916
+ report(
917
+ f"Permutações para {measure} ({measure_index}/{len(measures)}; "
918
+ f"{permutations} repetições)..."
919
+ )
920
+ within_left, within_right, between, difference, separation, left_values, right_values = _group_statistics(
921
+ matrix, left_indices, right_indices
922
+ )
923
+ p_homogeneity, p_separation = _permutation_p_values(
924
+ matrix,
925
+ size_left,
926
+ difference,
927
+ separation,
928
+ permutations,
929
+ random_state,
930
+ )
931
+ rows[measure] = {
932
+ "kind": kind,
933
+ "within_group_a": within_left,
934
+ "within_group_b": within_right,
935
+ "between_groups": between,
936
+ "within_difference": difference,
937
+ "separation": separation,
938
+ "cliffs_delta_within": _cliffs_delta(left_values, right_values),
939
+ "p_value_homogeneity": p_homogeneity,
940
+ "p_value_separation": p_separation,
941
+ }
942
+ summary = pd.DataFrame.from_dict(rows, orient="index")
943
+ names = (group_names[0], group_names[1])
944
+ within_columns = _within_columns(names, set(summary.columns) - {"within_group_a", "within_group_b"})
945
+ summary = summary.rename(columns=dict(zip(("within_group_a", "within_group_b"), within_columns, strict=True)))
946
+ summary.index.name = "measure"
947
+ for source, target in (
948
+ ("p_value_homogeneity", "p_value_homogeneity_holm"),
949
+ ("p_value_separation", "p_value_separation_holm"),
950
+ ):
951
+ adjusted = _holm(summary[source].to_dict())
952
+ summary[target] = pd.Series(adjusted)
953
+
954
+ result = GroupComparisonResult(
955
+ summary=summary,
956
+ group_names=names,
957
+ files=public_files,
958
+ metrics=analysis.metrics,
959
+ metadata={
960
+ **deepcopy(analysis.metadata),
961
+ "within_group_columns": list(within_columns),
962
+ "ast_timeout": ast_timeout,
963
+ "method": "file_label_permutation_v1",
964
+ "alternative": "two-sided",
965
+ "permutations": permutations,
966
+ "random_state": random_state,
967
+ "alpha": float(alpha),
968
+ "javascript_ast_included": javascript_ast_added,
969
+ "javascript_ast_cache": javascript_ast_cache,
970
+ "cache_dir": str(Path(cache_dir).resolve()) if cache_dir is not None else None,
971
+ "p_value_correction": "Holm separately by hypothesis family",
972
+ "metric_ids": analysis.metadata.get("metric_ids", {}),
973
+ "metric_dimensions": dimensions,
974
+ "overall_aggregation": analysis.metadata.get("overall_aggregation", {}),
975
+ },
976
+ )
977
+ if report is not None:
978
+ report("Comparação concluída.")
979
+ return result