codevariability 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codevariability/__init__.py +23 -0
- codevariability/analysis.py +544 -0
- codevariability/ast_tree_edit.py +350 -0
- codevariability/cli.py +75 -0
- codevariability/exceptions.py +2 -0
- codevariability/group_comparison.py +979 -0
- codevariability/interop.py +94 -0
- codevariability/metrics.py +89 -0
- codevariability/normalization.py +257 -0
- codevariability/output.py +76 -0
- codevariability/spreadsheet.py +68 -0
- codevariability/validation.py +115 -0
- codevariability-0.2.0.dist-info/METADATA +171 -0
- codevariability-0.2.0.dist-info/RECORD +18 -0
- codevariability-0.2.0.dist-info/WHEEL +5 -0
- codevariability-0.2.0.dist-info/entry_points.txt +2 -0
- codevariability-0.2.0.dist-info/licenses/LICENSE +21 -0
- codevariability-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Interoperability helpers for language-specific analysis adapters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from .exceptions import AnalysisError
|
|
11
|
+
from .validation import metric_name, similarity_matrix
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _unique_keys(items):
|
|
15
|
+
result = {}
|
|
16
|
+
for key, value in items:
|
|
17
|
+
if key in result:
|
|
18
|
+
raise AnalysisError(f"Chave JSON duplicada: {key}.")
|
|
19
|
+
result[key] = value
|
|
20
|
+
return result
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def load_matrix_json(path: str | Path) -> tuple[str, pd.DataFrame]:
|
|
24
|
+
"""Load the JSON matrix schema emitted by a codevariability language adapter."""
|
|
25
|
+
try:
|
|
26
|
+
payload = json.loads(
|
|
27
|
+
Path(path).read_text(encoding="utf-8"), object_pairs_hook=_unique_keys
|
|
28
|
+
)
|
|
29
|
+
except (OSError, UnicodeError, ValueError, RecursionError, TypeError) as exc:
|
|
30
|
+
raise AnalysisError(
|
|
31
|
+
f"Não foi possível ler o JSON de métrica externa: {exc}."
|
|
32
|
+
) from exc
|
|
33
|
+
if not isinstance(payload, dict):
|
|
34
|
+
raise AnalysisError(
|
|
35
|
+
"JSON de métrica externa inválido ou versão de esquema não suportada."
|
|
36
|
+
)
|
|
37
|
+
required = {"schema_version", "metric", "files", "matrix"}
|
|
38
|
+
missing = required - set(payload)
|
|
39
|
+
if missing or payload.get("schema_version") != "codevariability.matrix.v1":
|
|
40
|
+
raise AnalysisError(
|
|
41
|
+
"JSON de métrica externa inválido ou versão de esquema não suportada."
|
|
42
|
+
)
|
|
43
|
+
files = payload["files"]
|
|
44
|
+
if (
|
|
45
|
+
not isinstance(files, list)
|
|
46
|
+
or not files
|
|
47
|
+
or not all(isinstance(name, str) and name for name in files)
|
|
48
|
+
or len(files) != len(set(files))
|
|
49
|
+
):
|
|
50
|
+
raise AnalysisError(
|
|
51
|
+
"O JSON de métrica externa deve declarar nomes de arquivo únicos."
|
|
52
|
+
)
|
|
53
|
+
metric = metric_name(payload["metric"])
|
|
54
|
+
matrix = payload["matrix"]
|
|
55
|
+
if (
|
|
56
|
+
not isinstance(metric, str)
|
|
57
|
+
or not metric
|
|
58
|
+
or not isinstance(matrix, list)
|
|
59
|
+
or len(matrix) != len(files)
|
|
60
|
+
or any(not isinstance(row, list) or len(row) != len(files) for row in matrix)
|
|
61
|
+
):
|
|
62
|
+
raise AnalysisError(
|
|
63
|
+
"O JSON de métrica externa deve conter uma métrica e uma matriz quadrada compatível com os arquivos."
|
|
64
|
+
)
|
|
65
|
+
try:
|
|
66
|
+
frame = similarity_matrix(
|
|
67
|
+
pd.DataFrame(matrix, index=files, columns=files, dtype=object), files
|
|
68
|
+
)
|
|
69
|
+
except (TypeError, ValueError, OverflowError) as exc:
|
|
70
|
+
raise AnalysisError(
|
|
71
|
+
"A matriz externa deve conter somente números reais finitos."
|
|
72
|
+
) from exc
|
|
73
|
+
frame.attrs["metric_id"] = payload.get("metric_id")
|
|
74
|
+
adapter_metadata = payload.get("metadata")
|
|
75
|
+
if adapter_metadata is not None and not isinstance(adapter_metadata, dict):
|
|
76
|
+
raise AnalysisError("Os metadados do adaptador devem ser um objeto JSON.")
|
|
77
|
+
if frame.attrs["metric_id"] is not None and (
|
|
78
|
+
not isinstance(frame.attrs["metric_id"], str)
|
|
79
|
+
or not frame.attrs["metric_id"].strip()
|
|
80
|
+
):
|
|
81
|
+
raise AnalysisError("metric_id deve ser texto não vazio.")
|
|
82
|
+
if isinstance(adapter_metadata, dict):
|
|
83
|
+
frame.attrs["adapter_metadata"] = adapter_metadata
|
|
84
|
+
frame.attrs["normalization"] = adapter_metadata.get("normalization")
|
|
85
|
+
dimensions = adapter_metadata.get("metric_dimensions")
|
|
86
|
+
if isinstance(dimensions, dict):
|
|
87
|
+
frame.attrs["dimension"] = dimensions.get(metric)
|
|
88
|
+
elif dimensions is not None:
|
|
89
|
+
raise AnalysisError("metric_dimensions deve ser um objeto JSON.")
|
|
90
|
+
for attribute in ("normalization", "dimension"):
|
|
91
|
+
value = frame.attrs.get(attribute)
|
|
92
|
+
if value is not None and (not isinstance(value, str) or not value.strip()):
|
|
93
|
+
raise AnalysisError(f"{attribute} deve ser texto não vazio.")
|
|
94
|
+
return metric, frame
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Normalized pairwise similarity metrics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from itertools import combinations
|
|
6
|
+
from typing import Callable, Mapping, Sequence
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
from rapidfuzz.distance import LCSseq, Levenshtein
|
|
10
|
+
from sklearn.feature_extraction.text import CountVectorizer
|
|
11
|
+
from sklearn.metrics.pairwise import cosine_similarity
|
|
12
|
+
|
|
13
|
+
from .ast_tree_edit import (
|
|
14
|
+
AST_TREE_EDIT_METRIC,
|
|
15
|
+
AST_TREE_EDIT_METRIC_ID,
|
|
16
|
+
ast_tree_edit_similarity,
|
|
17
|
+
)
|
|
18
|
+
from .normalization import code_tokens, text_tokens
|
|
19
|
+
|
|
20
|
+
DEFAULT_METRICS = ("cosine", "jaccard", "lcs", "levenshtein")
|
|
21
|
+
METRICS = (*DEFAULT_METRICS, AST_TREE_EDIT_METRIC)
|
|
22
|
+
METRIC_IDS = {
|
|
23
|
+
"cosine": "cosine_bag_of_words_text_v1",
|
|
24
|
+
"jaccard": "jaccard_word_set_text_v1",
|
|
25
|
+
"lcs": "lcs_code_tokens_v1",
|
|
26
|
+
"levenshtein": "levenshtein_code_tokens_v1",
|
|
27
|
+
AST_TREE_EDIT_METRIC: AST_TREE_EDIT_METRIC_ID,
|
|
28
|
+
}
|
|
29
|
+
METRIC_DIMENSIONS = {
|
|
30
|
+
"cosine": "textual",
|
|
31
|
+
"jaccard": "textual",
|
|
32
|
+
"lcs": "syntactic_token_sequence",
|
|
33
|
+
"levenshtein": "syntactic_token_sequence",
|
|
34
|
+
AST_TREE_EDIT_METRIC: "structural_ast",
|
|
35
|
+
}
|
|
36
|
+
def _empty_matrix(names: Sequence[str]) -> pd.DataFrame:
|
|
37
|
+
matrix = pd.DataFrame(0.0, index=names, columns=names, dtype=float)
|
|
38
|
+
for name in names:
|
|
39
|
+
matrix.loc[name, name] = 1.0
|
|
40
|
+
return matrix
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def cosine(sources: Mapping[str, str]) -> pd.DataFrame:
|
|
44
|
+
names = list(sources)
|
|
45
|
+
corpus = [" ".join(text_tokens(sources[name])) for name in names]
|
|
46
|
+
if not any(corpus):
|
|
47
|
+
return _empty_matrix(names)
|
|
48
|
+
vectors = CountVectorizer(token_pattern=r"(?u)\b\w+\b").fit_transform(corpus)
|
|
49
|
+
return pd.DataFrame(cosine_similarity(vectors), index=names, columns=names)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def jaccard(sources: Mapping[str, str]) -> pd.DataFrame:
|
|
53
|
+
names = list(sources)
|
|
54
|
+
sets = {name: set(text_tokens(text)) for name, text in sources.items()}
|
|
55
|
+
matrix = _empty_matrix(names)
|
|
56
|
+
for left, right in combinations(names, 2):
|
|
57
|
+
union = sets[left] | sets[right]
|
|
58
|
+
value = len(sets[left] & sets[right]) / len(union) if union else 1.0
|
|
59
|
+
matrix.loc[left, right] = matrix.loc[right, left] = value
|
|
60
|
+
return matrix
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def lcs(sources: Mapping[str, str]) -> pd.DataFrame:
|
|
64
|
+
names = list(sources)
|
|
65
|
+
processed = {name: code_tokens(text, name) for name, text in sources.items()}
|
|
66
|
+
matrix = _empty_matrix(names)
|
|
67
|
+
for left, right in combinations(names, 2):
|
|
68
|
+
value = LCSseq.normalized_similarity(processed[left], processed[right])
|
|
69
|
+
matrix.loc[left, right] = matrix.loc[right, left] = value
|
|
70
|
+
return matrix
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def levenshtein(sources: Mapping[str, str]) -> pd.DataFrame:
|
|
74
|
+
names = list(sources)
|
|
75
|
+
processed = {name: code_tokens(text, name) for name, text in sources.items()}
|
|
76
|
+
matrix = _empty_matrix(names)
|
|
77
|
+
for left, right in combinations(names, 2):
|
|
78
|
+
value = Levenshtein.normalized_similarity(processed[left], processed[right])
|
|
79
|
+
matrix.loc[left, right] = matrix.loc[right, left] = value
|
|
80
|
+
return matrix
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
CALCULATORS: dict[str, Callable[..., pd.DataFrame]] = {
|
|
84
|
+
"cosine": cosine,
|
|
85
|
+
"jaccard": jaccard,
|
|
86
|
+
"lcs": lcs,
|
|
87
|
+
"levenshtein": levenshtein,
|
|
88
|
+
AST_TREE_EDIT_METRIC: ast_tree_edit_similarity,
|
|
89
|
+
}
|
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""Reproducible preprocessing for textual and code-token analyses."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import unicodedata
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from pygments import lex
|
|
11
|
+
from pygments.lexers import get_lexer_by_name, get_lexer_for_filename
|
|
12
|
+
from pygments.token import Comment, Operator, Text
|
|
13
|
+
from pygments.util import ClassNotFound
|
|
14
|
+
|
|
15
|
+
TEXT_NORMALIZATION = "unicode_words_casefold_v1"
|
|
16
|
+
CODE_TOKENIZATION = "pygments_lexemes_with_generic_fallback_v2"
|
|
17
|
+
|
|
18
|
+
_WORD = re.compile(r"(?u)\b\w+\b")
|
|
19
|
+
_GENERIC_CODE_TOKEN = re.compile(
|
|
20
|
+
r"""(?x)
|
|
21
|
+
(?:0[xX][0-9a-fA-F]+|0[bB][01]+|0[oO][0-7]+|\d+(?:\.\d+)?(?:[eE][+-]?\d+)?)
|
|
22
|
+
|(?:[^\W\d]\w*)
|
|
23
|
+
|(?:===|!==|>>>|<<=|>>=|\*\*=|//=|\?\?=|\*\*|//|==|!=|<=|>=|&&|\|\||\?\?|\?\.|=>|->|::|\.\.\.|\+\+|--|\+=|-=|\*=|/=|%=|&=|\|=|\^=|:=|<<|>>)
|
|
24
|
+
|(?:[^\s\w])
|
|
25
|
+
""",
|
|
26
|
+
flags=re.UNICODE,
|
|
27
|
+
)
|
|
28
|
+
_COMPOUND_OPERATORS = {
|
|
29
|
+
"===",
|
|
30
|
+
"!==",
|
|
31
|
+
">>>",
|
|
32
|
+
"<<=",
|
|
33
|
+
">>=",
|
|
34
|
+
"**",
|
|
35
|
+
"//",
|
|
36
|
+
"==",
|
|
37
|
+
"!=",
|
|
38
|
+
"<=",
|
|
39
|
+
">=",
|
|
40
|
+
"**=",
|
|
41
|
+
"//=",
|
|
42
|
+
"??=",
|
|
43
|
+
"&&",
|
|
44
|
+
"||",
|
|
45
|
+
"??",
|
|
46
|
+
"?.",
|
|
47
|
+
"=>",
|
|
48
|
+
"->",
|
|
49
|
+
"::",
|
|
50
|
+
"...",
|
|
51
|
+
"++",
|
|
52
|
+
"--",
|
|
53
|
+
"+=",
|
|
54
|
+
"-=",
|
|
55
|
+
"*=",
|
|
56
|
+
"/=",
|
|
57
|
+
"%=",
|
|
58
|
+
"&=",
|
|
59
|
+
"|=",
|
|
60
|
+
"^=",
|
|
61
|
+
":=",
|
|
62
|
+
"<<",
|
|
63
|
+
">>",
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True, slots=True)
|
|
68
|
+
class CodeSegment:
|
|
69
|
+
"""A code segment and its optional Markdown language identifier."""
|
|
70
|
+
|
|
71
|
+
code: str
|
|
72
|
+
language: str | None = None
|
|
73
|
+
start_line: int = 1
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def normalize_text(text: str) -> str:
|
|
77
|
+
"""Apply only language-independent textual normalization."""
|
|
78
|
+
|
|
79
|
+
normalized = (
|
|
80
|
+
unicodedata.normalize("NFC", text).replace("\r\n", "\n").replace("\r", "\n")
|
|
81
|
+
)
|
|
82
|
+
return " ".join(normalized.casefold().split())
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def text_tokens(text: str) -> list[str]:
|
|
86
|
+
"""Return case-folded Unicode words from the complete response."""
|
|
87
|
+
|
|
88
|
+
return _WORD.findall(normalize_text(text))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def fenced_code_segments(text: str) -> tuple[CodeSegment, ...]:
|
|
92
|
+
"""Scan closed backtick/tilde fences once, preserving code and line numbers.
|
|
93
|
+
|
|
94
|
+
Only a complete delimiter line closes a fence. Unterminated fences are
|
|
95
|
+
rejected, so malformed Markdown cannot silently become an empty program.
|
|
96
|
+
"""
|
|
97
|
+
from .exceptions import AnalysisError
|
|
98
|
+
|
|
99
|
+
result: list[CodeSegment] = []
|
|
100
|
+
delimiter: str | None = None
|
|
101
|
+
width = 0
|
|
102
|
+
language: str | None = None
|
|
103
|
+
start_line = 1
|
|
104
|
+
code: list[str] = []
|
|
105
|
+
for number, line in enumerate(text.splitlines(keepends=True), 1):
|
|
106
|
+
stripped = line.strip(" \t\r\n")
|
|
107
|
+
if delimiter is not None:
|
|
108
|
+
if (
|
|
109
|
+
stripped
|
|
110
|
+
and stripped[0] == delimiter
|
|
111
|
+
and len(stripped) >= width
|
|
112
|
+
and not stripped.strip(delimiter)
|
|
113
|
+
):
|
|
114
|
+
result.append(CodeSegment("".join(code), language, start_line))
|
|
115
|
+
delimiter = None
|
|
116
|
+
code = []
|
|
117
|
+
else:
|
|
118
|
+
code.append(line)
|
|
119
|
+
continue
|
|
120
|
+
candidate = line.lstrip(" \t")
|
|
121
|
+
if not candidate or candidate[0] not in {"`", "~"}:
|
|
122
|
+
continue
|
|
123
|
+
marker = candidate[0]
|
|
124
|
+
count = len(candidate) - len(candidate.lstrip(marker))
|
|
125
|
+
if count < 3:
|
|
126
|
+
continue
|
|
127
|
+
info = candidate[count:].strip()
|
|
128
|
+
if marker == "`" and "`" in info:
|
|
129
|
+
continue
|
|
130
|
+
delimiter, width = marker, count
|
|
131
|
+
language = info.split(maxsplit=1)[0] if info else None
|
|
132
|
+
start_line = number + 1
|
|
133
|
+
if delimiter is not None:
|
|
134
|
+
raise AnalysisError(f"Fence Markdown sem fechamento (linha {start_line - 1}).")
|
|
135
|
+
return tuple(result)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def code_segments(text: str) -> tuple[CodeSegment, ...]:
|
|
139
|
+
"""Extract closed fences, or retain the complete unmodified source."""
|
|
140
|
+
segments = fenced_code_segments(text)
|
|
141
|
+
return segments or (CodeSegment(text),)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _lexer(filename: str, language: str | None):
|
|
145
|
+
try:
|
|
146
|
+
if language:
|
|
147
|
+
return get_lexer_by_name(language, stripnl=False, ensurenl=False)
|
|
148
|
+
return get_lexer_for_filename(filename, stripnl=False, ensurenl=False)
|
|
149
|
+
except ClassNotFound:
|
|
150
|
+
return None
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _generic_tokens(code: str) -> list[str]:
|
|
154
|
+
# Resolve closing quotes once, backwards. Repeated failed searches from
|
|
155
|
+
# escaped quotes in an unterminated string must not rescan every suffix.
|
|
156
|
+
# next_end[q] is the first legal closing q from the next character;
|
|
157
|
+
# after_next[q] handles the two-character escape branch of the old grammar.
|
|
158
|
+
quotes = ('"', "'", '`')
|
|
159
|
+
next_end = [-1, -1, -1]
|
|
160
|
+
after_next = next_end
|
|
161
|
+
endings: dict[int, int] = {}
|
|
162
|
+
for index in range(len(code) - 1, -1, -1):
|
|
163
|
+
character = code[index]
|
|
164
|
+
current = []
|
|
165
|
+
for position, quote in enumerate(quotes):
|
|
166
|
+
if character == quote:
|
|
167
|
+
endings[index] = next_end[position]
|
|
168
|
+
current.append(index)
|
|
169
|
+
elif character == "\\":
|
|
170
|
+
current.append(after_next[position] if index + 1 < len(code) and code[index + 1] != "\n" else -1)
|
|
171
|
+
else:
|
|
172
|
+
current.append(next_end[position])
|
|
173
|
+
after_next, next_end = next_end, current
|
|
174
|
+
tokens = []
|
|
175
|
+
index = 0
|
|
176
|
+
while index < len(code):
|
|
177
|
+
end = endings.get(index, -1)
|
|
178
|
+
if end >= 0:
|
|
179
|
+
tokens.append(code[index:end + 1])
|
|
180
|
+
index = end + 1
|
|
181
|
+
else:
|
|
182
|
+
match = _GENERIC_CODE_TOKEN.match(code, index)
|
|
183
|
+
if match is None:
|
|
184
|
+
index += 1
|
|
185
|
+
else:
|
|
186
|
+
tokens.append(match.group(0))
|
|
187
|
+
index = match.end()
|
|
188
|
+
return tokens
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _pygments_tokens(code: str, lexer) -> list[str]:
|
|
192
|
+
result: list[str] = []
|
|
193
|
+
adjacent_operator = False
|
|
194
|
+
for token_type, value in lex(code, lexer):
|
|
195
|
+
if token_type in Comment or not value:
|
|
196
|
+
adjacent_operator = False
|
|
197
|
+
continue
|
|
198
|
+
if value.isspace():
|
|
199
|
+
adjacent_operator = False
|
|
200
|
+
continue
|
|
201
|
+
if token_type in Operator:
|
|
202
|
+
candidate = f"{result[-1]}{value}" if adjacent_operator else value
|
|
203
|
+
if adjacent_operator and candidate in _COMPOUND_OPERATORS:
|
|
204
|
+
result[-1] = candidate
|
|
205
|
+
else:
|
|
206
|
+
result.append(value)
|
|
207
|
+
adjacent_operator = True
|
|
208
|
+
continue
|
|
209
|
+
adjacent_operator = False
|
|
210
|
+
if token_type in Text:
|
|
211
|
+
result.extend(_generic_tokens(value))
|
|
212
|
+
else:
|
|
213
|
+
result.append(value.replace("\r\n", "\n").replace("\r", "\n"))
|
|
214
|
+
return result
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def code_tokens(text: str, filename: str) -> list[str]:
|
|
218
|
+
"""Return ordered code lexemes, excluding comments and whitespace.
|
|
219
|
+
|
|
220
|
+
Pygments supplies language lexers when the filename or Markdown fence has
|
|
221
|
+
a known language. Unknown languages use the documented generic tokenizer;
|
|
222
|
+
this fallback is deterministic but is not presented as a compiler lexer.
|
|
223
|
+
Fragment boundary markers prevent matches from crossing code blocks.
|
|
224
|
+
"""
|
|
225
|
+
|
|
226
|
+
segments = (
|
|
227
|
+
fenced_code_segments(text)
|
|
228
|
+
if Path(filename).suffix.lower() in {".md", ".markdown"}
|
|
229
|
+
else (CodeSegment(text),)
|
|
230
|
+
)
|
|
231
|
+
result: list[str] = []
|
|
232
|
+
for segment in segments:
|
|
233
|
+
lexer = _lexer(filename, segment.language)
|
|
234
|
+
tokens = (
|
|
235
|
+
_pygments_tokens(segment.code, lexer)
|
|
236
|
+
if lexer
|
|
237
|
+
else _generic_tokens(segment.code)
|
|
238
|
+
)
|
|
239
|
+
if not tokens:
|
|
240
|
+
continue
|
|
241
|
+
if result:
|
|
242
|
+
result.append("<CODE_FRAGMENT_BOUNDARY>")
|
|
243
|
+
result.extend(tokens)
|
|
244
|
+
return result
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def uses_python_code(text: str, filename: str) -> bool:
|
|
248
|
+
"""Route .py source or explicitly Python Markdown fences to the parser."""
|
|
249
|
+
suffix = Path(filename).suffix.lower()
|
|
250
|
+
if suffix == ".py":
|
|
251
|
+
return True
|
|
252
|
+
if suffix not in {".md", ".markdown"}:
|
|
253
|
+
return False
|
|
254
|
+
segments = fenced_code_segments(text)
|
|
255
|
+
return bool(segments) and all(
|
|
256
|
+
(segment.language or "").casefold() in {"py", "python"} for segment in segments
|
|
257
|
+
)
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Atomic output files: never follow an existing destination symlink."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import tempfile
|
|
8
|
+
from contextlib import contextmanager, suppress
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
from .exceptions import AnalysisError
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@contextmanager
|
|
17
|
+
def atomic_target(path: str | Path):
|
|
18
|
+
target = Path(path)
|
|
19
|
+
if target.is_symlink():
|
|
20
|
+
raise AnalysisError(
|
|
21
|
+
f"O destino de exportação não pode ser um link simbólico: {target}."
|
|
22
|
+
)
|
|
23
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
24
|
+
descriptor, name = tempfile.mkstemp(
|
|
25
|
+
prefix=".codevariability-", suffix=target.suffix, dir=target.parent
|
|
26
|
+
)
|
|
27
|
+
os.close(descriptor)
|
|
28
|
+
temporary = Path(name)
|
|
29
|
+
try:
|
|
30
|
+
yield temporary
|
|
31
|
+
if target.is_symlink():
|
|
32
|
+
raise AnalysisError(
|
|
33
|
+
f"O destino de exportação tornou-se um link simbólico: {target}."
|
|
34
|
+
)
|
|
35
|
+
os.replace(temporary, target)
|
|
36
|
+
finally:
|
|
37
|
+
temporary.unlink(missing_ok=True)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def frame_json(frame: pd.DataFrame, orient: str) -> object:
|
|
41
|
+
"""Preserve pandas' null conventions, reporting serialization errors uniformly."""
|
|
42
|
+
try:
|
|
43
|
+
return json.loads(frame.to_json(orient=orient))
|
|
44
|
+
except (TypeError, ValueError, OverflowError, RecursionError) as exc:
|
|
45
|
+
raise AnalysisError(
|
|
46
|
+
f"A tabela não pode ser serializada como JSON: {exc}."
|
|
47
|
+
) from exc
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def write_json(payload: object, path: str | Path) -> None:
|
|
51
|
+
try:
|
|
52
|
+
content = json.dumps(
|
|
53
|
+
payload, ensure_ascii=False, indent=2, allow_nan=False
|
|
54
|
+
).encode("utf-8")
|
|
55
|
+
except (TypeError, ValueError, OverflowError, RecursionError) as exc:
|
|
56
|
+
raise AnalysisError(
|
|
57
|
+
f"O resultado não pode ser serializado como JSON: {exc}."
|
|
58
|
+
) from exc
|
|
59
|
+
with atomic_target(path) as temporary:
|
|
60
|
+
temporary.write_bytes(content)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@contextmanager
|
|
64
|
+
def excel_writer(path: str | Path):
|
|
65
|
+
with atomic_target(path) as temporary:
|
|
66
|
+
# The outer handle closes even if openpyxl cannot save a failed workbook.
|
|
67
|
+
with temporary.open("w+b") as handle:
|
|
68
|
+
writer = pd.ExcelWriter(handle, engine="openpyxl")
|
|
69
|
+
try:
|
|
70
|
+
yield writer
|
|
71
|
+
except BaseException:
|
|
72
|
+
with suppress(Exception):
|
|
73
|
+
writer.close()
|
|
74
|
+
raise
|
|
75
|
+
else:
|
|
76
|
+
writer.close()
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Escape untrusted labels when exporting tables opened by spreadsheet apps."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
from .exceptions import AnalysisError
|
|
12
|
+
from .output import atomic_target
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _safe_cell(value: object, prefix: str) -> object:
|
|
16
|
+
if (
|
|
17
|
+
prefix == "'"
|
|
18
|
+
and isinstance(value, str)
|
|
19
|
+
and re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f\ud800-\udfff\ufffe\uffff]", value)
|
|
20
|
+
):
|
|
21
|
+
raise AnalysisError(
|
|
22
|
+
"O texto exportado contém caracteres incompatíveis com XML do Excel."
|
|
23
|
+
)
|
|
24
|
+
if isinstance(value, str) and value.lstrip().startswith(("=", "+", "-", "@")):
|
|
25
|
+
return prefix + value
|
|
26
|
+
return value
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _safe_index(index: pd.Index, prefix: str) -> pd.Index:
|
|
30
|
+
if isinstance(index, pd.MultiIndex):
|
|
31
|
+
return pd.MultiIndex.from_tuples(
|
|
32
|
+
[tuple(_safe_cell(part, prefix) for part in label) for label in index],
|
|
33
|
+
names=[_safe_cell(name, prefix) for name in index.names],
|
|
34
|
+
)
|
|
35
|
+
return pd.Index(
|
|
36
|
+
[_safe_cell(label, prefix) for label in index],
|
|
37
|
+
name=_safe_cell(index.name, prefix),
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _safe_frame(frame: pd.DataFrame, prefix: str) -> pd.DataFrame:
|
|
42
|
+
safe = frame.copy()
|
|
43
|
+
for column in safe.columns:
|
|
44
|
+
if safe[column].dtype == object or pd.api.types.is_string_dtype(
|
|
45
|
+
safe[column].dtype
|
|
46
|
+
):
|
|
47
|
+
safe[column] = safe[column].map(lambda value: _safe_cell(value, prefix))
|
|
48
|
+
safe.index = _safe_index(safe.index, prefix)
|
|
49
|
+
safe.columns = _safe_index(safe.columns, prefix)
|
|
50
|
+
return safe
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def safe_to_csv(frame: pd.DataFrame, path: str | Path, **kwargs: object) -> None:
|
|
54
|
+
kwargs.setdefault("quoting", csv.QUOTE_ALL)
|
|
55
|
+
safe = _safe_frame(frame, "\t")
|
|
56
|
+
with atomic_target(path) as temporary:
|
|
57
|
+
try:
|
|
58
|
+
safe.to_csv(temporary, **kwargs)
|
|
59
|
+
except UnicodeError as exc:
|
|
60
|
+
raise AnalysisError(
|
|
61
|
+
"O texto exportado como CSV deve ser representável em UTF-8."
|
|
62
|
+
) from exc
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def safe_to_excel(
|
|
66
|
+
frame: pd.DataFrame, writer: pd.ExcelWriter, **kwargs: object
|
|
67
|
+
) -> None:
|
|
68
|
+
_safe_frame(frame, "'").to_excel(writer, **kwargs)
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""Shared validation for public analysis and adapter inputs."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping, Sequence
|
|
6
|
+
from copy import deepcopy
|
|
7
|
+
from numbers import Real
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import pandas as pd
|
|
11
|
+
|
|
12
|
+
from .exceptions import AnalysisError
|
|
13
|
+
|
|
14
|
+
DIMENSION_SCORE_COLUMNS = {
|
|
15
|
+
"textual": "textual_score",
|
|
16
|
+
"syntactic_token_sequence": "syntactic_score",
|
|
17
|
+
"structural_ast": "structural_score",
|
|
18
|
+
}
|
|
19
|
+
RESERVED_COLUMNS = {
|
|
20
|
+
"file",
|
|
21
|
+
"rank",
|
|
22
|
+
"score",
|
|
23
|
+
"overall_score",
|
|
24
|
+
"overall_dispersion",
|
|
25
|
+
*DIMENSION_SCORE_COLUMNS.values(),
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def metric_name(value: object) -> str:
|
|
30
|
+
if not isinstance(value, str) or not value.strip() or value in RESERVED_COLUMNS:
|
|
31
|
+
raise AnalysisError(
|
|
32
|
+
"O nome da métrica deve ser texto não vazio e não pode ser uma coluna reservada."
|
|
33
|
+
)
|
|
34
|
+
try:
|
|
35
|
+
value.encode("utf-8")
|
|
36
|
+
except UnicodeError as exc:
|
|
37
|
+
raise AnalysisError(
|
|
38
|
+
"O nome da métrica deve ser representável em UTF-8."
|
|
39
|
+
) from exc
|
|
40
|
+
return value
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def dimensions(metrics: Sequence[str], metadata: Mapping[str, object]) -> None:
|
|
44
|
+
for field in ("metric_ids", "metric_normalizations"):
|
|
45
|
+
values = metadata.get(field, {})
|
|
46
|
+
if not isinstance(values, Mapping) or any(
|
|
47
|
+
not isinstance(value, str) or not value.strip() for value in values.values()
|
|
48
|
+
):
|
|
49
|
+
raise AnalysisError(
|
|
50
|
+
f"{field} deve mapear métricas para strings não vazias."
|
|
51
|
+
)
|
|
52
|
+
if not isinstance(metadata.get("external_adapters", {}), Mapping):
|
|
53
|
+
raise AnalysisError("external_adapters deve ser um mapeamento.")
|
|
54
|
+
declared = metadata.get("metric_dimensions", {})
|
|
55
|
+
if not isinstance(declared, Mapping):
|
|
56
|
+
raise AnalysisError("metric_dimensions deve ser um mapeamento.")
|
|
57
|
+
columns: dict[str, str] = {}
|
|
58
|
+
for metric in metrics:
|
|
59
|
+
dimension = declared.get(metric, metric)
|
|
60
|
+
if not isinstance(dimension, str) or not dimension.strip():
|
|
61
|
+
raise AnalysisError("A dimensão de cada métrica deve ser texto não vazio.")
|
|
62
|
+
column = DIMENSION_SCORE_COLUMNS.get(dimension, f"{dimension}_score")
|
|
63
|
+
if (
|
|
64
|
+
column in {"file", "rank", "score", "overall_score", "overall_dispersion"}
|
|
65
|
+
or column in metrics
|
|
66
|
+
):
|
|
67
|
+
raise AnalysisError(
|
|
68
|
+
f"A dimensão '{dimension}' colide com uma coluna reservada ou métrica."
|
|
69
|
+
)
|
|
70
|
+
if column in columns and columns[column] != dimension:
|
|
71
|
+
raise AnalysisError(
|
|
72
|
+
"Dimensões distintas não podem gerar a mesma coluna de score."
|
|
73
|
+
)
|
|
74
|
+
columns[column] = dimension
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def similarity_matrix(matrix: object, files: Sequence[str]) -> pd.DataFrame:
|
|
78
|
+
if not isinstance(matrix, pd.DataFrame):
|
|
79
|
+
raise AnalysisError("A matriz de similaridade deve ser um pandas.DataFrame.")
|
|
80
|
+
if not matrix.index.is_unique or not matrix.columns.is_unique:
|
|
81
|
+
raise AnalysisError(
|
|
82
|
+
"Índices e colunas da matriz externa não podem conter nomes duplicados."
|
|
83
|
+
)
|
|
84
|
+
if (
|
|
85
|
+
list(matrix.shape) != [len(files), len(files)]
|
|
86
|
+
or set(matrix.index) != set(files)
|
|
87
|
+
or set(matrix.columns) != set(files)
|
|
88
|
+
):
|
|
89
|
+
raise AnalysisError(
|
|
90
|
+
"Índices e colunas da matriz externa devem corresponder exatamente aos arquivos analisados."
|
|
91
|
+
)
|
|
92
|
+
if any(
|
|
93
|
+
not isinstance(value, Real) or isinstance(value, (bool, np.bool_))
|
|
94
|
+
for value in matrix.to_numpy().flat
|
|
95
|
+
):
|
|
96
|
+
raise AnalysisError(
|
|
97
|
+
"A matriz externa deve conter somente números reais, sem booleanos ou strings."
|
|
98
|
+
)
|
|
99
|
+
try:
|
|
100
|
+
aligned = matrix.loc[list(files), list(files)].astype(float).copy(deep=True)
|
|
101
|
+
except (TypeError, ValueError, OverflowError) as exc:
|
|
102
|
+
raise AnalysisError(
|
|
103
|
+
"A matriz externa deve conter somente valores numéricos finitos."
|
|
104
|
+
) from exc
|
|
105
|
+
values = aligned.to_numpy()
|
|
106
|
+
if not np.isfinite(values).all() or ((values < 0) | (values > 1)).any():
|
|
107
|
+
raise AnalysisError(
|
|
108
|
+
"A matriz externa deve conter somente similaridades numéricas entre 0 e 1."
|
|
109
|
+
)
|
|
110
|
+
if not (values.diagonal() == 1).all():
|
|
111
|
+
raise AnalysisError("A diagonal da matriz externa deve ser igual a 1.")
|
|
112
|
+
if not aligned.equals(aligned.T):
|
|
113
|
+
raise AnalysisError("A matriz externa deve ser simétrica.")
|
|
114
|
+
aligned.attrs = deepcopy(matrix.attrs)
|
|
115
|
+
return aligned
|