codevariability 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,94 @@
1
+ """Interoperability helpers for language-specific analysis adapters."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from pathlib import Path
7
+
8
+ import pandas as pd
9
+
10
+ from .exceptions import AnalysisError
11
+ from .validation import metric_name, similarity_matrix
12
+
13
+
14
+ def _unique_keys(items):
15
+ result = {}
16
+ for key, value in items:
17
+ if key in result:
18
+ raise AnalysisError(f"Chave JSON duplicada: {key}.")
19
+ result[key] = value
20
+ return result
21
+
22
+
23
+ def load_matrix_json(path: str | Path) -> tuple[str, pd.DataFrame]:
24
+ """Load the JSON matrix schema emitted by a codevariability language adapter."""
25
+ try:
26
+ payload = json.loads(
27
+ Path(path).read_text(encoding="utf-8"), object_pairs_hook=_unique_keys
28
+ )
29
+ except (OSError, UnicodeError, ValueError, RecursionError, TypeError) as exc:
30
+ raise AnalysisError(
31
+ f"Não foi possível ler o JSON de métrica externa: {exc}."
32
+ ) from exc
33
+ if not isinstance(payload, dict):
34
+ raise AnalysisError(
35
+ "JSON de métrica externa inválido ou versão de esquema não suportada."
36
+ )
37
+ required = {"schema_version", "metric", "files", "matrix"}
38
+ missing = required - set(payload)
39
+ if missing or payload.get("schema_version") != "codevariability.matrix.v1":
40
+ raise AnalysisError(
41
+ "JSON de métrica externa inválido ou versão de esquema não suportada."
42
+ )
43
+ files = payload["files"]
44
+ if (
45
+ not isinstance(files, list)
46
+ or not files
47
+ or not all(isinstance(name, str) and name for name in files)
48
+ or len(files) != len(set(files))
49
+ ):
50
+ raise AnalysisError(
51
+ "O JSON de métrica externa deve declarar nomes de arquivo únicos."
52
+ )
53
+ metric = metric_name(payload["metric"])
54
+ matrix = payload["matrix"]
55
+ if (
56
+ not isinstance(metric, str)
57
+ or not metric
58
+ or not isinstance(matrix, list)
59
+ or len(matrix) != len(files)
60
+ or any(not isinstance(row, list) or len(row) != len(files) for row in matrix)
61
+ ):
62
+ raise AnalysisError(
63
+ "O JSON de métrica externa deve conter uma métrica e uma matriz quadrada compatível com os arquivos."
64
+ )
65
+ try:
66
+ frame = similarity_matrix(
67
+ pd.DataFrame(matrix, index=files, columns=files, dtype=object), files
68
+ )
69
+ except (TypeError, ValueError, OverflowError) as exc:
70
+ raise AnalysisError(
71
+ "A matriz externa deve conter somente números reais finitos."
72
+ ) from exc
73
+ frame.attrs["metric_id"] = payload.get("metric_id")
74
+ adapter_metadata = payload.get("metadata")
75
+ if adapter_metadata is not None and not isinstance(adapter_metadata, dict):
76
+ raise AnalysisError("Os metadados do adaptador devem ser um objeto JSON.")
77
+ if frame.attrs["metric_id"] is not None and (
78
+ not isinstance(frame.attrs["metric_id"], str)
79
+ or not frame.attrs["metric_id"].strip()
80
+ ):
81
+ raise AnalysisError("metric_id deve ser texto não vazio.")
82
+ if isinstance(adapter_metadata, dict):
83
+ frame.attrs["adapter_metadata"] = adapter_metadata
84
+ frame.attrs["normalization"] = adapter_metadata.get("normalization")
85
+ dimensions = adapter_metadata.get("metric_dimensions")
86
+ if isinstance(dimensions, dict):
87
+ frame.attrs["dimension"] = dimensions.get(metric)
88
+ elif dimensions is not None:
89
+ raise AnalysisError("metric_dimensions deve ser um objeto JSON.")
90
+ for attribute in ("normalization", "dimension"):
91
+ value = frame.attrs.get(attribute)
92
+ if value is not None and (not isinstance(value, str) or not value.strip()):
93
+ raise AnalysisError(f"{attribute} deve ser texto não vazio.")
94
+ return metric, frame
@@ -0,0 +1,89 @@
1
+ """Normalized pairwise similarity metrics."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from itertools import combinations
6
+ from typing import Callable, Mapping, Sequence
7
+
8
+ import pandas as pd
9
+ from rapidfuzz.distance import LCSseq, Levenshtein
10
+ from sklearn.feature_extraction.text import CountVectorizer
11
+ from sklearn.metrics.pairwise import cosine_similarity
12
+
13
+ from .ast_tree_edit import (
14
+ AST_TREE_EDIT_METRIC,
15
+ AST_TREE_EDIT_METRIC_ID,
16
+ ast_tree_edit_similarity,
17
+ )
18
+ from .normalization import code_tokens, text_tokens
19
+
20
+ DEFAULT_METRICS = ("cosine", "jaccard", "lcs", "levenshtein")
21
+ METRICS = (*DEFAULT_METRICS, AST_TREE_EDIT_METRIC)
22
+ METRIC_IDS = {
23
+ "cosine": "cosine_bag_of_words_text_v1",
24
+ "jaccard": "jaccard_word_set_text_v1",
25
+ "lcs": "lcs_code_tokens_v1",
26
+ "levenshtein": "levenshtein_code_tokens_v1",
27
+ AST_TREE_EDIT_METRIC: AST_TREE_EDIT_METRIC_ID,
28
+ }
29
+ METRIC_DIMENSIONS = {
30
+ "cosine": "textual",
31
+ "jaccard": "textual",
32
+ "lcs": "syntactic_token_sequence",
33
+ "levenshtein": "syntactic_token_sequence",
34
+ AST_TREE_EDIT_METRIC: "structural_ast",
35
+ }
36
+ def _empty_matrix(names: Sequence[str]) -> pd.DataFrame:
37
+ matrix = pd.DataFrame(0.0, index=names, columns=names, dtype=float)
38
+ for name in names:
39
+ matrix.loc[name, name] = 1.0
40
+ return matrix
41
+
42
+
43
+ def cosine(sources: Mapping[str, str]) -> pd.DataFrame:
44
+ names = list(sources)
45
+ corpus = [" ".join(text_tokens(sources[name])) for name in names]
46
+ if not any(corpus):
47
+ return _empty_matrix(names)
48
+ vectors = CountVectorizer(token_pattern=r"(?u)\b\w+\b").fit_transform(corpus)
49
+ return pd.DataFrame(cosine_similarity(vectors), index=names, columns=names)
50
+
51
+
52
+ def jaccard(sources: Mapping[str, str]) -> pd.DataFrame:
53
+ names = list(sources)
54
+ sets = {name: set(text_tokens(text)) for name, text in sources.items()}
55
+ matrix = _empty_matrix(names)
56
+ for left, right in combinations(names, 2):
57
+ union = sets[left] | sets[right]
58
+ value = len(sets[left] & sets[right]) / len(union) if union else 1.0
59
+ matrix.loc[left, right] = matrix.loc[right, left] = value
60
+ return matrix
61
+
62
+
63
+ def lcs(sources: Mapping[str, str]) -> pd.DataFrame:
64
+ names = list(sources)
65
+ processed = {name: code_tokens(text, name) for name, text in sources.items()}
66
+ matrix = _empty_matrix(names)
67
+ for left, right in combinations(names, 2):
68
+ value = LCSseq.normalized_similarity(processed[left], processed[right])
69
+ matrix.loc[left, right] = matrix.loc[right, left] = value
70
+ return matrix
71
+
72
+
73
+ def levenshtein(sources: Mapping[str, str]) -> pd.DataFrame:
74
+ names = list(sources)
75
+ processed = {name: code_tokens(text, name) for name, text in sources.items()}
76
+ matrix = _empty_matrix(names)
77
+ for left, right in combinations(names, 2):
78
+ value = Levenshtein.normalized_similarity(processed[left], processed[right])
79
+ matrix.loc[left, right] = matrix.loc[right, left] = value
80
+ return matrix
81
+
82
+
83
+ CALCULATORS: dict[str, Callable[..., pd.DataFrame]] = {
84
+ "cosine": cosine,
85
+ "jaccard": jaccard,
86
+ "lcs": lcs,
87
+ "levenshtein": levenshtein,
88
+ AST_TREE_EDIT_METRIC: ast_tree_edit_similarity,
89
+ }
@@ -0,0 +1,257 @@
1
+ """Reproducible preprocessing for textual and code-token analyses."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import unicodedata
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+
10
+ from pygments import lex
11
+ from pygments.lexers import get_lexer_by_name, get_lexer_for_filename
12
+ from pygments.token import Comment, Operator, Text
13
+ from pygments.util import ClassNotFound
14
+
15
+ TEXT_NORMALIZATION = "unicode_words_casefold_v1"
16
+ CODE_TOKENIZATION = "pygments_lexemes_with_generic_fallback_v2"
17
+
18
+ _WORD = re.compile(r"(?u)\b\w+\b")
19
+ _GENERIC_CODE_TOKEN = re.compile(
20
+ r"""(?x)
21
+ (?:0[xX][0-9a-fA-F]+|0[bB][01]+|0[oO][0-7]+|\d+(?:\.\d+)?(?:[eE][+-]?\d+)?)
22
+ |(?:[^\W\d]\w*)
23
+ |(?:===|!==|>>>|<<=|>>=|\*\*=|//=|\?\?=|\*\*|//|==|!=|<=|>=|&&|\|\||\?\?|\?\.|=>|->|::|\.\.\.|\+\+|--|\+=|-=|\*=|/=|%=|&=|\|=|\^=|:=|<<|>>)
24
+ |(?:[^\s\w])
25
+ """,
26
+ flags=re.UNICODE,
27
+ )
28
+ _COMPOUND_OPERATORS = {
29
+ "===",
30
+ "!==",
31
+ ">>>",
32
+ "<<=",
33
+ ">>=",
34
+ "**",
35
+ "//",
36
+ "==",
37
+ "!=",
38
+ "<=",
39
+ ">=",
40
+ "**=",
41
+ "//=",
42
+ "??=",
43
+ "&&",
44
+ "||",
45
+ "??",
46
+ "?.",
47
+ "=>",
48
+ "->",
49
+ "::",
50
+ "...",
51
+ "++",
52
+ "--",
53
+ "+=",
54
+ "-=",
55
+ "*=",
56
+ "/=",
57
+ "%=",
58
+ "&=",
59
+ "|=",
60
+ "^=",
61
+ ":=",
62
+ "<<",
63
+ ">>",
64
+ }
65
+
66
+
67
+ @dataclass(frozen=True, slots=True)
68
+ class CodeSegment:
69
+ """A code segment and its optional Markdown language identifier."""
70
+
71
+ code: str
72
+ language: str | None = None
73
+ start_line: int = 1
74
+
75
+
76
+ def normalize_text(text: str) -> str:
77
+ """Apply only language-independent textual normalization."""
78
+
79
+ normalized = (
80
+ unicodedata.normalize("NFC", text).replace("\r\n", "\n").replace("\r", "\n")
81
+ )
82
+ return " ".join(normalized.casefold().split())
83
+
84
+
85
+ def text_tokens(text: str) -> list[str]:
86
+ """Return case-folded Unicode words from the complete response."""
87
+
88
+ return _WORD.findall(normalize_text(text))
89
+
90
+
91
+ def fenced_code_segments(text: str) -> tuple[CodeSegment, ...]:
92
+ """Scan closed backtick/tilde fences once, preserving code and line numbers.
93
+
94
+ Only a complete delimiter line closes a fence. Unterminated fences are
95
+ rejected, so malformed Markdown cannot silently become an empty program.
96
+ """
97
+ from .exceptions import AnalysisError
98
+
99
+ result: list[CodeSegment] = []
100
+ delimiter: str | None = None
101
+ width = 0
102
+ language: str | None = None
103
+ start_line = 1
104
+ code: list[str] = []
105
+ for number, line in enumerate(text.splitlines(keepends=True), 1):
106
+ stripped = line.strip(" \t\r\n")
107
+ if delimiter is not None:
108
+ if (
109
+ stripped
110
+ and stripped[0] == delimiter
111
+ and len(stripped) >= width
112
+ and not stripped.strip(delimiter)
113
+ ):
114
+ result.append(CodeSegment("".join(code), language, start_line))
115
+ delimiter = None
116
+ code = []
117
+ else:
118
+ code.append(line)
119
+ continue
120
+ candidate = line.lstrip(" \t")
121
+ if not candidate or candidate[0] not in {"`", "~"}:
122
+ continue
123
+ marker = candidate[0]
124
+ count = len(candidate) - len(candidate.lstrip(marker))
125
+ if count < 3:
126
+ continue
127
+ info = candidate[count:].strip()
128
+ if marker == "`" and "`" in info:
129
+ continue
130
+ delimiter, width = marker, count
131
+ language = info.split(maxsplit=1)[0] if info else None
132
+ start_line = number + 1
133
+ if delimiter is not None:
134
+ raise AnalysisError(f"Fence Markdown sem fechamento (linha {start_line - 1}).")
135
+ return tuple(result)
136
+
137
+
138
+ def code_segments(text: str) -> tuple[CodeSegment, ...]:
139
+ """Extract closed fences, or retain the complete unmodified source."""
140
+ segments = fenced_code_segments(text)
141
+ return segments or (CodeSegment(text),)
142
+
143
+
144
+ def _lexer(filename: str, language: str | None):
145
+ try:
146
+ if language:
147
+ return get_lexer_by_name(language, stripnl=False, ensurenl=False)
148
+ return get_lexer_for_filename(filename, stripnl=False, ensurenl=False)
149
+ except ClassNotFound:
150
+ return None
151
+
152
+
153
+ def _generic_tokens(code: str) -> list[str]:
154
+ # Resolve closing quotes once, backwards. Repeated failed searches from
155
+ # escaped quotes in an unterminated string must not rescan every suffix.
156
+ # next_end[q] is the first legal closing q from the next character;
157
+ # after_next[q] handles the two-character escape branch of the old grammar.
158
+ quotes = ('"', "'", '`')
159
+ next_end = [-1, -1, -1]
160
+ after_next = next_end
161
+ endings: dict[int, int] = {}
162
+ for index in range(len(code) - 1, -1, -1):
163
+ character = code[index]
164
+ current = []
165
+ for position, quote in enumerate(quotes):
166
+ if character == quote:
167
+ endings[index] = next_end[position]
168
+ current.append(index)
169
+ elif character == "\\":
170
+ current.append(after_next[position] if index + 1 < len(code) and code[index + 1] != "\n" else -1)
171
+ else:
172
+ current.append(next_end[position])
173
+ after_next, next_end = next_end, current
174
+ tokens = []
175
+ index = 0
176
+ while index < len(code):
177
+ end = endings.get(index, -1)
178
+ if end >= 0:
179
+ tokens.append(code[index:end + 1])
180
+ index = end + 1
181
+ else:
182
+ match = _GENERIC_CODE_TOKEN.match(code, index)
183
+ if match is None:
184
+ index += 1
185
+ else:
186
+ tokens.append(match.group(0))
187
+ index = match.end()
188
+ return tokens
189
+
190
+
191
+ def _pygments_tokens(code: str, lexer) -> list[str]:
192
+ result: list[str] = []
193
+ adjacent_operator = False
194
+ for token_type, value in lex(code, lexer):
195
+ if token_type in Comment or not value:
196
+ adjacent_operator = False
197
+ continue
198
+ if value.isspace():
199
+ adjacent_operator = False
200
+ continue
201
+ if token_type in Operator:
202
+ candidate = f"{result[-1]}{value}" if adjacent_operator else value
203
+ if adjacent_operator and candidate in _COMPOUND_OPERATORS:
204
+ result[-1] = candidate
205
+ else:
206
+ result.append(value)
207
+ adjacent_operator = True
208
+ continue
209
+ adjacent_operator = False
210
+ if token_type in Text:
211
+ result.extend(_generic_tokens(value))
212
+ else:
213
+ result.append(value.replace("\r\n", "\n").replace("\r", "\n"))
214
+ return result
215
+
216
+
217
+ def code_tokens(text: str, filename: str) -> list[str]:
218
+ """Return ordered code lexemes, excluding comments and whitespace.
219
+
220
+ Pygments supplies language lexers when the filename or Markdown fence has
221
+ a known language. Unknown languages use the documented generic tokenizer;
222
+ this fallback is deterministic but is not presented as a compiler lexer.
223
+ Fragment boundary markers prevent matches from crossing code blocks.
224
+ """
225
+
226
+ segments = (
227
+ fenced_code_segments(text)
228
+ if Path(filename).suffix.lower() in {".md", ".markdown"}
229
+ else (CodeSegment(text),)
230
+ )
231
+ result: list[str] = []
232
+ for segment in segments:
233
+ lexer = _lexer(filename, segment.language)
234
+ tokens = (
235
+ _pygments_tokens(segment.code, lexer)
236
+ if lexer
237
+ else _generic_tokens(segment.code)
238
+ )
239
+ if not tokens:
240
+ continue
241
+ if result:
242
+ result.append("<CODE_FRAGMENT_BOUNDARY>")
243
+ result.extend(tokens)
244
+ return result
245
+
246
+
247
+ def uses_python_code(text: str, filename: str) -> bool:
248
+ """Route .py source or explicitly Python Markdown fences to the parser."""
249
+ suffix = Path(filename).suffix.lower()
250
+ if suffix == ".py":
251
+ return True
252
+ if suffix not in {".md", ".markdown"}:
253
+ return False
254
+ segments = fenced_code_segments(text)
255
+ return bool(segments) and all(
256
+ (segment.language or "").casefold() in {"py", "python"} for segment in segments
257
+ )
@@ -0,0 +1,76 @@
1
+ """Atomic output files: never follow an existing destination symlink."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import tempfile
8
+ from contextlib import contextmanager, suppress
9
+ from pathlib import Path
10
+
11
+ import pandas as pd
12
+
13
+ from .exceptions import AnalysisError
14
+
15
+
16
+ @contextmanager
17
+ def atomic_target(path: str | Path):
18
+ target = Path(path)
19
+ if target.is_symlink():
20
+ raise AnalysisError(
21
+ f"O destino de exportação não pode ser um link simbólico: {target}."
22
+ )
23
+ target.parent.mkdir(parents=True, exist_ok=True)
24
+ descriptor, name = tempfile.mkstemp(
25
+ prefix=".codevariability-", suffix=target.suffix, dir=target.parent
26
+ )
27
+ os.close(descriptor)
28
+ temporary = Path(name)
29
+ try:
30
+ yield temporary
31
+ if target.is_symlink():
32
+ raise AnalysisError(
33
+ f"O destino de exportação tornou-se um link simbólico: {target}."
34
+ )
35
+ os.replace(temporary, target)
36
+ finally:
37
+ temporary.unlink(missing_ok=True)
38
+
39
+
40
+ def frame_json(frame: pd.DataFrame, orient: str) -> object:
41
+ """Preserve pandas' null conventions, reporting serialization errors uniformly."""
42
+ try:
43
+ return json.loads(frame.to_json(orient=orient))
44
+ except (TypeError, ValueError, OverflowError, RecursionError) as exc:
45
+ raise AnalysisError(
46
+ f"A tabela não pode ser serializada como JSON: {exc}."
47
+ ) from exc
48
+
49
+
50
+ def write_json(payload: object, path: str | Path) -> None:
51
+ try:
52
+ content = json.dumps(
53
+ payload, ensure_ascii=False, indent=2, allow_nan=False
54
+ ).encode("utf-8")
55
+ except (TypeError, ValueError, OverflowError, RecursionError) as exc:
56
+ raise AnalysisError(
57
+ f"O resultado não pode ser serializado como JSON: {exc}."
58
+ ) from exc
59
+ with atomic_target(path) as temporary:
60
+ temporary.write_bytes(content)
61
+
62
+
63
+ @contextmanager
64
+ def excel_writer(path: str | Path):
65
+ with atomic_target(path) as temporary:
66
+ # The outer handle closes even if openpyxl cannot save a failed workbook.
67
+ with temporary.open("w+b") as handle:
68
+ writer = pd.ExcelWriter(handle, engine="openpyxl")
69
+ try:
70
+ yield writer
71
+ except BaseException:
72
+ with suppress(Exception):
73
+ writer.close()
74
+ raise
75
+ else:
76
+ writer.close()
@@ -0,0 +1,68 @@
1
+ """Escape untrusted labels when exporting tables opened by spreadsheet apps."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ import re
7
+ from pathlib import Path
8
+
9
+ import pandas as pd
10
+
11
+ from .exceptions import AnalysisError
12
+ from .output import atomic_target
13
+
14
+
15
+ def _safe_cell(value: object, prefix: str) -> object:
16
+ if (
17
+ prefix == "'"
18
+ and isinstance(value, str)
19
+ and re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f\ud800-\udfff\ufffe\uffff]", value)
20
+ ):
21
+ raise AnalysisError(
22
+ "O texto exportado contém caracteres incompatíveis com XML do Excel."
23
+ )
24
+ if isinstance(value, str) and value.lstrip().startswith(("=", "+", "-", "@")):
25
+ return prefix + value
26
+ return value
27
+
28
+
29
+ def _safe_index(index: pd.Index, prefix: str) -> pd.Index:
30
+ if isinstance(index, pd.MultiIndex):
31
+ return pd.MultiIndex.from_tuples(
32
+ [tuple(_safe_cell(part, prefix) for part in label) for label in index],
33
+ names=[_safe_cell(name, prefix) for name in index.names],
34
+ )
35
+ return pd.Index(
36
+ [_safe_cell(label, prefix) for label in index],
37
+ name=_safe_cell(index.name, prefix),
38
+ )
39
+
40
+
41
+ def _safe_frame(frame: pd.DataFrame, prefix: str) -> pd.DataFrame:
42
+ safe = frame.copy()
43
+ for column in safe.columns:
44
+ if safe[column].dtype == object or pd.api.types.is_string_dtype(
45
+ safe[column].dtype
46
+ ):
47
+ safe[column] = safe[column].map(lambda value: _safe_cell(value, prefix))
48
+ safe.index = _safe_index(safe.index, prefix)
49
+ safe.columns = _safe_index(safe.columns, prefix)
50
+ return safe
51
+
52
+
53
+ def safe_to_csv(frame: pd.DataFrame, path: str | Path, **kwargs: object) -> None:
54
+ kwargs.setdefault("quoting", csv.QUOTE_ALL)
55
+ safe = _safe_frame(frame, "\t")
56
+ with atomic_target(path) as temporary:
57
+ try:
58
+ safe.to_csv(temporary, **kwargs)
59
+ except UnicodeError as exc:
60
+ raise AnalysisError(
61
+ "O texto exportado como CSV deve ser representável em UTF-8."
62
+ ) from exc
63
+
64
+
65
+ def safe_to_excel(
66
+ frame: pd.DataFrame, writer: pd.ExcelWriter, **kwargs: object
67
+ ) -> None:
68
+ _safe_frame(frame, "'").to_excel(writer, **kwargs)
@@ -0,0 +1,115 @@
1
+ """Shared validation for public analysis and adapter inputs."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping, Sequence
6
+ from copy import deepcopy
7
+ from numbers import Real
8
+
9
+ import numpy as np
10
+ import pandas as pd
11
+
12
+ from .exceptions import AnalysisError
13
+
14
+ DIMENSION_SCORE_COLUMNS = {
15
+ "textual": "textual_score",
16
+ "syntactic_token_sequence": "syntactic_score",
17
+ "structural_ast": "structural_score",
18
+ }
19
+ RESERVED_COLUMNS = {
20
+ "file",
21
+ "rank",
22
+ "score",
23
+ "overall_score",
24
+ "overall_dispersion",
25
+ *DIMENSION_SCORE_COLUMNS.values(),
26
+ }
27
+
28
+
29
+ def metric_name(value: object) -> str:
30
+ if not isinstance(value, str) or not value.strip() or value in RESERVED_COLUMNS:
31
+ raise AnalysisError(
32
+ "O nome da métrica deve ser texto não vazio e não pode ser uma coluna reservada."
33
+ )
34
+ try:
35
+ value.encode("utf-8")
36
+ except UnicodeError as exc:
37
+ raise AnalysisError(
38
+ "O nome da métrica deve ser representável em UTF-8."
39
+ ) from exc
40
+ return value
41
+
42
+
43
+ def dimensions(metrics: Sequence[str], metadata: Mapping[str, object]) -> None:
44
+ for field in ("metric_ids", "metric_normalizations"):
45
+ values = metadata.get(field, {})
46
+ if not isinstance(values, Mapping) or any(
47
+ not isinstance(value, str) or not value.strip() for value in values.values()
48
+ ):
49
+ raise AnalysisError(
50
+ f"{field} deve mapear métricas para strings não vazias."
51
+ )
52
+ if not isinstance(metadata.get("external_adapters", {}), Mapping):
53
+ raise AnalysisError("external_adapters deve ser um mapeamento.")
54
+ declared = metadata.get("metric_dimensions", {})
55
+ if not isinstance(declared, Mapping):
56
+ raise AnalysisError("metric_dimensions deve ser um mapeamento.")
57
+ columns: dict[str, str] = {}
58
+ for metric in metrics:
59
+ dimension = declared.get(metric, metric)
60
+ if not isinstance(dimension, str) or not dimension.strip():
61
+ raise AnalysisError("A dimensão de cada métrica deve ser texto não vazio.")
62
+ column = DIMENSION_SCORE_COLUMNS.get(dimension, f"{dimension}_score")
63
+ if (
64
+ column in {"file", "rank", "score", "overall_score", "overall_dispersion"}
65
+ or column in metrics
66
+ ):
67
+ raise AnalysisError(
68
+ f"A dimensão '{dimension}' colide com uma coluna reservada ou métrica."
69
+ )
70
+ if column in columns and columns[column] != dimension:
71
+ raise AnalysisError(
72
+ "Dimensões distintas não podem gerar a mesma coluna de score."
73
+ )
74
+ columns[column] = dimension
75
+
76
+
77
+ def similarity_matrix(matrix: object, files: Sequence[str]) -> pd.DataFrame:
78
+ if not isinstance(matrix, pd.DataFrame):
79
+ raise AnalysisError("A matriz de similaridade deve ser um pandas.DataFrame.")
80
+ if not matrix.index.is_unique or not matrix.columns.is_unique:
81
+ raise AnalysisError(
82
+ "Índices e colunas da matriz externa não podem conter nomes duplicados."
83
+ )
84
+ if (
85
+ list(matrix.shape) != [len(files), len(files)]
86
+ or set(matrix.index) != set(files)
87
+ or set(matrix.columns) != set(files)
88
+ ):
89
+ raise AnalysisError(
90
+ "Índices e colunas da matriz externa devem corresponder exatamente aos arquivos analisados."
91
+ )
92
+ if any(
93
+ not isinstance(value, Real) or isinstance(value, (bool, np.bool_))
94
+ for value in matrix.to_numpy().flat
95
+ ):
96
+ raise AnalysisError(
97
+ "A matriz externa deve conter somente números reais, sem booleanos ou strings."
98
+ )
99
+ try:
100
+ aligned = matrix.loc[list(files), list(files)].astype(float).copy(deep=True)
101
+ except (TypeError, ValueError, OverflowError) as exc:
102
+ raise AnalysisError(
103
+ "A matriz externa deve conter somente valores numéricos finitos."
104
+ ) from exc
105
+ values = aligned.to_numpy()
106
+ if not np.isfinite(values).all() or ((values < 0) | (values > 1)).any():
107
+ raise AnalysisError(
108
+ "A matriz externa deve conter somente similaridades numéricas entre 0 e 1."
109
+ )
110
+ if not (values.diagonal() == 1).all():
111
+ raise AnalysisError("A diagonal da matriz externa deve ser igual a 1.")
112
+ if not aligned.equals(aligned.T):
113
+ raise AnalysisError("A matriz externa deve ser simétrica.")
114
+ aligned.attrs = deepcopy(matrix.attrs)
115
+ return aligned