evalsuite-python 0.1.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,202 @@
1
+ """Centralised input validation.
2
+
3
+ Inputs are converted to NumPy once, checked once, and never modified in place.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import warnings
9
+ from typing import Any, Literal, Optional
10
+
11
+ import numpy as np
12
+ from numpy.typing import NDArray
13
+
14
+ from .exceptions import InputValidationError, UndefinedMetricWarning
15
+ from .types import ArrayLike, FloatArray, ZeroDivision
16
+
17
+ __all__ = [
18
+ "TargetType",
19
+ "check_consistent_length",
20
+ "resolve_labels",
21
+ "safe_divide",
22
+ "target_type",
23
+ "to_numpy",
24
+ "validate_probabilities",
25
+ "validate_sample_weight",
26
+ "validate_zero_division",
27
+ ]
28
+
29
+ TargetType = Literal["binary", "multiclass", "multilabel", "continuous"]
30
+
31
+
32
+ def to_numpy(x: ArrayLike, name: str, *, allow_2d: bool = False) -> NDArray[Any]:
33
+ """Convert array-likes (lists, NumPy arrays, pandas objects) to a NumPy array without copying
34
+ when possible. Raises :class:`InputValidationError` for None, scalars, empty input or bad shapes."""
35
+ if x is None:
36
+ raise InputValidationError(f"{name} is required but was None.")
37
+ values = getattr(x, "to_numpy", None)
38
+ arr = np.asarray(values() if callable(values) else x)
39
+ if arr.ndim == 0:
40
+ raise InputValidationError(f"{name} must be one-dimensional array-like; received a scalar.")
41
+ if arr.ndim == 2 and arr.shape[1] == 1 and not allow_2d:
42
+ arr = arr.ravel()
43
+ if arr.ndim > (2 if allow_2d else 1):
44
+ expected = "1-D or 2-D" if allow_2d else "1-D"
45
+ raise InputValidationError(f"{name} must be {expected}; received an array with shape {arr.shape}.")
46
+ if arr.shape[0] == 0:
47
+ raise InputValidationError(f"{name} is empty. Provide at least one observation.")
48
+ return arr
49
+
50
+
51
+ def check_finite(arr: NDArray[Any], name: str) -> None:
52
+ if np.issubdtype(arr.dtype, np.number) and not np.all(np.isfinite(arr)):
53
+ n_nan = int(np.isnan(arr).sum())
54
+ n_inf = int(np.isinf(arr).sum())
55
+ raise InputValidationError(
56
+ f"{name} contains {n_nan} NaN and {n_inf} infinite value(s). Remove or impute them before evaluating."
57
+ )
58
+
59
+
60
+ def check_consistent_length(**arrays: Optional[NDArray[Any]]) -> int:
61
+ """All given arrays must have the same number of observations; returns that number."""
62
+ lengths = {name: a.shape[0] for name, a in arrays.items() if a is not None}
63
+ if len(set(lengths.values())) > 1:
64
+ detail = " and ".join(f"{k}={v}" for k, v in lengths.items())
65
+ names = " and ".join(lengths)
66
+ raise InputValidationError(f"{names} must contain the same number of observations. Received {detail}.")
67
+ return int(next(iter(lengths.values())))
68
+
69
+
70
+ def validate_sample_weight(sample_weight: Optional[ArrayLike], n: int) -> Optional[FloatArray]:
71
+ if sample_weight is None:
72
+ return None
73
+ w = to_numpy(sample_weight, "sample_weight").astype(np.float64, copy=False)
74
+ check_finite(w, "sample_weight")
75
+ if w.shape[0] != n:
76
+ raise InputValidationError(
77
+ f"sample_weight must contain one weight per observation. Received {w.shape[0]} for {n} observations."
78
+ )
79
+ if np.any(w < 0):
80
+ raise InputValidationError("sample_weight must be non-negative.")
81
+ if w.sum() == 0:
82
+ raise InputValidationError("sample_weight must not sum to zero.")
83
+ return w
84
+
85
+
86
+ def target_type(y: NDArray[Any]) -> TargetType:
87
+ """Infer the type of a label array."""
88
+ if y.ndim == 2:
89
+ if y.shape[1] > 1 and np.isin(y, (0, 1)).all():
90
+ return "multilabel"
91
+ if np.issubdtype(y.dtype, np.floating) and not np.all(np.mod(y, 1) == 0):
92
+ return "continuous" # multi-output regression
93
+ raise InputValidationError(
94
+ "2-D targets must be a binary indicator matrix (0/1, one column per label) for multilabel tasks; "
95
+ f"received shape {y.shape} with values other than 0 and 1."
96
+ )
97
+ if np.issubdtype(y.dtype, np.floating) and not np.all(np.mod(y, 1) == 0):
98
+ return "continuous"
99
+ try:
100
+ n_unique = np.unique(y).shape[0]
101
+ except TypeError as exc:
102
+ raise InputValidationError(
103
+ "Labels must be mutually comparable (for example all integers or all strings); found a mix of types."
104
+ ) from exc
105
+ return "binary" if n_unique <= 2 else "multiclass"
106
+
107
+
108
+ def resolve_labels(
109
+ y_true: NDArray[Any], y_pred: Optional[NDArray[Any]], labels: Optional[ArrayLike]
110
+ ) -> NDArray[Any]:
111
+ """Sorted labels present in y_true or y_pred, or the user's explicit list (order kept)."""
112
+ if labels is not None:
113
+ lab = to_numpy(labels, "labels")
114
+ if np.unique(lab).shape[0] != lab.shape[0]:
115
+ raise InputValidationError("labels contains duplicates.")
116
+ return lab
117
+ present = y_true if y_pred is None else np.concatenate([y_true, y_pred])
118
+ try:
119
+ return np.unique(present)
120
+ except TypeError as exc:
121
+ raise InputValidationError(
122
+ "Labels in y_true and y_pred must be mutually comparable (for example all integers or all strings)."
123
+ ) from exc
124
+
125
+
126
+ def validate_probabilities(
127
+ y_prob: ArrayLike,
128
+ n: int,
129
+ *,
130
+ n_classes: Optional[int] = None,
131
+ name: str = "y_prob",
132
+ rows_sum_to_one: bool = True,
133
+ unit: str = "class",
134
+ ) -> FloatArray:
135
+ """Probabilities in [0, 1]. For multiclass (2-D), one column per class and rows summing to 1.
136
+ Multilabel probabilities are independent per label (``rows_sum_to_one=False``)."""
137
+ p = to_numpy(y_prob, name, allow_2d=True).astype(np.float64, copy=False)
138
+ check_finite(p, name)
139
+ if p.shape[0] != n:
140
+ raise InputValidationError(
141
+ f"{name} must contain one row per observation. Received {p.shape[0]} for {n} observations."
142
+ )
143
+ if np.any(p < 0) or np.any(p > 1):
144
+ raise InputValidationError(
145
+ f"{name} must contain probabilities in [0, 1]; found values from {float(np.min(p)):.4g} to "
146
+ f"{float(np.max(p)):.4g}. "
147
+ "If these are scores or logits, convert them to probabilities first."
148
+ )
149
+ if p.ndim == 2:
150
+ if n_classes is not None and p.shape[1] != n_classes:
151
+ raise InputValidationError(
152
+ f"{name} has {p.shape[1]} columns but there are {n_classes} {unit}es. "
153
+ f"Provide one probability column per {unit}, in label order."
154
+ if unit == "class"
155
+ else f"{name} has {p.shape[1]} columns but there are {n_classes} {unit}s. "
156
+ f"Provide one probability column per {unit}."
157
+ )
158
+ sums = p.sum(axis=1)
159
+ if rows_sum_to_one and not np.allclose(sums, 1.0, atol=1e-6):
160
+ worst = float(np.abs(sums - 1).max())
161
+ raise InputValidationError(
162
+ f"Each row of {name} must sum to 1 for multiclass probabilities (largest deviation {worst:.3g})."
163
+ )
164
+ return p
165
+
166
+
167
+ def validate_zero_division(zero_division: ZeroDivision) -> ZeroDivision:
168
+ if zero_division == "warn":
169
+ return zero_division
170
+ if isinstance(zero_division, (int, float)) and (np.isnan(zero_division) or 0 <= zero_division <= 1):
171
+ return float(zero_division)
172
+ raise InputValidationError('zero_division must be "warn", 0, 1 or np.nan.')
173
+
174
+
175
+ def safe_divide(
176
+ num: NDArray[np.float64] | float,
177
+ den: NDArray[np.float64] | float,
178
+ *,
179
+ zero_division: ZeroDivision = "warn",
180
+ metric: str = "metric",
181
+ ) -> NDArray[np.float64]:
182
+ """Element-wise num/den. Where den == 0 the result is ``zero_division``.
183
+
184
+ ``"warn"`` (default) returns 0 and emits :class:`UndefinedMetricWarning`, matching scikit-learn's
185
+ convention; pass ``zero_division=np.nan`` to propagate undefined values instead. Never silent.
186
+ """
187
+ num_a = np.asarray(num, dtype=np.float64)
188
+ den_a = np.asarray(den, dtype=np.float64)
189
+ zero = den_a == 0
190
+ out = np.divide(num_a, den_a, out=np.zeros(np.broadcast(num_a, den_a).shape), where=~zero)
191
+ if np.any(zero):
192
+ if zero_division == "warn":
193
+ warnings.warn(
194
+ f"{metric} is undefined for {int(np.sum(zero))} case(s) because the denominator is zero; "
195
+ "using 0. Set zero_division=0, 1 or np.nan to choose the value explicitly.",
196
+ UndefinedMetricWarning,
197
+ stacklevel=4,
198
+ )
199
+ out[zero] = 0.0
200
+ else:
201
+ out[zero] = float(zero_division)
202
+ return out
evalsuite/py.typed ADDED
File without changes
@@ -0,0 +1,41 @@
1
+ """Regression metrics."""
2
+
3
+ from .metrics import (
4
+ adjusted_r2,
5
+ explained_variance,
6
+ huber_loss,
7
+ mae,
8
+ mape,
9
+ max_error,
10
+ mean_bias_error,
11
+ median_absolute_error,
12
+ mse,
13
+ msle,
14
+ quantile_loss,
15
+ r2,
16
+ rae,
17
+ rmse,
18
+ rmsle,
19
+ rse,
20
+ smape,
21
+ )
22
+
23
+ __all__ = [
24
+ "adjusted_r2",
25
+ "explained_variance",
26
+ "huber_loss",
27
+ "mae",
28
+ "mape",
29
+ "max_error",
30
+ "mean_bias_error",
31
+ "median_absolute_error",
32
+ "mse",
33
+ "msle",
34
+ "quantile_loss",
35
+ "r2",
36
+ "rae",
37
+ "rmse",
38
+ "rmsle",
39
+ "rse",
40
+ "smape",
41
+ ]