evalsuite-python 0.1.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,587 @@
1
+ """Regression metrics.
2
+
3
+ Single- and multi-output targets (2-D arrays, one column per output) are supported. ``multioutput``
4
+ is ``"uniform_average"`` (default), ``"raw_values"`` (one value per output) or an array of output weights.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import warnings
10
+ from typing import Any, Optional, Union
11
+
12
+ import numpy as np
13
+
14
+ from ..core.exceptions import InputValidationError, MetricInputError, UndefinedMetricWarning
15
+ from ..core.registry import register
16
+ from ..core.result import MetricResult
17
+ from ..core.types import ArrayLike, FloatArray
18
+ from ..core.validation import check_consistent_length, check_finite, to_numpy, validate_sample_weight
19
+
20
+ __all__ = [
21
+ "adjusted_r2",
22
+ "explained_variance",
23
+ "huber_loss",
24
+ "mae",
25
+ "mape",
26
+ "max_error",
27
+ "mean_bias_error",
28
+ "median_absolute_error",
29
+ "mse",
30
+ "msle",
31
+ "quantile_loss",
32
+ "r2",
33
+ "rae",
34
+ "rmse",
35
+ "rmsle",
36
+ "rse",
37
+ "smape",
38
+ ]
39
+
40
+ _R = "regression"
41
+ Multioutput = Union[str, ArrayLike]
42
+ _REF_HYNDMAN = (
43
+ "Hyndman RJ, Koehler AB. Another look at measures of forecast accuracy. Int J Forecast. 2006;22(4):679-688."
44
+ )
45
+ _REF_WILLMOTT = "Willmott CJ, Matsuura K. Advantages of the mean absolute error (MAE) over the root mean square error (RMSE) in assessing average model performance. Clim Res. 2005;30:79-82."
46
+
47
+
48
+ class _Inputs:
49
+ """Validated regression arrays as float64 (n, k)."""
50
+
51
+ def __init__(self, y_true: ArrayLike, y_pred: ArrayLike, sample_weight: Optional[ArrayLike]) -> None:
52
+ yt = to_numpy(y_true, "y_true", allow_2d=True)
53
+ yp = to_numpy(y_pred, "y_pred", allow_2d=True)
54
+ for arr, name in ((yt, "y_true"), (yp, "y_pred")):
55
+ if not np.issubdtype(arr.dtype, np.number) or np.issubdtype(arr.dtype, np.bool_):
56
+ raise InputValidationError(
57
+ f"{name} must be numeric for regression metrics; got dtype {arr.dtype}."
58
+ )
59
+ yt = yt.astype(np.float64, copy=False)
60
+ yp = yp.astype(np.float64, copy=False)
61
+ check_finite(yt, "y_true")
62
+ check_finite(yp, "y_pred")
63
+ self.n = check_consistent_length(y_true=yt, y_pred=yp)
64
+ if yt.ndim != yp.ndim or (yt.ndim == 2 and yt.shape[1] != yp.shape[1]):
65
+ raise InputValidationError(
66
+ f"y_true has shape {yt.shape} but y_pred has shape {yp.shape}; they must match."
67
+ )
68
+ self.multi = yt.ndim == 2
69
+ self.y_true: FloatArray = yt if self.multi else yt[:, None]
70
+ self.y_pred: FloatArray = yp if self.multi else yp[:, None]
71
+ self.weight = validate_sample_weight(sample_weight, self.n)
72
+
73
+ @property
74
+ def w(self) -> FloatArray:
75
+ return np.ones(self.n) if self.weight is None else self.weight
76
+
77
+ @property
78
+ def error(self) -> FloatArray:
79
+ return self.y_pred - self.y_true
80
+
81
+ def mean(self, values: FloatArray) -> FloatArray:
82
+ """Weighted mean over observations, per output column."""
83
+ return np.average(values, axis=0, weights=self.w)
84
+
85
+
86
+ def _finish(
87
+ per_output: FloatArray,
88
+ inp: _Inputs,
89
+ multioutput: Multioutput,
90
+ metric: str,
91
+ name: str,
92
+ params: Optional[dict[str, Any]] = None,
93
+ ) -> MetricResult:
94
+ params = {**(params or {})}
95
+ if inp.multi:
96
+ params["multioutput"] = multioutput if isinstance(multioutput, str) else "weights"
97
+ if not inp.multi:
98
+ return MetricResult(metric, name, float(per_output[0]), params)
99
+ if isinstance(multioutput, str):
100
+ if multioutput == "raw_values":
101
+ return MetricResult(metric, name, per_output, params, labels=tuple(range(per_output.shape[0])))
102
+ if multioutput == "uniform_average":
103
+ return MetricResult(metric, name, float(per_output.mean()), params)
104
+ raise InputValidationError("multioutput must be 'uniform_average', 'raw_values' or an array of weights.")
105
+ weights = to_numpy(multioutput, "multioutput").astype(float)
106
+ if weights.shape[0] != per_output.shape[0] or np.any(weights < 0) or weights.sum() == 0:
107
+ raise InputValidationError("multioutput weights must be non-negative, one per output, not all zero.")
108
+ return MetricResult(metric, name, float(np.average(per_output, weights=weights)), params)
109
+
110
+
111
+ def _weighted_median(values: FloatArray, w: FloatArray) -> float:
112
+ """Median minimising Σ wᵢ|xᵢ − m|: the smallest value whose cumulative weight reaches half the total."""
113
+ order = np.argsort(values, kind="mergesort")
114
+ v, cw = values[order], np.cumsum(w[order])
115
+ return float(v[np.searchsorted(cw, 0.5 * cw[-1])])
116
+
117
+
118
+ # ---- metrics --------------------------------------------------------------------------------------
119
+ @register(
120
+ category=_R,
121
+ task="regression",
122
+ name="Mean absolute error",
123
+ definition="Average absolute difference between predictions and targets, in the target's units.",
124
+ formula="(1/n) Σ |ŷᵢ − yᵢ|",
125
+ range="[0, ∞)",
126
+ input_requirements=("y_true", "y_pred"),
127
+ references=(_REF_WILLMOTT,),
128
+ higher_is_better=False,
129
+ )
130
+ def mae(
131
+ y_true: ArrayLike,
132
+ y_pred: ArrayLike,
133
+ *,
134
+ sample_weight: Optional[ArrayLike] = None,
135
+ multioutput: Multioutput = "uniform_average",
136
+ ) -> MetricResult:
137
+ """Mean absolute error."""
138
+ inp = _Inputs(y_true, y_pred, sample_weight)
139
+ return _finish(inp.mean(np.abs(inp.error)), inp, multioutput, "mae", "MAE")
140
+
141
+
142
+ @register(
143
+ category=_R,
144
+ task="regression",
145
+ name="Mean squared error",
146
+ definition="Average squared difference between predictions and targets; penalises large errors.",
147
+ formula="(1/n) Σ (ŷᵢ − yᵢ)²",
148
+ range="[0, ∞)",
149
+ input_requirements=("y_true", "y_pred"),
150
+ references=(_REF_WILLMOTT,),
151
+ higher_is_better=False,
152
+ )
153
+ def mse(
154
+ y_true: ArrayLike,
155
+ y_pred: ArrayLike,
156
+ *,
157
+ sample_weight: Optional[ArrayLike] = None,
158
+ multioutput: Multioutput = "uniform_average",
159
+ ) -> MetricResult:
160
+ """Mean squared error."""
161
+ inp = _Inputs(y_true, y_pred, sample_weight)
162
+ return _finish(inp.mean(inp.error**2), inp, multioutput, "mse", "MSE")
163
+
164
+
165
+ @register(
166
+ category=_R,
167
+ task="regression",
168
+ name="Root mean squared error",
169
+ definition="Square root of the mean squared error, in the target's units.",
170
+ formula="√((1/n) Σ (ŷᵢ − yᵢ)²)",
171
+ range="[0, ∞)",
172
+ input_requirements=("y_true", "y_pred"),
173
+ references=(
174
+ _REF_WILLMOTT,
175
+ "Chai T, Draxler RR. Root mean square error (RMSE) or mean absolute error (MAE)? Geosci Model Dev. 2014;7:1247-1250.",
176
+ ),
177
+ higher_is_better=False,
178
+ )
179
+ def rmse(
180
+ y_true: ArrayLike,
181
+ y_pred: ArrayLike,
182
+ *,
183
+ sample_weight: Optional[ArrayLike] = None,
184
+ multioutput: Multioutput = "uniform_average",
185
+ ) -> MetricResult:
186
+ """Root mean squared error (square root taken per output, then averaged)."""
187
+ inp = _Inputs(y_true, y_pred, sample_weight)
188
+ return _finish(np.sqrt(inp.mean(inp.error**2)), inp, multioutput, "rmse", "RMSE")
189
+
190
+
191
+ def _r2_per_output(inp: _Inputs) -> FloatArray:
192
+ w = inp.w[:, None]
193
+ ss_res = (w * inp.error**2).sum(0)
194
+ mean = inp.mean(inp.y_true)
195
+ ss_tot = (w * (inp.y_true - mean) ** 2).sum(0)
196
+ out = np.empty_like(ss_res)
197
+ const = ss_tot == 0
198
+ out[~const] = 1 - ss_res[~const] / ss_tot[~const]
199
+ if np.any(const):
200
+ warnings.warn(
201
+ "R² is undefined when y_true is constant; returning 1.0 for perfect predictions and 0.0 otherwise "
202
+ "(the scikit-learn convention).",
203
+ UndefinedMetricWarning,
204
+ stacklevel=3,
205
+ )
206
+ out[const] = np.where(ss_res[const] == 0, 1.0, 0.0)
207
+ return out
208
+
209
+
210
+ @register(
211
+ category=_R,
212
+ task="regression",
213
+ name="Coefficient of determination (R²)",
214
+ definition="Proportion of the variance in the target explained by the predictions; can be negative for "
215
+ "models worse than predicting the mean.",
216
+ formula="1 − Σ(yᵢ − ŷᵢ)² / Σ(yᵢ − ȳ)²",
217
+ range="(−∞, 1]",
218
+ input_requirements=("y_true", "y_pred"),
219
+ references=("Kvålseth TO. Cautionary note about R². Am Stat. 1985;39(4):279-285.",),
220
+ )
221
+ def r2(
222
+ y_true: ArrayLike,
223
+ y_pred: ArrayLike,
224
+ *,
225
+ sample_weight: Optional[ArrayLike] = None,
226
+ multioutput: Multioutput = "uniform_average",
227
+ ) -> MetricResult:
228
+ """Coefficient of determination."""
229
+ inp = _Inputs(y_true, y_pred, sample_weight)
230
+ return _finish(_r2_per_output(inp), inp, multioutput, "r2", "R²")
231
+
232
+
233
+ @register(
234
+ category=_R,
235
+ task="regression",
236
+ name="Adjusted R²",
237
+ definition="R² penalised for the number of predictors, for comparing models with different numbers of features.",
238
+ formula="1 − (1 − R²)(n − 1)/(n − p − 1)",
239
+ range="(−∞, 1]",
240
+ input_requirements=("y_true", "y_pred", "n_features"),
241
+ references=("Theil H. Economic Forecasts and Policy. North-Holland; 1961.",),
242
+ )
243
+ def adjusted_r2(
244
+ y_true: ArrayLike, y_pred: ArrayLike, *, n_features: int, sample_weight: Optional[ArrayLike] = None
245
+ ) -> MetricResult:
246
+ """Adjusted R². Needs ``n_features`` (number of predictors, excluding the intercept)."""
247
+ inp = _Inputs(y_true, y_pred, sample_weight)
248
+ if inp.multi:
249
+ raise InputValidationError("adjusted_r2 supports single-output targets only.")
250
+ if not isinstance(n_features, (int, np.integer)) or n_features < 0:
251
+ raise InputValidationError("n_features must be a non-negative integer.")
252
+ if inp.n - n_features - 1 <= 0:
253
+ raise MetricInputError(
254
+ f"Adjusted R² needs more observations than predictors + 1 (n={inp.n}, n_features={n_features})."
255
+ )
256
+ r = float(_r2_per_output(inp)[0])
257
+ value = 1 - (1 - r) * (inp.n - 1) / (inp.n - n_features - 1)
258
+ return MetricResult("adjusted_r2", "Adjusted R²", value, {"n_features": int(n_features)})
259
+
260
+
261
+ @register(
262
+ category=_R,
263
+ task="regression",
264
+ name="Mean absolute percentage error",
265
+ definition="Average absolute error relative to the true value, as a fraction (multiply by 100 for percent).",
266
+ formula="(1/n) Σ |ŷᵢ − yᵢ| / |yᵢ|",
267
+ range="[0, ∞)",
268
+ input_requirements=("y_true ≠ 0", "y_pred"),
269
+ references=(_REF_HYNDMAN,),
270
+ higher_is_better=False,
271
+ )
272
+ def mape(
273
+ y_true: ArrayLike,
274
+ y_pred: ArrayLike,
275
+ *,
276
+ sample_weight: Optional[ArrayLike] = None,
277
+ multioutput: Multioutput = "uniform_average",
278
+ ) -> MetricResult:
279
+ """MAPE as a fraction. Refuses zero targets (division by zero) instead of silently using a tiny epsilon."""
280
+ inp = _Inputs(y_true, y_pred, sample_weight)
281
+ zeros = int((inp.y_true == 0).sum())
282
+ if zeros:
283
+ raise MetricInputError(
284
+ f"MAPE is undefined because y_true contains {zeros} zero value(s). "
285
+ "Use smape, mae or rae instead, or exclude zero targets explicitly."
286
+ )
287
+ return _finish(inp.mean(np.abs(inp.error) / np.abs(inp.y_true)), inp, multioutput, "mape", "MAPE")
288
+
289
+
290
+ @register(
291
+ category=_R,
292
+ task="regression",
293
+ name="Symmetric mean absolute percentage error",
294
+ definition="Absolute error relative to the mean magnitude of target and prediction; a pair that is both "
295
+ "zero contributes 0.",
296
+ formula="(1/n) Σ 2|ŷᵢ − yᵢ| / (|yᵢ| + |ŷᵢ|)",
297
+ range="[0, 2]",
298
+ input_requirements=("y_true", "y_pred"),
299
+ references=(
300
+ _REF_HYNDMAN,
301
+ "Makridakis S. Accuracy measures: theoretical and practical concerns. Int J Forecast. 1993;9(4):527-529.",
302
+ ),
303
+ higher_is_better=False,
304
+ )
305
+ def smape(
306
+ y_true: ArrayLike,
307
+ y_pred: ArrayLike,
308
+ *,
309
+ sample_weight: Optional[ArrayLike] = None,
310
+ multioutput: Multioutput = "uniform_average",
311
+ ) -> MetricResult:
312
+ """Symmetric MAPE (fraction, range [0, 2])."""
313
+ inp = _Inputs(y_true, y_pred, sample_weight)
314
+ den = np.abs(inp.y_true) + np.abs(inp.y_pred)
315
+ terms = np.divide(2 * np.abs(inp.error), den, out=np.zeros_like(den), where=den != 0)
316
+ return _finish(inp.mean(terms), inp, multioutput, "smape", "sMAPE")
317
+
318
+
319
+ def _check_log_domain(inp: _Inputs, name: str) -> None:
320
+ if np.any(inp.y_true < 0) or np.any(inp.y_pred < 0):
321
+ raise MetricInputError(f"{name} needs non-negative y_true and y_pred (it compares log(1 + y)).")
322
+
323
+
324
+ @register(
325
+ category=_R,
326
+ task="regression",
327
+ name="Mean squared logarithmic error",
328
+ definition="Mean squared difference of log(1 + y); emphasises relative error and penalises under-prediction.",
329
+ formula="(1/n) Σ (log(1 + ŷᵢ) − log(1 + yᵢ))²",
330
+ range="[0, ∞)",
331
+ input_requirements=("y_true ≥ 0", "y_pred ≥ 0"),
332
+ references=(_REF_HYNDMAN,),
333
+ higher_is_better=False,
334
+ )
335
+ def msle(
336
+ y_true: ArrayLike,
337
+ y_pred: ArrayLike,
338
+ *,
339
+ sample_weight: Optional[ArrayLike] = None,
340
+ multioutput: Multioutput = "uniform_average",
341
+ ) -> MetricResult:
342
+ """Mean squared logarithmic error."""
343
+ inp = _Inputs(y_true, y_pred, sample_weight)
344
+ _check_log_domain(inp, "MSLE")
345
+ return _finish(inp.mean((np.log1p(inp.y_pred) - np.log1p(inp.y_true)) ** 2), inp, multioutput, "msle", "MSLE")
346
+
347
+
348
+ @register(
349
+ category=_R,
350
+ task="regression",
351
+ name="Root mean squared logarithmic error",
352
+ definition="Square root of the mean squared logarithmic error.",
353
+ formula="√MSLE",
354
+ range="[0, ∞)",
355
+ input_requirements=("y_true ≥ 0", "y_pred ≥ 0"),
356
+ references=(_REF_HYNDMAN,),
357
+ higher_is_better=False,
358
+ )
359
+ def rmsle(
360
+ y_true: ArrayLike,
361
+ y_pred: ArrayLike,
362
+ *,
363
+ sample_weight: Optional[ArrayLike] = None,
364
+ multioutput: Multioutput = "uniform_average",
365
+ ) -> MetricResult:
366
+ """Root mean squared logarithmic error."""
367
+ inp = _Inputs(y_true, y_pred, sample_weight)
368
+ _check_log_domain(inp, "RMSLE")
369
+ per = np.sqrt(inp.mean((np.log1p(inp.y_pred) - np.log1p(inp.y_true)) ** 2))
370
+ return _finish(per, inp, multioutput, "rmsle", "RMSLE")
371
+
372
+
373
+ @register(
374
+ category=_R,
375
+ task="regression",
376
+ name="Median absolute error",
377
+ definition="Median of the absolute errors; robust to outliers.",
378
+ formula="median(|ŷᵢ − yᵢ|)",
379
+ range="[0, ∞)",
380
+ input_requirements=("y_true", "y_pred"),
381
+ references=("Rousseeuw PJ, Leroy AM. Robust Regression and Outlier Detection. Wiley; 1987.",),
382
+ higher_is_better=False,
383
+ )
384
+ def median_absolute_error(
385
+ y_true: ArrayLike,
386
+ y_pred: ArrayLike,
387
+ *,
388
+ sample_weight: Optional[ArrayLike] = None,
389
+ multioutput: Multioutput = "uniform_average",
390
+ ) -> MetricResult:
391
+ """Median absolute error (weighted median when ``sample_weight`` is given)."""
392
+ inp = _Inputs(y_true, y_pred, sample_weight)
393
+ abs_err = np.abs(inp.error)
394
+ if inp.weight is None:
395
+ per = np.median(abs_err, axis=0)
396
+ else:
397
+ per = np.array([_weighted_median(abs_err[:, j], inp.w) for j in range(abs_err.shape[1])])
398
+ return _finish(per, inp, multioutput, "median_absolute_error", "Median absolute error")
399
+
400
+
401
+ @register(
402
+ category=_R,
403
+ task="regression",
404
+ name="Explained variance",
405
+ definition="Proportion of target variance explained, ignoring systematic bias (unlike R²).",
406
+ formula="1 − Var(y − ŷ) / Var(y)",
407
+ range="(−∞, 1]",
408
+ input_requirements=("y_true", "y_pred"),
409
+ references=("Draper NR, Smith H. Applied Regression Analysis. 3rd ed. Wiley; 1998.",),
410
+ )
411
+ def explained_variance(
412
+ y_true: ArrayLike,
413
+ y_pred: ArrayLike,
414
+ *,
415
+ sample_weight: Optional[ArrayLike] = None,
416
+ multioutput: Multioutput = "uniform_average",
417
+ ) -> MetricResult:
418
+ """Explained variance score."""
419
+ inp = _Inputs(y_true, y_pred, sample_weight)
420
+ resid = inp.y_true - inp.y_pred
421
+ var_res = inp.mean((resid - inp.mean(resid)) ** 2)
422
+ var_true = inp.mean((inp.y_true - inp.mean(inp.y_true)) ** 2)
423
+ out = np.empty_like(var_res)
424
+ const = var_true == 0
425
+ out[~const] = 1 - var_res[~const] / var_true[~const]
426
+ if np.any(const):
427
+ warnings.warn(
428
+ "Explained variance is undefined when y_true is constant; returning 1.0 for zero residual "
429
+ "variance and 0.0 otherwise.",
430
+ UndefinedMetricWarning,
431
+ stacklevel=2,
432
+ )
433
+ out[const] = np.where(var_res[const] == 0, 1.0, 0.0)
434
+ return _finish(out, inp, multioutput, "explained_variance", "Explained variance")
435
+
436
+
437
+ @register(
438
+ category=_R,
439
+ task="regression",
440
+ name="Maximum error",
441
+ definition="Largest absolute error: the worst case.",
442
+ formula="maxᵢ |ŷᵢ − yᵢ|",
443
+ range="[0, ∞)",
444
+ input_requirements=("y_true", "y_pred"),
445
+ references=("Hyndman RJ, Athanasopoulos G. Forecasting: Principles and Practice. 3rd ed. OTexts; 2021.",),
446
+ higher_is_better=False,
447
+ )
448
+ def max_error(y_true: ArrayLike, y_pred: ArrayLike) -> MetricResult:
449
+ """Maximum absolute error (single output, unweighted)."""
450
+ inp = _Inputs(y_true, y_pred, None)
451
+ if inp.multi:
452
+ raise InputValidationError("max_error supports single-output targets only.")
453
+ return MetricResult("max_error", "Max error", float(np.abs(inp.error).max()))
454
+
455
+
456
+ @register(
457
+ category=_R,
458
+ task="regression",
459
+ name="Mean bias error",
460
+ definition="Average signed error: positive when the model over-predicts on average.",
461
+ formula="(1/n) Σ (ŷᵢ − yᵢ)",
462
+ range="(−∞, ∞) (0 = unbiased)",
463
+ input_requirements=("y_true", "y_pred"),
464
+ references=(_REF_WILLMOTT,),
465
+ higher_is_better=None,
466
+ )
467
+ def mean_bias_error(
468
+ y_true: ArrayLike,
469
+ y_pred: ArrayLike,
470
+ *,
471
+ sample_weight: Optional[ArrayLike] = None,
472
+ multioutput: Multioutput = "uniform_average",
473
+ ) -> MetricResult:
474
+ """Mean bias error (prediction minus truth)."""
475
+ inp = _Inputs(y_true, y_pred, sample_weight)
476
+ return _finish(inp.mean(inp.error), inp, multioutput, "mean_bias_error", "Mean bias error")
477
+
478
+
479
+ @register(
480
+ category=_R,
481
+ task="regression",
482
+ name="Quantile (pinball) loss",
483
+ definition="Asymmetric absolute loss for evaluating a predicted alpha-quantile.",
484
+ formula="(1/n) Σ max(α(yᵢ − ŷᵢ), (α − 1)(yᵢ − ŷᵢ))",
485
+ range="[0, ∞)",
486
+ input_requirements=("y_true", "y_pred", "alpha"),
487
+ references=("Koenker R, Bassett G. Regression quantiles. Econometrica. 1978;46(1):33-50.",),
488
+ higher_is_better=False,
489
+ )
490
+ def quantile_loss(
491
+ y_true: ArrayLike,
492
+ y_pred: ArrayLike,
493
+ *,
494
+ alpha: float = 0.5,
495
+ sample_weight: Optional[ArrayLike] = None,
496
+ multioutput: Multioutput = "uniform_average",
497
+ ) -> MetricResult:
498
+ """Pinball loss for the ``alpha`` quantile (alpha=0.5 gives half the MAE)."""
499
+ if not (isinstance(alpha, (int, float)) and 0 < alpha < 1):
500
+ raise InputValidationError("alpha must be strictly between 0 and 1.")
501
+ inp = _Inputs(y_true, y_pred, sample_weight)
502
+ diff = inp.y_true - inp.y_pred
503
+ loss = np.maximum(alpha * diff, (alpha - 1) * diff)
504
+ return _finish(inp.mean(loss), inp, multioutput, "quantile_loss", "Quantile loss", {"alpha": float(alpha)})
505
+
506
+
507
+ @register(
508
+ category=_R,
509
+ task="regression",
510
+ name="Huber loss",
511
+ definition="Squared error for small residuals and linear error beyond delta; robust to outliers.",
512
+ formula="½e² if |e| ≤ δ, else δ(|e| − ½δ)",
513
+ range="[0, ∞)",
514
+ input_requirements=("y_true", "y_pred", "delta"),
515
+ references=("Huber PJ. Robust estimation of a location parameter. Ann Math Stat. 1964;35(1):73-101.",),
516
+ higher_is_better=False,
517
+ )
518
+ def huber_loss(
519
+ y_true: ArrayLike,
520
+ y_pred: ArrayLike,
521
+ *,
522
+ delta: float = 1.0,
523
+ sample_weight: Optional[ArrayLike] = None,
524
+ multioutput: Multioutput = "uniform_average",
525
+ ) -> MetricResult:
526
+ """Mean Huber loss."""
527
+ if not (isinstance(delta, (int, float)) and delta > 0):
528
+ raise InputValidationError("delta must be positive.")
529
+ inp = _Inputs(y_true, y_pred, sample_weight)
530
+ a = np.abs(inp.error)
531
+ loss = np.where(a <= delta, 0.5 * a**2, delta * (a - 0.5 * delta))
532
+ return _finish(inp.mean(loss), inp, multioutput, "huber_loss", "Huber loss", {"delta": float(delta)})
533
+
534
+
535
+ def _relative(inp: _Inputs, power: int, name: str) -> FloatArray:
536
+ num = (inp.w[:, None] * np.abs(inp.error) ** power).sum(0)
537
+ den = (inp.w[:, None] * np.abs(inp.y_true - inp.mean(inp.y_true)) ** power).sum(0)
538
+ if np.any(den == 0):
539
+ raise MetricInputError(f"{name} is undefined when y_true is constant (zero baseline error).")
540
+ ratio: FloatArray = num / den
541
+ return ratio
542
+
543
+
544
+ @register(
545
+ category=_R,
546
+ task="regression",
547
+ name="Relative absolute error",
548
+ definition="Total absolute error relative to that of always predicting the mean; below 1 beats the mean.",
549
+ formula="Σ|ŷᵢ − yᵢ| / Σ|yᵢ − ȳ|",
550
+ range="[0, ∞)",
551
+ input_requirements=("y_true", "y_pred"),
552
+ references=("Witten IH, Frank E, Hall MA. Data Mining. 3rd ed. Morgan Kaufmann; 2011.",),
553
+ higher_is_better=False,
554
+ )
555
+ def rae(
556
+ y_true: ArrayLike,
557
+ y_pred: ArrayLike,
558
+ *,
559
+ sample_weight: Optional[ArrayLike] = None,
560
+ multioutput: Multioutput = "uniform_average",
561
+ ) -> MetricResult:
562
+ """Relative absolute error."""
563
+ inp = _Inputs(y_true, y_pred, sample_weight)
564
+ return _finish(_relative(inp, 1, "RAE"), inp, multioutput, "rae", "RAE")
565
+
566
+
567
+ @register(
568
+ category=_R,
569
+ task="regression",
570
+ name="Relative squared error",
571
+ definition="Total squared error relative to that of always predicting the mean (equals 1 − R²).",
572
+ formula="Σ(ŷᵢ − yᵢ)² / Σ(yᵢ − ȳ)²",
573
+ range="[0, ∞)",
574
+ input_requirements=("y_true", "y_pred"),
575
+ references=("Witten IH, Frank E, Hall MA. Data Mining. 3rd ed. Morgan Kaufmann; 2011.",),
576
+ higher_is_better=False,
577
+ )
578
+ def rse(
579
+ y_true: ArrayLike,
580
+ y_pred: ArrayLike,
581
+ *,
582
+ sample_weight: Optional[ArrayLike] = None,
583
+ multioutput: Multioutput = "uniform_average",
584
+ ) -> MetricResult:
585
+ """Relative squared error."""
586
+ inp = _Inputs(y_true, y_pred, sample_weight)
587
+ return _finish(_relative(inp, 2, "RSE"), inp, multioutput, "rse", "RSE")
evalsuite/version.py ADDED
@@ -0,0 +1,3 @@
1
+ """Package version (single source of truth, read by the build backend)."""
2
+
3
+ __version__ = "0.1.0a1"