ordboost 0.2.1__tar.gz → 0.3.0.dev0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ordboost
3
- Version: 0.2.1
3
+ Version: 0.3.0.dev0
4
4
  Summary: Non-parametric discrete ordinal gradient boosting with probabilistic prediction calibration.
5
5
  Author-email: Mathias Roesler <mathias.roesler@auckland.ac.nz>
6
6
  License: Apache-2.0
@@ -15,9 +15,11 @@ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
15
  Requires-Python: >=3.10
16
16
  Description-Content-Type: text/markdown
17
17
  License-File: LICENSE
18
+ Requires-Dist: numba>=0.67.0
18
19
  Requires-Dist: numpy>=1.22.0
19
20
  Requires-Dist: scikit-learn>=1.1.0
20
21
  Requires-Dist: scipy>=1.8.0
22
+ Requires-Dist: scores>=2.6.0
21
23
  Provides-Extra: dev
22
24
  Requires-Dist: pytest>=7.0.0; extra == "dev"
23
25
  Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
@@ -26,6 +28,7 @@ Provides-Extra: docs
26
28
  Requires-Dist: mkdocs>=1.5.0; extra == "docs"
27
29
  Requires-Dist: mkdocs-material>=9.5.0; extra == "docs"
28
30
  Requires-Dist: mkdocstrings[python]>=0.24.0; extra == "docs"
31
+ Requires-Dist: mkdocs-include-markdown-plugin; extra == "docs"
29
32
  Dynamic: license-file
30
33
 
31
34
  # OrdBoost
@@ -18,9 +18,11 @@ classifiers = [
18
18
  ]
19
19
 
20
20
  dependencies = [
21
+ "numba>=0.67.0",
21
22
  "numpy>=1.22.0",
22
23
  "scikit-learn>=1.1.0",
23
24
  "scipy>=1.8.0",
25
+ "scores>=2.6.0",
24
26
  ]
25
27
 
26
28
  [project.optional-dependencies]
@@ -33,6 +35,7 @@ docs = [
33
35
  "mkdocs>=1.5.0",
34
36
  "mkdocs-material>=9.5.0",
35
37
  "mkdocstrings[python]>=0.24.0",
38
+ "mkdocs-include-markdown-plugin",
36
39
  ]
37
40
 
38
41
  [build-system]
@@ -21,7 +21,7 @@ from ordboost.metrics import (
21
21
  )
22
22
  from ordboost.models import OrdBoostClassifier, OrdBoostRegressor
23
23
 
24
- __version__ = "0.2.1"
24
+ __version__ = "0.3.0.dev0"
25
25
 
26
26
  __all__ = [
27
27
  # Models
@@ -0,0 +1,526 @@
1
+ """Predictive probability distributions for discrete ordinal outcomes."""
2
+
3
+ from abc import ABC, abstractmethod
4
+ from typing import Union
5
+
6
+ import numpy as np
7
+ from numpy.typing import ArrayLike
8
+
9
+
10
+ class PredictiveDistribution(ABC):
11
+ """Abstract base class for all predictive probability distributions.
12
+
13
+ Defines a unified interface for extracting point estimates, quantiles,
14
+ prediction intervals, and cumulative probabilities regardless of whether
15
+ the distribution is discrete or continuous.
16
+
17
+ Methods
18
+ -------
19
+ mean()
20
+ Calculate expected values across samples.
21
+ median()
22
+ Calculate 50th percentile predictions across samples.
23
+ ppf(q)
24
+ Calculate percent point function (inverse CDF / quantiles).
25
+ interval(alpha=0.10)
26
+ Calculate central prediction bounds for a given significance level.
27
+
28
+ """
29
+
30
+ @abstractmethod
31
+ def mean(self) -> np.ndarray:
32
+ """Calculate the expected value for each sample.
33
+
34
+ Returns
35
+ -------
36
+ np.ndarray
37
+ 1D array of shape (n_samples,) containing expected values in
38
+ physical target units.
39
+
40
+ Notes
41
+ -----
42
+ Implementations must ensure the underlying distribution's total
43
+ probability mass is fully accounted for (e.g. PMF rows summing to
44
+ 1.0, or a CDF grid spanning exactly [0.0, 1.0]) for this to return
45
+ an unbiased estimate. Concrete subclasses enforce this at
46
+ construction time rather than here.
47
+
48
+ """
49
+ ...
50
+
51
+ @abstractmethod
52
+ def _ppf(self, q_arr: np.ndarray) -> np.ndarray:
53
+ """Compute the percent point function for a validated quantile array.
54
+
55
+ Called by the public `ppf` after `q` has already been validated to
56
+ be a scalar or 1D array with values in [0.0, 1.0]. Subclasses
57
+ should not re-validate `q_arr`.
58
+
59
+ Parameters
60
+ ----------
61
+ q_arr : np.ndarray
62
+ Validated quantile level(s): either a 0D (scalar) or 1D array,
63
+ with all values in [0.0, 1.0].
64
+
65
+ Returns
66
+ -------
67
+ np.ndarray
68
+ If `q_arr` is 0D, returns a 1D array of shape (n_samples,).
69
+ If `q_arr` is 1D of length `n_quantiles`, returns a 2D array
70
+ of shape (n_samples, n_quantiles).
71
+
72
+ """
73
+ ...
74
+
75
+ def ppf(self, q: Union[float, ArrayLike]) -> np.ndarray:
76
+ """Percent Point Function (inverse CDF / quantile calculation).
77
+
78
+ Validates `q` once, then delegates to the subclass's `_ppf`.
79
+
80
+ Parameters
81
+ ----------
82
+ q : float | ArrayLike
83
+ Quantile level(s). Can be a single scalar or a 1D array-like
84
+ of quantiles, with all values in [0.0, 1.0].
85
+
86
+ Returns
87
+ -------
88
+ np.ndarray
89
+ If `q` is a scalar, returns a 1D array of shape (n_samples,).
90
+ If `q` is a 1D array of length `n_quantiles`, returns a 2D
91
+ array of shape (n_samples, n_quantiles).
92
+
93
+ Raises
94
+ ------
95
+ ValueError
96
+ If `q` is not a scalar or 1D array, or if any value in `q`
97
+ lies outside [0.0, 1.0].
98
+
99
+ """
100
+ q_arr = np.asarray(q, dtype=float)
101
+ if q_arr.ndim not in (0, 1):
102
+ raise ValueError(
103
+ f"Expected 'q' to be a scalar or 1D array, got a {q_arr.ndim}D array."
104
+ )
105
+ if np.any((q_arr < 0.0) | (q_arr > 1.0)):
106
+ raise ValueError("All quantiles in 'q' must lie within [0.0, 1.0].")
107
+ return self._ppf(q_arr)
108
+
109
+ def median(self) -> np.ndarray:
110
+ """Calculate the 50th percentile (median) prediction for each sample.
111
+
112
+ Returns
113
+ -------
114
+ np.ndarray
115
+ 1D array of shape (n_samples,) containing median predictions.
116
+
117
+ """
118
+ return self.ppf(0.5)
119
+
120
+ def interval(self, alpha: float = 0.10) -> tuple[np.ndarray, np.ndarray]:
121
+ """Calculate central prediction bounds for a given significance level.
122
+
123
+ Parameters
124
+ ----------
125
+ alpha : float, default=0.10
126
+ Significance level (e.g., alpha=0.10 yields a 90% central interval).
127
+
128
+ Returns
129
+ -------
130
+ tuple[np.ndarray, np.ndarray]
131
+ Tuple of (lower_bounds, upper_bounds), each as a 1D array.
132
+
133
+ Raises
134
+ ------
135
+ ValueError
136
+ If `alpha` is not strictly within (0.0, 1.0).
137
+
138
+ """
139
+ if not 0.0 < alpha < 1.0:
140
+ raise ValueError("Significance level 'alpha' must be between 0.0 and 1.0.")
141
+
142
+ lower_q = alpha / 2.0
143
+ upper_q = 1.0 - (alpha / 2.0)
144
+ bounds = self.ppf(np.array([lower_q, upper_q]))
145
+ return bounds[:, 0], bounds[:, 1]
146
+
147
+ @staticmethod
148
+ def _validate_strictly_ascending(name: str, arr: np.ndarray) -> None:
149
+ """Validate that a 1D array is strictly increasing.
150
+
151
+ Shared by subclasses whose interpolation/lookup logic (e.g.
152
+ `np.interp`, `np.argmax` over a CDF) requires an ordered axis to
153
+ behave correctly -- silently, not with an error, if violated.
154
+
155
+ Parameters
156
+ ----------
157
+ name : str
158
+ Name of the array being validated, used in the error message.
159
+ arr : np.ndarray
160
+ 1D array to validate.
161
+
162
+ Raises
163
+ ------
164
+ ValueError
165
+ If `arr` is not strictly increasing.
166
+
167
+ """
168
+ if np.any(np.diff(arr) <= 0.0):
169
+ raise ValueError(f"'{name}' must be strictly ascending.")
170
+
171
+
172
+ class DiscretePredictiveDistribution(PredictiveDistribution):
173
+ """Encapsulates a discrete Probability Mass Function (PMF) matrix.
174
+
175
+ Provides vectorized utilities for computing cumulative distribution
176
+ functions (CDF), percent point functions (quantiles/PPF), expected
177
+ values, medians, and prediction intervals across samples.
178
+
179
+ Parameters
180
+ ----------
181
+ pmf : np.ndarray
182
+ A 2D float array of shape (n_samples, n_classes) representing the
183
+ predicted probability for each discrete target class. Values
184
+ along each row must sum to 1.0 (within floating-point tolerance).
185
+ classes : np.ndarray
186
+ A 1D array of shape (n_classes,) representing the physical
187
+ ordinal class labels in strictly ascending order.
188
+
189
+ Attributes
190
+ ----------
191
+ pmf : np.ndarray
192
+ A 2D float array of shape (n_samples, n_classes) containing
193
+ validated predicted class probabilities as a read-only object.
194
+ classes : np.ndarray
195
+ A 1D array of shape (n_classes,) containing the validated ordinal
196
+ class labels as a read-only object.
197
+ cdf : np.ndarray
198
+ A 2D float array of shape (n_samples, n_classes) containing
199
+ cumulative probabilities computed from `pmf`. The final column is
200
+ forced to exactly 1.0 to guard against floating-point PMF row
201
+ sums fractionally below 1.0.
202
+
203
+ Methods
204
+ -------
205
+ mean()
206
+ Calculate the expected value for each sample.
207
+ ppf(q)
208
+ Calculate the percent point function (inverse CDF / quantiles).
209
+ median()
210
+ Calculate the 50th percentile prediction for each sample.
211
+ interval(alpha=0.10)
212
+ Calculate central prediction bounds for a given significance level.
213
+
214
+ Raises
215
+ ------
216
+ ValueError
217
+ If `pmf` is not a 2D array, `classes` is not a 1D array, the
218
+ number of columns in `pmf` does not match the length of
219
+ `classes`, `classes` is not strictly ascending, or any row of
220
+ `pmf` does not sum to 1.0 within tolerance.
221
+
222
+ """
223
+
224
+ _PMF_SUM_ATOL = 1e-6
225
+
226
+ def __init__(self, pmf: np.ndarray, classes: np.ndarray) -> None:
227
+ pmf_arr = np.asarray(pmf, dtype=float)
228
+ classes_arr = np.asarray(classes)
229
+
230
+ if pmf_arr.ndim != 2:
231
+ raise ValueError(
232
+ f"Expected 'pmf' to be a 2D array of shape (n_samples, n_classes), "
233
+ f"got shape {pmf_arr.shape}."
234
+ )
235
+ if classes_arr.ndim != 1:
236
+ raise ValueError(
237
+ f"Expected 'classes' to be a 1D array, got shape {classes_arr.shape}."
238
+ )
239
+ if pmf_arr.shape[1] != classes_arr.shape[0]:
240
+ raise ValueError(
241
+ f"Mismatch between PMF class dimension ({pmf_arr.shape[1]}) "
242
+ f"and classes array length ({classes_arr.shape[0]})."
243
+ )
244
+
245
+ self._validate_strictly_ascending("classes", classes_arr.astype(float))
246
+ self._validate_normalized_pmf(pmf_arr)
247
+
248
+ self.pmf = pmf_arr.copy()
249
+ self.classes = classes_arr.copy()
250
+ self.pmf.flags.writeable = False # Convert to read-only
251
+ self.classes.flags.writeable = False # Convert to read-only
252
+ self._cdf: np.ndarray | None = None
253
+
254
+ @staticmethod
255
+ def _validate_normalized_pmf(pmf: np.ndarray, atol: float = _PMF_SUM_ATOL) -> None:
256
+ """Validate that every row of a PMF matrix sums to 1.0.
257
+
258
+ Parameters
259
+ ----------
260
+ pmf : np.ndarray
261
+ 2D array of shape (n_samples, n_classes).
262
+ atol : float, default=1e-6
263
+ Absolute tolerance for the row-sum comparison.
264
+
265
+ Raises
266
+ ------
267
+ ValueError
268
+ If any row of `pmf` does not sum to 1.0 within `atol`.
269
+
270
+ """
271
+ row_sums = pmf.sum(axis=1)
272
+ if not np.allclose(row_sums, 1.0, atol=atol):
273
+ raise ValueError(
274
+ "Every row of 'pmf' must sum to 1.0 within tolerance; "
275
+ f"got sums ranging [{row_sums.min():.6f}, {row_sums.max():.6f}]."
276
+ )
277
+
278
+ @property
279
+ def cdf(self) -> np.ndarray:
280
+ """Compute the Cumulative Distribution Function (CDF) array.
281
+
282
+ Returns
283
+ -------
284
+ np.ndarray
285
+ 2D array of shape (n_samples, n_classes) containing cumulative
286
+ probabilities. The final column is forced to exactly 1.0 to
287
+ guard against `ppf(1.0)` silently returning the wrong (lowest)
288
+ class if a PMF row's floating-point sum falls fractionally
289
+ short of 1.0.
290
+
291
+ """
292
+ if self._cdf is None:
293
+ cum = np.clip(np.cumsum(self.pmf, axis=1), 0.0, 1.0)
294
+ cum[:, -1] = 1.0
295
+ self._cdf = cum
296
+ return self._cdf
297
+
298
+ def mean(self) -> np.ndarray:
299
+ """Calculate the expected value (mean) for each sample.
300
+
301
+ Returns
302
+ -------
303
+ np.ndarray
304
+ 1D array of shape (n_samples,) representing expected values in
305
+ physical class units.
306
+
307
+ """
308
+ return np.dot(self.pmf, self.classes)
309
+
310
+ def _ppf(self, q_arr: np.ndarray) -> np.ndarray:
311
+ """Map validated quantile probabilities to discrete class levels.
312
+
313
+ Parameters
314
+ ----------
315
+ q_arr : np.ndarray
316
+ Validated quantile level(s); see `PredictiveDistribution._ppf`.
317
+
318
+ Returns
319
+ -------
320
+ np.ndarray
321
+ See `PredictiveDistribution._ppf`.
322
+
323
+ """
324
+ cdf = self.cdf
325
+ if q_arr.ndim == 0:
326
+ indices = np.argmax(cdf >= q_arr, axis=1)
327
+ return self.classes[indices]
328
+
329
+ cdf_expanded = cdf[:, np.newaxis, :]
330
+ q_expanded = q_arr[np.newaxis, :, np.newaxis]
331
+ indices = np.argmax(cdf_expanded >= q_expanded, axis=2)
332
+ return self.classes[indices]
333
+
334
+
335
+ class ContinuousPredictiveDistribution(PredictiveDistribution):
336
+ """Encapsulates a continuous predictive Cumulative Distribution Function (CDF).
337
+
338
+ Provides vectorized utilities for computing expected continuous
339
+ values, medians, percent point functions (quantiles/PPF), central
340
+ prediction intervals, and continuous CDF probabilities across samples.
341
+
342
+ Parameters
343
+ ----------
344
+ grid_y : np.ndarray
345
+ 1D float array of shape (n_grid_points,) representing continuous
346
+ physical target grid values in strictly ascending order.
347
+ grid_cdf : np.ndarray
348
+ 2D float array of shape (n_samples, n_grid_points) containing
349
+ evaluated cumulative probabilities across grid points. Every
350
+ row's first value must be 0.0 and last value must be 1.0 (within
351
+ tolerance) for `mean()` to return an unbiased estimate.
352
+
353
+ Attributes
354
+ ----------
355
+ grid_y : np.ndarray
356
+ 1D float array containing grid values as a read-only object.
357
+ grid_cdf : np.ndarray
358
+ 2D float array containing cumulative probabilities, clipped to
359
+ [0.0, 1.0] as a read-only object.
360
+
361
+ Methods
362
+ -------
363
+ mean()
364
+ Calculate expected continuous values via numerical integration.
365
+ ppf(q)
366
+ Calculate percent point function (inverse CDF / quantiles).
367
+ median()
368
+ Calculate the 50th percentile prediction for each sample.
369
+ interval(alpha=0.10)
370
+ Calculate central prediction bounds for a given significance level.
371
+ cdf(y)
372
+ Evaluate continuous CDF probability P(Y <= y) at physical value y.
373
+
374
+ Raises
375
+ ------
376
+ ValueError
377
+ If `grid_y` is not 1D, `grid_cdf` is not 2D, shape dimensions
378
+ mismatch, `grid_y` is not strictly ascending, or any row of
379
+ `grid_cdf` does not start at 0.0 or end at 1.0 within tolerance.
380
+
381
+ """
382
+
383
+ _CDF_BOUNDARY_ATOL = 1e-6
384
+
385
+ def __init__(self, grid_y: np.ndarray, grid_cdf: np.ndarray) -> None:
386
+ y_arr = np.asarray(grid_y, dtype=float)
387
+ cdf_arr = np.asarray(grid_cdf, dtype=float)
388
+
389
+ if y_arr.ndim != 1 or cdf_arr.ndim != 2:
390
+ raise ValueError("Invalid array dimensions for grid_y or grid_cdf.")
391
+ if cdf_arr.shape[1] != y_arr.shape[0]:
392
+ raise ValueError("Grid CDF column dimension must match grid_y length.")
393
+
394
+ self._validate_strictly_ascending("grid_y", y_arr)
395
+ cdf_arr = np.clip(cdf_arr, 0.0, 1.0)
396
+ self._validate_normalized_cdf_grid(cdf_arr)
397
+
398
+ self.grid_y = y_arr.copy()
399
+ self.grid_y.flags.writeable = False # Convert to read-only
400
+ self.grid_cdf = cdf_arr.copy()
401
+ self.grid_cdf.flags.writeable = False # Convert to read-only
402
+ self._n_samples = cdf_arr.shape[0]
403
+
404
+ @staticmethod
405
+ def _validate_normalized_cdf_grid(
406
+ grid_cdf: np.ndarray, atol: float = _CDF_BOUNDARY_ATOL
407
+ ) -> None:
408
+ """Validate that every row of a CDF grid starts at 0.0 and ends at 1.0.
409
+
410
+ Parameters
411
+ ----------
412
+ grid_cdf : np.ndarray
413
+ 2D array of shape (n_samples, n_grid_points), already clipped
414
+ to [0.0, 1.0].
415
+ atol : float, default=1e-6
416
+ Absolute tolerance for the boundary comparisons.
417
+
418
+ Raises
419
+ ------
420
+ ValueError
421
+ If any row's first value is not close to 0.0, or last value
422
+ is not close to 1.0.
423
+
424
+ """
425
+ if not np.allclose(grid_cdf[:, 0], 0.0, atol=atol):
426
+ raise ValueError(
427
+ "Every row of 'grid_cdf' must start at 0.0 (grid_cdf[:, 0])."
428
+ )
429
+ if not np.allclose(grid_cdf[:, -1], 1.0, atol=atol):
430
+ raise ValueError(
431
+ "Every row of 'grid_cdf' must end at 1.0 (grid_cdf[:, -1])."
432
+ )
433
+
434
+ def mean(self) -> np.ndarray:
435
+ """Calculate expected continuous values via numerical integration.
436
+
437
+ Uses the identity ``E[Y] = grid_y[0] + integral_{grid_y[0]}^{grid_y[-1]}
438
+ (1 - F(y)) dy``, evaluated by the trapezoidal rule. Correctness
439
+ relies on `grid_cdf` starting at 0.0 and ending at 1.0, which is
440
+ enforced at construction time.
441
+
442
+ Returns
443
+ -------
444
+ np.ndarray
445
+ 1D array of shape (n_samples,) containing expected physical
446
+ values.
447
+
448
+ """
449
+ dy = np.diff(self.grid_y)
450
+ avg_prob = 1.0 - 0.5 * (self.grid_cdf[:, :-1] + self.grid_cdf[:, 1:])
451
+ return np.sum(avg_prob * dy, axis=1) + self.grid_y[0]
452
+
453
+ def _ppf(self, q_arr: np.ndarray) -> np.ndarray:
454
+ """Interpolate continuous target values at validated quantile levels.
455
+
456
+ Parameters
457
+ ----------
458
+ q_arr : np.ndarray
459
+ Validated quantile level(s); see `PredictiveDistribution._ppf`.
460
+
461
+ Returns
462
+ -------
463
+ np.ndarray
464
+ See `PredictiveDistribution._ppf`.
465
+
466
+ """
467
+ n_samples, n_grid = self.grid_cdf.shape
468
+
469
+ if q_arr.ndim == 0:
470
+ q_val = float(q_arr)
471
+ idx = np.clip(
472
+ np.count_nonzero(self.grid_cdf <= q_val, axis=1) - 1, 0, n_grid - 2
473
+ )
474
+ rows = np.arange(n_samples)
475
+ q0, q1 = self.grid_cdf[rows, idx], self.grid_cdf[rows, idx + 1]
476
+ denom = q1 - q0
477
+ t = np.divide(q_val - q0, denom, out=np.zeros_like(q0), where=denom != 0)
478
+ t = np.clip(t, 0.0, 1.0)
479
+ return (1.0 - t) * self.grid_y[idx] + t * self.grid_y[idx + 1]
480
+
481
+ return np.stack([self._ppf(np.asarray(q)) for q in q_arr], axis=1)
482
+
483
+ def cdf(self, y: Union[float, ArrayLike]) -> np.ndarray:
484
+ """Evaluate the continuous CDF, P(Y <= y), at physical value(s) `y`.
485
+
486
+ Parameters
487
+ ----------
488
+ y : float | ArrayLike
489
+ Physical target value(s) at which to evaluate the CDF. A
490
+ scalar is broadcast across all samples; a 1D array of length
491
+ `n_samples` evaluates each sample at its own `y` value.
492
+
493
+ Returns
494
+ -------
495
+ np.ndarray
496
+ 1D array of shape (n_samples,) containing P(Y <= y) for each
497
+ sample, linearly interpolated (or exactly looked up) against
498
+ `grid_y`/`grid_cdf`, with flat extrapolation to 0.0 below
499
+ `grid_y[0]` and 1.0 above `grid_y[-1]`.
500
+
501
+ Raises
502
+ ------
503
+ ValueError
504
+ If `y` is neither a scalar nor a 1D array of length
505
+ `n_samples`.
506
+
507
+ """
508
+ y_arr = np.asarray(y, dtype=float)
509
+ if y_arr.ndim == 0:
510
+ return np.array(
511
+ [
512
+ np.interp(float(y_arr), self.grid_y, self.grid_cdf[i])
513
+ for i in range(self._n_samples)
514
+ ]
515
+ )
516
+ if y_arr.ndim != 1 or y_arr.shape[0] != self._n_samples:
517
+ raise ValueError(
518
+ f"Expected 'y' to be a scalar or 1D array of length "
519
+ f"{self._n_samples}, got shape {y_arr.shape}."
520
+ )
521
+ return np.array(
522
+ [
523
+ np.interp(y_arr[i], self.grid_y, self.grid_cdf[i])
524
+ for i in range(self._n_samples)
525
+ ]
526
+ )