tppis 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tppis/__init__.py +29 -0
- tppis/_validation.py +150 -0
- tppis/criteria.py +220 -0
- tppis/datasets/__init__.py +26 -0
- tppis/datasets/real.py +97 -0
- tppis/datasets/simulate.py +188 -0
- tppis/exceptions.py +56 -0
- tppis/factors.py +37 -0
- tppis/kernel.py +159 -0
- tppis/metrics.py +127 -0
- tppis/py.typed +0 -0
- tppis/screeners.py +504 -0
- tppis/spectral.py +183 -0
- tppis/tuning.py +309 -0
- tppis-0.1.0.dist-info/METADATA +154 -0
- tppis-0.1.0.dist-info/RECORD +18 -0
- tppis-0.1.0.dist-info/WHEEL +4 -0
- tppis-0.1.0.dist-info/licenses/LICENSE +21 -0
tppis/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""TPPIS: Truncated Preconditioned Profiled Independence Screening."""
|
|
2
|
+
|
|
3
|
+
from tppis.datasets import (
|
|
4
|
+
make_example1,
|
|
5
|
+
make_example2,
|
|
6
|
+
make_example3,
|
|
7
|
+
make_example4,
|
|
8
|
+
)
|
|
9
|
+
from tppis.exceptions import TPPISError
|
|
10
|
+
from tppis.metrics import screening_scores
|
|
11
|
+
from tppis.screeners import FPSIS, FPSISBIC, PPIS, SIS, TPPIS, screen
|
|
12
|
+
|
|
13
|
+
__version__ = "0.1.0"
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"FPSIS",
|
|
17
|
+
"FPSISBIC",
|
|
18
|
+
"PPIS",
|
|
19
|
+
"SIS",
|
|
20
|
+
"TPPIS",
|
|
21
|
+
"TPPISError",
|
|
22
|
+
"make_example1",
|
|
23
|
+
"make_example2",
|
|
24
|
+
"make_example3",
|
|
25
|
+
"make_example4",
|
|
26
|
+
"screen",
|
|
27
|
+
"screening_scores",
|
|
28
|
+
"__version__",
|
|
29
|
+
]
|
tppis/_validation.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Input checking, standardization and tie-stable ranking.
|
|
2
|
+
|
|
3
|
+
Standardization uses the population standard deviation (``ddof=0``), matching
|
|
4
|
+
ambiguity A-5 of the specification and ``sklearn.preprocessing.StandardScaler``.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import warnings
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
from numpy.typing import ArrayLike, NDArray
|
|
14
|
+
|
|
15
|
+
from tppis.exceptions import ConstantFeatureWarning, TPPISError
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"TPPISError",
|
|
19
|
+
"apply_standardize",
|
|
20
|
+
"as_float_arrays",
|
|
21
|
+
"check_literal",
|
|
22
|
+
"rank_by_abs",
|
|
23
|
+
"standardize",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def as_float_arrays(
|
|
28
|
+
X: ArrayLike,
|
|
29
|
+
y: ArrayLike,
|
|
30
|
+
) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
|
|
31
|
+
"""Cast ``X`` and ``y`` to contiguous float64 arrays with matching ``n``.
|
|
32
|
+
|
|
33
|
+
Parameters
|
|
34
|
+
----------
|
|
35
|
+
X :
|
|
36
|
+
Predictor matrix of shape ``(n, p)``.
|
|
37
|
+
y :
|
|
38
|
+
Response vector of shape ``(n,)``.
|
|
39
|
+
|
|
40
|
+
Returns
|
|
41
|
+
-------
|
|
42
|
+
X, y
|
|
43
|
+
Contiguous ``float64`` copies.
|
|
44
|
+
"""
|
|
45
|
+
X_arr = np.ascontiguousarray(np.asarray(X, dtype=np.float64))
|
|
46
|
+
y_arr = np.ascontiguousarray(np.asarray(y, dtype=np.float64).reshape(-1))
|
|
47
|
+
if X_arr.ndim != 2:
|
|
48
|
+
raise TPPISError(f"X must be 2-dimensional, got shape {X_arr.shape}.")
|
|
49
|
+
n, p = X_arr.shape
|
|
50
|
+
if n < 2:
|
|
51
|
+
raise TPPISError(f"n must be at least 2, got {n}.")
|
|
52
|
+
if p < 1:
|
|
53
|
+
raise TPPISError("X must have at least one column.")
|
|
54
|
+
if y_arr.shape[0] != n:
|
|
55
|
+
raise TPPISError(
|
|
56
|
+
f"X and y must share the first dimension, "
|
|
57
|
+
f"got n={n} and y.size={y_arr.size}."
|
|
58
|
+
)
|
|
59
|
+
if not np.isfinite(X_arr).all():
|
|
60
|
+
raise TPPISError("X contains non-finite values.")
|
|
61
|
+
if not np.isfinite(y_arr).all():
|
|
62
|
+
raise TPPISError("y contains non-finite values.")
|
|
63
|
+
return X_arr, y_arr
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def standardize(
|
|
67
|
+
X: NDArray[np.float64],
|
|
68
|
+
y: NDArray[np.float64],
|
|
69
|
+
*,
|
|
70
|
+
enabled: bool = True,
|
|
71
|
+
) -> tuple[
|
|
72
|
+
NDArray[np.float64],
|
|
73
|
+
NDArray[np.float64],
|
|
74
|
+
NDArray[np.float64],
|
|
75
|
+
NDArray[np.float64],
|
|
76
|
+
float,
|
|
77
|
+
]:
|
|
78
|
+
"""Center ``y`` and optionally column-standardize ``X`` with ``ddof=0``.
|
|
79
|
+
|
|
80
|
+
Parameters
|
|
81
|
+
----------
|
|
82
|
+
X, y :
|
|
83
|
+
Validated arrays from :func:`as_float_arrays`.
|
|
84
|
+
enabled :
|
|
85
|
+
If False, ``X`` is returned unchanged and column scales are ones.
|
|
86
|
+
|
|
87
|
+
Returns
|
|
88
|
+
-------
|
|
89
|
+
X_std, y_c, x_mean, x_scale, y_mean
|
|
90
|
+
"""
|
|
91
|
+
y_mean = float(y.mean())
|
|
92
|
+
y_c = y - y_mean
|
|
93
|
+
x_mean = X.mean(axis=0)
|
|
94
|
+
if not enabled:
|
|
95
|
+
return X, y_c, x_mean, np.ones(X.shape[1], dtype=np.float64), y_mean
|
|
96
|
+
x_scale = X.std(axis=0, ddof=0)
|
|
97
|
+
zero = x_scale <= np.finfo(np.float64).eps * max(X.shape)
|
|
98
|
+
if np.any(zero):
|
|
99
|
+
warnings.warn(
|
|
100
|
+
f"{int(zero.sum())} constant column(s) left unscaled.",
|
|
101
|
+
ConstantFeatureWarning,
|
|
102
|
+
stacklevel=2,
|
|
103
|
+
)
|
|
104
|
+
x_scale = x_scale.copy()
|
|
105
|
+
x_scale[zero] = 1.0
|
|
106
|
+
X_std = (X - x_mean) / x_scale
|
|
107
|
+
return X_std, y_c, x_mean, x_scale, y_mean
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def rank_by_abs(
|
|
111
|
+
omega: NDArray[np.float64],
|
|
112
|
+
) -> NDArray[np.intp]:
|
|
113
|
+
"""Descending ``|omega|`` order, ties broken by ascending column index (A-7).
|
|
114
|
+
|
|
115
|
+
Parameters
|
|
116
|
+
----------
|
|
117
|
+
omega :
|
|
118
|
+
Marginal importance scores of length ``p``.
|
|
119
|
+
|
|
120
|
+
Returns
|
|
121
|
+
-------
|
|
122
|
+
order
|
|
123
|
+
Integer indices of shape ``(p,)``.
|
|
124
|
+
"""
|
|
125
|
+
p = omega.shape[0]
|
|
126
|
+
keys = np.empty((p, 2), dtype=np.float64)
|
|
127
|
+
keys[:, 0] = -np.abs(omega)
|
|
128
|
+
keys[:, 1] = np.arange(p, dtype=np.float64)
|
|
129
|
+
return np.lexsort((keys[:, 1], keys[:, 0])).astype(np.intp)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def apply_standardize(
|
|
133
|
+
X: ArrayLike,
|
|
134
|
+
x_mean: NDArray[np.float64],
|
|
135
|
+
x_scale: NDArray[np.float64],
|
|
136
|
+
) -> NDArray[np.float64]:
|
|
137
|
+
"""Apply a previously fitted column standardization to new ``X``."""
|
|
138
|
+
X_arr = np.ascontiguousarray(np.asarray(X, dtype=np.float64))
|
|
139
|
+
if X_arr.ndim != 2:
|
|
140
|
+
raise TPPISError(f"X must be 2-dimensional, got shape {X_arr.shape}.")
|
|
141
|
+
if X_arr.shape[1] != x_mean.shape[0]:
|
|
142
|
+
raise TPPISError(f"X has {X_arr.shape[1]} columns, expected {x_mean.shape[0]}.")
|
|
143
|
+
return (X_arr - x_mean) / x_scale
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def check_literal(name: str, value: Any, allowed: set[str]) -> str:
|
|
147
|
+
"""Validate a string option against a closed set."""
|
|
148
|
+
if value not in allowed:
|
|
149
|
+
raise TPPISError(f"{name} must be one of {sorted(allowed)}, got {value!r}.")
|
|
150
|
+
return str(value)
|
tppis/criteria.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""Refit and the BIC-type criterion (10).
|
|
2
|
+
|
|
3
|
+
The coefficient inside the criterion is ordinary least squares of the original
|
|
4
|
+
columns on the centered response. That is the residual whose value matches
|
|
5
|
+
Tables 1–4. Equation (8) fits the transformed design instead, and that residual
|
|
6
|
+
does not.
|
|
7
|
+
|
|
8
|
+
The Gram matrices for ``k = 1, 2, ...`` are nested leading blocks, so an
|
|
9
|
+
incremental Cholesky sweep produces every ``beta_hat(M_k)`` in ``O(K^3 / 3)``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import warnings
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
from numpy.typing import NDArray
|
|
19
|
+
from scipy.linalg import solve_triangular
|
|
20
|
+
|
|
21
|
+
from tppis._validation import check_literal
|
|
22
|
+
from tppis.exceptions import BoundarySelectionWarning, TPPISError, TppisWarning
|
|
23
|
+
|
|
24
|
+
# Floor for log(RSS). Combined with A-9 this keeps BIC finite.
|
|
25
|
+
_RSS_FLOOR = 1e-300
|
|
26
|
+
# Relative pivot tolerance for the incremental Cholesky path.
|
|
27
|
+
_CHOL_EPS = 1e-12
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class CriterionSweep:
|
|
32
|
+
"""BIC path over ``k = 1 .. K`` for one ``(d, alpha)`` pair."""
|
|
33
|
+
|
|
34
|
+
k: NDArray[np.intp]
|
|
35
|
+
bic: NDArray[np.float64]
|
|
36
|
+
rss: NDArray[np.float64]
|
|
37
|
+
best_k: int
|
|
38
|
+
best_bic: float
|
|
39
|
+
best_beta: NDArray[np.float64]
|
|
40
|
+
on_boundary: bool
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def bic_value(
|
|
44
|
+
rss: float,
|
|
45
|
+
k: int,
|
|
46
|
+
n: int,
|
|
47
|
+
p: int,
|
|
48
|
+
*,
|
|
49
|
+
kind: str = "paper",
|
|
50
|
+
) -> float:
|
|
51
|
+
"""BIC-type criterion (10).
|
|
52
|
+
|
|
53
|
+
Parameters
|
|
54
|
+
----------
|
|
55
|
+
rss :
|
|
56
|
+
``|| y - X(M_k) @ beta_hat(M_k) ||^2`` on the original scale.
|
|
57
|
+
k, n, p :
|
|
58
|
+
Model size, sample size, number of predictors.
|
|
59
|
+
kind :
|
|
60
|
+
``"paper"`` is ``log(RSS) + (log(p)/n) * k * log(n)`` (A-4).
|
|
61
|
+
``"rss_mean"`` replaces ``RSS`` with ``RSS / n``.
|
|
62
|
+
|
|
63
|
+
Returns
|
|
64
|
+
-------
|
|
65
|
+
bic
|
|
66
|
+
Scalar criterion value. Natural logarithm.
|
|
67
|
+
"""
|
|
68
|
+
kind = check_literal("bic", kind, {"paper", "rss_mean"})
|
|
69
|
+
if rss < _RSS_FLOOR:
|
|
70
|
+
warnings.warn(
|
|
71
|
+
f"RSS={rss} was clamped at {_RSS_FLOOR} before taking the log.",
|
|
72
|
+
TppisWarning,
|
|
73
|
+
stacklevel=2,
|
|
74
|
+
)
|
|
75
|
+
rss = _RSS_FLOOR
|
|
76
|
+
scale = rss if kind == "paper" else rss / n
|
|
77
|
+
return float(np.log(scale) + (np.log(p) / n) * k * np.log(n))
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def default_k_max(p: int, n: int, n_retained: int, override: int | None) -> int:
|
|
81
|
+
"""A-9 cap ``k_max = min(p, n_retained, n - 1)``.
|
|
82
|
+
|
|
83
|
+
``k <= p - 1`` is not the constraint that keeps least squares valid.
|
|
84
|
+
After centering, the columns of ``X`` span at most ``n - 1`` dimensions,
|
|
85
|
+
so ``k >= n`` interpolates ``y`` and ``log(RSS)`` is unbounded below.
|
|
86
|
+
An override at or above ``n`` is rejected. An override above the retained
|
|
87
|
+
rank is still clipped to that rank.
|
|
88
|
+
"""
|
|
89
|
+
cap = min(p, n_retained, n - 1)
|
|
90
|
+
if cap < 1:
|
|
91
|
+
raise TPPISError(
|
|
92
|
+
f"k_max collapsed to {cap}; n_retained={n_retained}, n={n}, p={p}."
|
|
93
|
+
)
|
|
94
|
+
if override is None:
|
|
95
|
+
return cap
|
|
96
|
+
if override < 1:
|
|
97
|
+
raise TPPISError(f"k_max must be at least 1, got {override}.")
|
|
98
|
+
if int(override) >= n:
|
|
99
|
+
raise TPPISError(
|
|
100
|
+
f"k_max={override} must be at most n-1={n - 1}. "
|
|
101
|
+
"A column-centered design has rank at most n-1, so k >= n makes "
|
|
102
|
+
"log(RSS) unbounded below. k <= p-1 is not that constraint."
|
|
103
|
+
)
|
|
104
|
+
return min(int(override), cap)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _chol_solve(
|
|
108
|
+
L: NDArray[np.float64], rhs: NDArray[np.float64]
|
|
109
|
+
) -> NDArray[np.float64]:
|
|
110
|
+
z = solve_triangular(L, rhs, lower=True, check_finite=False)
|
|
111
|
+
beta = solve_triangular(L.T, z, lower=False, check_finite=False)
|
|
112
|
+
return np.asarray(beta, dtype=np.float64)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _min_norm(
|
|
116
|
+
gram: NDArray[np.float64], rhs: NDArray[np.float64]
|
|
117
|
+
) -> NDArray[np.float64]:
|
|
118
|
+
"""Minimum-norm least-squares solve of a possibly singular Gram block (A-9)."""
|
|
119
|
+
beta, *_ = np.linalg.lstsq(gram, rhs, rcond=None)
|
|
120
|
+
return np.asarray(beta, dtype=np.float64)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def incremental_refit(
|
|
124
|
+
F: NDArray[np.float64],
|
|
125
|
+
rhs: NDArray[np.float64],
|
|
126
|
+
X_sel: NDArray[np.float64],
|
|
127
|
+
y: NDArray[np.float64],
|
|
128
|
+
p: int,
|
|
129
|
+
*,
|
|
130
|
+
kind: str = "paper",
|
|
131
|
+
) -> CriterionSweep:
|
|
132
|
+
"""Sweep ``k = 1 .. K`` with nested Cholesky updates of equation (8).
|
|
133
|
+
|
|
134
|
+
Parameters
|
|
135
|
+
----------
|
|
136
|
+
F :
|
|
137
|
+
Factor matrix of shape ``(K, s)`` such that
|
|
138
|
+
``Gram_k = F[:k] @ F[:k].T``. For TPPIS/PPIS this is ``V_S(M)``;
|
|
139
|
+
for FPSIS it is ``V_S(M) * mu_S``; for SIS it is ``X(M).T``.
|
|
140
|
+
rhs :
|
|
141
|
+
``X_hat(M_K).T @ y_hat`` restricted to the ranked columns, length ``K``.
|
|
142
|
+
Equals ``omega[order[:K]]``.
|
|
143
|
+
X_sel, y :
|
|
144
|
+
Ranked original (standardized) columns and the centered response, used
|
|
145
|
+
only for the residual of equation (10).
|
|
146
|
+
p :
|
|
147
|
+
Original number of predictors, for the BIC penalty.
|
|
148
|
+
kind :
|
|
149
|
+
Criterion flavour, see :func:`bic_value`.
|
|
150
|
+
|
|
151
|
+
Returns
|
|
152
|
+
-------
|
|
153
|
+
CriterionSweep
|
|
154
|
+
Full path and the minimizing ``k``.
|
|
155
|
+
"""
|
|
156
|
+
K = int(F.shape[0])
|
|
157
|
+
if K < 1:
|
|
158
|
+
raise TPPISError("Cannot sweep an empty candidate set.")
|
|
159
|
+
n = int(y.shape[0])
|
|
160
|
+
L = np.zeros((K, K), dtype=np.float64)
|
|
161
|
+
bic = np.empty(K, dtype=np.float64)
|
|
162
|
+
rss = np.empty(K, dtype=np.float64)
|
|
163
|
+
best_k = 1
|
|
164
|
+
best_bic = np.inf
|
|
165
|
+
best_beta = np.empty(0, dtype=np.float64)
|
|
166
|
+
use_direct = False
|
|
167
|
+
|
|
168
|
+
for k in range(1, K + 1):
|
|
169
|
+
gk = F[:k] @ F[k - 1]
|
|
170
|
+
if not use_direct:
|
|
171
|
+
if k == 1:
|
|
172
|
+
pivot = float(gk[0])
|
|
173
|
+
if pivot <= _CHOL_EPS * max(1.0, float(np.abs(F[0] @ F[0]))):
|
|
174
|
+
use_direct = True
|
|
175
|
+
else:
|
|
176
|
+
L[0, 0] = np.sqrt(pivot)
|
|
177
|
+
else:
|
|
178
|
+
try:
|
|
179
|
+
ell = solve_triangular(
|
|
180
|
+
L[: k - 1, : k - 1], gk[: k - 1], lower=True, check_finite=False
|
|
181
|
+
)
|
|
182
|
+
rem = float(gk[k - 1] - ell @ ell)
|
|
183
|
+
if rem <= _CHOL_EPS * max(1.0, float(np.abs(gk[k - 1]))):
|
|
184
|
+
use_direct = True
|
|
185
|
+
else:
|
|
186
|
+
L[k - 1, : k - 1] = ell
|
|
187
|
+
L[k - 1, k - 1] = np.sqrt(rem)
|
|
188
|
+
except np.linalg.LinAlgError:
|
|
189
|
+
use_direct = True
|
|
190
|
+
gram = F[:k] @ F[:k].T
|
|
191
|
+
if use_direct:
|
|
192
|
+
beta = _min_norm(gram, rhs[:k])
|
|
193
|
+
else:
|
|
194
|
+
beta = _chol_solve(L[:k, :k], rhs[:k])
|
|
195
|
+
resid = y - X_sel[:, :k] @ beta
|
|
196
|
+
rss_k = float(resid @ resid)
|
|
197
|
+
bic_k = bic_value(rss_k, k, n, p, kind=kind)
|
|
198
|
+
rss[k - 1] = rss_k
|
|
199
|
+
bic[k - 1] = bic_k
|
|
200
|
+
if bic_k < best_bic:
|
|
201
|
+
best_bic = bic_k
|
|
202
|
+
best_k = k
|
|
203
|
+
best_beta = beta
|
|
204
|
+
|
|
205
|
+
on_boundary = best_k in {1, K}
|
|
206
|
+
if on_boundary:
|
|
207
|
+
warnings.warn(
|
|
208
|
+
f"BIC minimum landed on the k-boundary at k={best_k} (k_max={K}).",
|
|
209
|
+
BoundarySelectionWarning,
|
|
210
|
+
stacklevel=2,
|
|
211
|
+
)
|
|
212
|
+
return CriterionSweep(
|
|
213
|
+
k=np.arange(1, K + 1, dtype=np.intp),
|
|
214
|
+
bic=bic,
|
|
215
|
+
rss=rss,
|
|
216
|
+
best_k=best_k,
|
|
217
|
+
best_bic=best_bic,
|
|
218
|
+
best_beta=best_beta,
|
|
219
|
+
on_boundary=on_boundary,
|
|
220
|
+
)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Simulation generators and real-data fetchers."""
|
|
2
|
+
|
|
3
|
+
from tppis.datasets.real import fetch_hydraulic, fetch_sp500_info
|
|
4
|
+
from tppis.datasets.simulate import (
|
|
5
|
+
SimulatedData,
|
|
6
|
+
example1_sigma,
|
|
7
|
+
example2_sigma,
|
|
8
|
+
make_example1,
|
|
9
|
+
make_example2,
|
|
10
|
+
make_example3,
|
|
11
|
+
make_example4,
|
|
12
|
+
min_eigenvalue,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"SimulatedData",
|
|
17
|
+
"example1_sigma",
|
|
18
|
+
"example2_sigma",
|
|
19
|
+
"fetch_hydraulic",
|
|
20
|
+
"fetch_sp500_info",
|
|
21
|
+
"make_example1",
|
|
22
|
+
"make_example2",
|
|
23
|
+
"make_example3",
|
|
24
|
+
"make_example4",
|
|
25
|
+
"min_eigenvalue",
|
|
26
|
+
]
|
tppis/datasets/real.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Fetch scripts for the two real datasets in Section 5 of the paper.
|
|
2
|
+
|
|
3
|
+
Neither dataset is redistributed. These helpers download from the public
|
|
4
|
+
sources the paper cites, cache locally, and raise a clear error if a source
|
|
5
|
+
is unavailable.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import os
|
|
12
|
+
import urllib.request
|
|
13
|
+
from collections.abc import Callable
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
DEFAULT_CACHE = Path(os.environ.get("TPPIS_DATA", Path.home() / ".tppis_data"))
|
|
17
|
+
|
|
18
|
+
# Sources as cited in Tanaka and Matsui (2023), Section 5.
|
|
19
|
+
HYDRAULIC_INFO = {
|
|
20
|
+
"name": "condition monitoring of hydraulic systems",
|
|
21
|
+
"paper_ref": "[17] Helwig, Pignanelli, Schutze, I2MTC 2015",
|
|
22
|
+
"url": "https://archive.ics.uci.edu/static/public/447/condition+monitoring+of+hydraulic+systems.zip",
|
|
23
|
+
"notes": (
|
|
24
|
+
"The paper uses n=1449 rows taken under stable system settings, "
|
|
25
|
+
"response = accumulator pressure (130/115/100/90), p=43680 from 17 sensors."
|
|
26
|
+
),
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
SP500_INFO = {
|
|
30
|
+
"name": "S&P 500, year 2020",
|
|
31
|
+
"paper_ref": "[18] FRED SP500; [19] Kaggle S&P 500 stocks",
|
|
32
|
+
"fred_url": "https://fred.stlouisfed.org/series/SP500",
|
|
33
|
+
"kaggle_url": "https://www.kaggle.com/hanseopark/sp-500-stocks-value-with-financial-statement",
|
|
34
|
+
"notes": (
|
|
35
|
+
"The paper uses 253 trading days in 2020. Response = S&P 500 index; "
|
|
36
|
+
"predictors = constituent stock prices (some companies contribute more "
|
|
37
|
+
"than one series)."
|
|
38
|
+
),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _sha256(path: Path) -> str:
|
|
43
|
+
digest = hashlib.sha256()
|
|
44
|
+
with path.open("rb") as handle:
|
|
45
|
+
for chunk in iter(lambda: handle.read(1 << 20), b""):
|
|
46
|
+
digest.update(chunk)
|
|
47
|
+
return digest.hexdigest()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _download(url: str, dest: Path, *, timeout: int = 60) -> Path:
|
|
51
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
52
|
+
if dest.exists():
|
|
53
|
+
return dest
|
|
54
|
+
tmp = dest.with_suffix(dest.suffix + ".part")
|
|
55
|
+
try:
|
|
56
|
+
with urllib.request.urlopen(url, timeout=timeout) as src, tmp.open("wb") as out:
|
|
57
|
+
out.write(src.read())
|
|
58
|
+
tmp.replace(dest)
|
|
59
|
+
except Exception as exc: # noqa: BLE001
|
|
60
|
+
if tmp.exists():
|
|
61
|
+
tmp.unlink()
|
|
62
|
+
raise RuntimeError(
|
|
63
|
+
f"Could not download {url}. Fetch the file by hand and place it at {dest}."
|
|
64
|
+
) from exc
|
|
65
|
+
return dest
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def fetch_hydraulic(
|
|
69
|
+
cache_dir: Path | None = None,
|
|
70
|
+
*,
|
|
71
|
+
downloader: Callable[[str, Path], Path] | None = None,
|
|
72
|
+
) -> Path:
|
|
73
|
+
"""Download the UCI hydraulic-systems archive into ``cache_dir``.
|
|
74
|
+
|
|
75
|
+
Returns the path to the zip. Parsing into the paper's ``(X, y)`` layout is
|
|
76
|
+
left to ``reproduction/run_real_data.py``, because the 17-sensor expansion
|
|
77
|
+
is study-specific.
|
|
78
|
+
"""
|
|
79
|
+
root = cache_dir or DEFAULT_CACHE
|
|
80
|
+
dest = root / "hydraulic" / "hydraulic_systems.zip"
|
|
81
|
+
fetch = downloader or (lambda url, path: _download(url, path))
|
|
82
|
+
return fetch(HYDRAULIC_INFO["url"], dest)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def fetch_sp500_info() -> dict[str, str]:
|
|
86
|
+
"""Return the documented S&P 500 source URLs. No automatic download.
|
|
87
|
+
|
|
88
|
+
The Kaggle source requires an account, so this function only documents
|
|
89
|
+
the locations. ``reproduction/run_real_data.py`` reads a user-supplied
|
|
90
|
+
directory of CSVs.
|
|
91
|
+
"""
|
|
92
|
+
return dict(SP500_INFO)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def verify_checksum(path: Path, expected: str) -> bool:
|
|
96
|
+
"""Compare a file to a published SHA-256 hex digest."""
|
|
97
|
+
return _sha256(path) == expected.lower()
|