tppis 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tppis/__init__.py ADDED
@@ -0,0 +1,29 @@
1
+ """TPPIS: Truncated Preconditioned Profiled Independence Screening."""
2
+
3
+ from tppis.datasets import (
4
+ make_example1,
5
+ make_example2,
6
+ make_example3,
7
+ make_example4,
8
+ )
9
+ from tppis.exceptions import TPPISError
10
+ from tppis.metrics import screening_scores
11
+ from tppis.screeners import FPSIS, FPSISBIC, PPIS, SIS, TPPIS, screen
12
+
13
+ __version__ = "0.1.0"
14
+
15
+ __all__ = [
16
+ "FPSIS",
17
+ "FPSISBIC",
18
+ "PPIS",
19
+ "SIS",
20
+ "TPPIS",
21
+ "TPPISError",
22
+ "make_example1",
23
+ "make_example2",
24
+ "make_example3",
25
+ "make_example4",
26
+ "screen",
27
+ "screening_scores",
28
+ "__version__",
29
+ ]
tppis/_validation.py ADDED
@@ -0,0 +1,150 @@
1
+ """Input checking, standardization and tie-stable ranking.
2
+
3
+ Standardization uses the population standard deviation (``ddof=0``), matching
4
+ ambiguity A-5 of the specification and ``sklearn.preprocessing.StandardScaler``.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import warnings
10
+ from typing import Any
11
+
12
+ import numpy as np
13
+ from numpy.typing import ArrayLike, NDArray
14
+
15
+ from tppis.exceptions import ConstantFeatureWarning, TPPISError
16
+
17
+ __all__ = [
18
+ "TPPISError",
19
+ "apply_standardize",
20
+ "as_float_arrays",
21
+ "check_literal",
22
+ "rank_by_abs",
23
+ "standardize",
24
+ ]
25
+
26
+
27
+ def as_float_arrays(
28
+ X: ArrayLike,
29
+ y: ArrayLike,
30
+ ) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
31
+ """Cast ``X`` and ``y`` to contiguous float64 arrays with matching ``n``.
32
+
33
+ Parameters
34
+ ----------
35
+ X :
36
+ Predictor matrix of shape ``(n, p)``.
37
+ y :
38
+ Response vector of shape ``(n,)``.
39
+
40
+ Returns
41
+ -------
42
+ X, y
43
+ Contiguous ``float64`` copies.
44
+ """
45
+ X_arr = np.ascontiguousarray(np.asarray(X, dtype=np.float64))
46
+ y_arr = np.ascontiguousarray(np.asarray(y, dtype=np.float64).reshape(-1))
47
+ if X_arr.ndim != 2:
48
+ raise TPPISError(f"X must be 2-dimensional, got shape {X_arr.shape}.")
49
+ n, p = X_arr.shape
50
+ if n < 2:
51
+ raise TPPISError(f"n must be at least 2, got {n}.")
52
+ if p < 1:
53
+ raise TPPISError("X must have at least one column.")
54
+ if y_arr.shape[0] != n:
55
+ raise TPPISError(
56
+ f"X and y must share the first dimension, "
57
+ f"got n={n} and y.size={y_arr.size}."
58
+ )
59
+ if not np.isfinite(X_arr).all():
60
+ raise TPPISError("X contains non-finite values.")
61
+ if not np.isfinite(y_arr).all():
62
+ raise TPPISError("y contains non-finite values.")
63
+ return X_arr, y_arr
64
+
65
+
66
+ def standardize(
67
+ X: NDArray[np.float64],
68
+ y: NDArray[np.float64],
69
+ *,
70
+ enabled: bool = True,
71
+ ) -> tuple[
72
+ NDArray[np.float64],
73
+ NDArray[np.float64],
74
+ NDArray[np.float64],
75
+ NDArray[np.float64],
76
+ float,
77
+ ]:
78
+ """Center ``y`` and optionally column-standardize ``X`` with ``ddof=0``.
79
+
80
+ Parameters
81
+ ----------
82
+ X, y :
83
+ Validated arrays from :func:`as_float_arrays`.
84
+ enabled :
85
+ If False, ``X`` is returned unchanged and column scales are ones.
86
+
87
+ Returns
88
+ -------
89
+ X_std, y_c, x_mean, x_scale, y_mean
90
+ """
91
+ y_mean = float(y.mean())
92
+ y_c = y - y_mean
93
+ x_mean = X.mean(axis=0)
94
+ if not enabled:
95
+ return X, y_c, x_mean, np.ones(X.shape[1], dtype=np.float64), y_mean
96
+ x_scale = X.std(axis=0, ddof=0)
97
+ zero = x_scale <= np.finfo(np.float64).eps * max(X.shape)
98
+ if np.any(zero):
99
+ warnings.warn(
100
+ f"{int(zero.sum())} constant column(s) left unscaled.",
101
+ ConstantFeatureWarning,
102
+ stacklevel=2,
103
+ )
104
+ x_scale = x_scale.copy()
105
+ x_scale[zero] = 1.0
106
+ X_std = (X - x_mean) / x_scale
107
+ return X_std, y_c, x_mean, x_scale, y_mean
108
+
109
+
110
+ def rank_by_abs(
111
+ omega: NDArray[np.float64],
112
+ ) -> NDArray[np.intp]:
113
+ """Descending ``|omega|`` order, ties broken by ascending column index (A-7).
114
+
115
+ Parameters
116
+ ----------
117
+ omega :
118
+ Marginal importance scores of length ``p``.
119
+
120
+ Returns
121
+ -------
122
+ order
123
+ Integer indices of shape ``(p,)``.
124
+ """
125
+ p = omega.shape[0]
126
+ keys = np.empty((p, 2), dtype=np.float64)
127
+ keys[:, 0] = -np.abs(omega)
128
+ keys[:, 1] = np.arange(p, dtype=np.float64)
129
+ return np.lexsort((keys[:, 1], keys[:, 0])).astype(np.intp)
130
+
131
+
132
+ def apply_standardize(
133
+ X: ArrayLike,
134
+ x_mean: NDArray[np.float64],
135
+ x_scale: NDArray[np.float64],
136
+ ) -> NDArray[np.float64]:
137
+ """Apply a previously fitted column standardization to new ``X``."""
138
+ X_arr = np.ascontiguousarray(np.asarray(X, dtype=np.float64))
139
+ if X_arr.ndim != 2:
140
+ raise TPPISError(f"X must be 2-dimensional, got shape {X_arr.shape}.")
141
+ if X_arr.shape[1] != x_mean.shape[0]:
142
+ raise TPPISError(f"X has {X_arr.shape[1]} columns, expected {x_mean.shape[0]}.")
143
+ return (X_arr - x_mean) / x_scale
144
+
145
+
146
+ def check_literal(name: str, value: Any, allowed: set[str]) -> str:
147
+ """Validate a string option against a closed set."""
148
+ if value not in allowed:
149
+ raise TPPISError(f"{name} must be one of {sorted(allowed)}, got {value!r}.")
150
+ return str(value)
tppis/criteria.py ADDED
@@ -0,0 +1,220 @@
1
+ """Refit and the BIC-type criterion (10).
2
+
3
+ The coefficient inside the criterion is ordinary least squares of the original
4
+ columns on the centered response. That is the residual whose value matches
5
+ Tables 1–4. Equation (8) fits the transformed design instead, and that residual
6
+ does not.
7
+
8
+ The Gram matrices for ``k = 1, 2, ...`` are nested leading blocks, so an
9
+ incremental Cholesky sweep produces every ``beta_hat(M_k)`` in ``O(K^3 / 3)``.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import warnings
15
+ from dataclasses import dataclass
16
+
17
+ import numpy as np
18
+ from numpy.typing import NDArray
19
+ from scipy.linalg import solve_triangular
20
+
21
+ from tppis._validation import check_literal
22
+ from tppis.exceptions import BoundarySelectionWarning, TPPISError, TppisWarning
23
+
24
+ # Floor for log(RSS). Combined with A-9 this keeps BIC finite.
25
+ _RSS_FLOOR = 1e-300
26
+ # Relative pivot tolerance for the incremental Cholesky path.
27
+ _CHOL_EPS = 1e-12
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class CriterionSweep:
32
+ """BIC path over ``k = 1 .. K`` for one ``(d, alpha)`` pair."""
33
+
34
+ k: NDArray[np.intp]
35
+ bic: NDArray[np.float64]
36
+ rss: NDArray[np.float64]
37
+ best_k: int
38
+ best_bic: float
39
+ best_beta: NDArray[np.float64]
40
+ on_boundary: bool
41
+
42
+
43
+ def bic_value(
44
+ rss: float,
45
+ k: int,
46
+ n: int,
47
+ p: int,
48
+ *,
49
+ kind: str = "paper",
50
+ ) -> float:
51
+ """BIC-type criterion (10).
52
+
53
+ Parameters
54
+ ----------
55
+ rss :
56
+ ``|| y - X(M_k) @ beta_hat(M_k) ||^2`` on the original scale.
57
+ k, n, p :
58
+ Model size, sample size, number of predictors.
59
+ kind :
60
+ ``"paper"`` is ``log(RSS) + (log(p)/n) * k * log(n)`` (A-4).
61
+ ``"rss_mean"`` replaces ``RSS`` with ``RSS / n``.
62
+
63
+ Returns
64
+ -------
65
+ bic
66
+ Scalar criterion value. Natural logarithm.
67
+ """
68
+ kind = check_literal("bic", kind, {"paper", "rss_mean"})
69
+ if rss < _RSS_FLOOR:
70
+ warnings.warn(
71
+ f"RSS={rss} was clamped at {_RSS_FLOOR} before taking the log.",
72
+ TppisWarning,
73
+ stacklevel=2,
74
+ )
75
+ rss = _RSS_FLOOR
76
+ scale = rss if kind == "paper" else rss / n
77
+ return float(np.log(scale) + (np.log(p) / n) * k * np.log(n))
78
+
79
+
80
+ def default_k_max(p: int, n: int, n_retained: int, override: int | None) -> int:
81
+ """A-9 cap ``k_max = min(p, n_retained, n - 1)``.
82
+
83
+ ``k <= p - 1`` is not the constraint that keeps least squares valid.
84
+ After centering, the columns of ``X`` span at most ``n - 1`` dimensions,
85
+ so ``k >= n`` interpolates ``y`` and ``log(RSS)`` is unbounded below.
86
+ An override at or above ``n`` is rejected. An override above the retained
87
+ rank is still clipped to that rank.
88
+ """
89
+ cap = min(p, n_retained, n - 1)
90
+ if cap < 1:
91
+ raise TPPISError(
92
+ f"k_max collapsed to {cap}; n_retained={n_retained}, n={n}, p={p}."
93
+ )
94
+ if override is None:
95
+ return cap
96
+ if override < 1:
97
+ raise TPPISError(f"k_max must be at least 1, got {override}.")
98
+ if int(override) >= n:
99
+ raise TPPISError(
100
+ f"k_max={override} must be at most n-1={n - 1}. "
101
+ "A column-centered design has rank at most n-1, so k >= n makes "
102
+ "log(RSS) unbounded below. k <= p-1 is not that constraint."
103
+ )
104
+ return min(int(override), cap)
105
+
106
+
107
+ def _chol_solve(
108
+ L: NDArray[np.float64], rhs: NDArray[np.float64]
109
+ ) -> NDArray[np.float64]:
110
+ z = solve_triangular(L, rhs, lower=True, check_finite=False)
111
+ beta = solve_triangular(L.T, z, lower=False, check_finite=False)
112
+ return np.asarray(beta, dtype=np.float64)
113
+
114
+
115
+ def _min_norm(
116
+ gram: NDArray[np.float64], rhs: NDArray[np.float64]
117
+ ) -> NDArray[np.float64]:
118
+ """Minimum-norm least-squares solve of a possibly singular Gram block (A-9)."""
119
+ beta, *_ = np.linalg.lstsq(gram, rhs, rcond=None)
120
+ return np.asarray(beta, dtype=np.float64)
121
+
122
+
123
+ def incremental_refit(
124
+ F: NDArray[np.float64],
125
+ rhs: NDArray[np.float64],
126
+ X_sel: NDArray[np.float64],
127
+ y: NDArray[np.float64],
128
+ p: int,
129
+ *,
130
+ kind: str = "paper",
131
+ ) -> CriterionSweep:
132
+ """Sweep ``k = 1 .. K`` with nested Cholesky updates of equation (8).
133
+
134
+ Parameters
135
+ ----------
136
+ F :
137
+ Factor matrix of shape ``(K, s)`` such that
138
+ ``Gram_k = F[:k] @ F[:k].T``. For TPPIS/PPIS this is ``V_S(M)``;
139
+ for FPSIS it is ``V_S(M) * mu_S``; for SIS it is ``X(M).T``.
140
+ rhs :
141
+ ``X_hat(M_K).T @ y_hat`` restricted to the ranked columns, length ``K``.
142
+ Equals ``omega[order[:K]]``.
143
+ X_sel, y :
144
+ Ranked original (standardized) columns and the centered response, used
145
+ only for the residual of equation (10).
146
+ p :
147
+ Original number of predictors, for the BIC penalty.
148
+ kind :
149
+ Criterion flavour, see :func:`bic_value`.
150
+
151
+ Returns
152
+ -------
153
+ CriterionSweep
154
+ Full path and the minimizing ``k``.
155
+ """
156
+ K = int(F.shape[0])
157
+ if K < 1:
158
+ raise TPPISError("Cannot sweep an empty candidate set.")
159
+ n = int(y.shape[0])
160
+ L = np.zeros((K, K), dtype=np.float64)
161
+ bic = np.empty(K, dtype=np.float64)
162
+ rss = np.empty(K, dtype=np.float64)
163
+ best_k = 1
164
+ best_bic = np.inf
165
+ best_beta = np.empty(0, dtype=np.float64)
166
+ use_direct = False
167
+
168
+ for k in range(1, K + 1):
169
+ gk = F[:k] @ F[k - 1]
170
+ if not use_direct:
171
+ if k == 1:
172
+ pivot = float(gk[0])
173
+ if pivot <= _CHOL_EPS * max(1.0, float(np.abs(F[0] @ F[0]))):
174
+ use_direct = True
175
+ else:
176
+ L[0, 0] = np.sqrt(pivot)
177
+ else:
178
+ try:
179
+ ell = solve_triangular(
180
+ L[: k - 1, : k - 1], gk[: k - 1], lower=True, check_finite=False
181
+ )
182
+ rem = float(gk[k - 1] - ell @ ell)
183
+ if rem <= _CHOL_EPS * max(1.0, float(np.abs(gk[k - 1]))):
184
+ use_direct = True
185
+ else:
186
+ L[k - 1, : k - 1] = ell
187
+ L[k - 1, k - 1] = np.sqrt(rem)
188
+ except np.linalg.LinAlgError:
189
+ use_direct = True
190
+ gram = F[:k] @ F[:k].T
191
+ if use_direct:
192
+ beta = _min_norm(gram, rhs[:k])
193
+ else:
194
+ beta = _chol_solve(L[:k, :k], rhs[:k])
195
+ resid = y - X_sel[:, :k] @ beta
196
+ rss_k = float(resid @ resid)
197
+ bic_k = bic_value(rss_k, k, n, p, kind=kind)
198
+ rss[k - 1] = rss_k
199
+ bic[k - 1] = bic_k
200
+ if bic_k < best_bic:
201
+ best_bic = bic_k
202
+ best_k = k
203
+ best_beta = beta
204
+
205
+ on_boundary = best_k in {1, K}
206
+ if on_boundary:
207
+ warnings.warn(
208
+ f"BIC minimum landed on the k-boundary at k={best_k} (k_max={K}).",
209
+ BoundarySelectionWarning,
210
+ stacklevel=2,
211
+ )
212
+ return CriterionSweep(
213
+ k=np.arange(1, K + 1, dtype=np.intp),
214
+ bic=bic,
215
+ rss=rss,
216
+ best_k=best_k,
217
+ best_bic=best_bic,
218
+ best_beta=best_beta,
219
+ on_boundary=on_boundary,
220
+ )
@@ -0,0 +1,26 @@
1
+ """Simulation generators and real-data fetchers."""
2
+
3
+ from tppis.datasets.real import fetch_hydraulic, fetch_sp500_info
4
+ from tppis.datasets.simulate import (
5
+ SimulatedData,
6
+ example1_sigma,
7
+ example2_sigma,
8
+ make_example1,
9
+ make_example2,
10
+ make_example3,
11
+ make_example4,
12
+ min_eigenvalue,
13
+ )
14
+
15
+ __all__ = [
16
+ "SimulatedData",
17
+ "example1_sigma",
18
+ "example2_sigma",
19
+ "fetch_hydraulic",
20
+ "fetch_sp500_info",
21
+ "make_example1",
22
+ "make_example2",
23
+ "make_example3",
24
+ "make_example4",
25
+ "min_eigenvalue",
26
+ ]
tppis/datasets/real.py ADDED
@@ -0,0 +1,97 @@
1
+ """Fetch scripts for the two real datasets in Section 5 of the paper.
2
+
3
+ Neither dataset is redistributed. These helpers download from the public
4
+ sources the paper cites, cache locally, and raise a clear error if a source
5
+ is unavailable.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ import os
12
+ import urllib.request
13
+ from collections.abc import Callable
14
+ from pathlib import Path
15
+
16
+ DEFAULT_CACHE = Path(os.environ.get("TPPIS_DATA", Path.home() / ".tppis_data"))
17
+
18
+ # Sources as cited in Tanaka and Matsui (2023), Section 5.
19
+ HYDRAULIC_INFO = {
20
+ "name": "condition monitoring of hydraulic systems",
21
+ "paper_ref": "[17] Helwig, Pignanelli, Schutze, I2MTC 2015",
22
+ "url": "https://archive.ics.uci.edu/static/public/447/condition+monitoring+of+hydraulic+systems.zip",
23
+ "notes": (
24
+ "The paper uses n=1449 rows taken under stable system settings, "
25
+ "response = accumulator pressure (130/115/100/90), p=43680 from 17 sensors."
26
+ ),
27
+ }
28
+
29
+ SP500_INFO = {
30
+ "name": "S&P 500, year 2020",
31
+ "paper_ref": "[18] FRED SP500; [19] Kaggle S&P 500 stocks",
32
+ "fred_url": "https://fred.stlouisfed.org/series/SP500",
33
+ "kaggle_url": "https://www.kaggle.com/hanseopark/sp-500-stocks-value-with-financial-statement",
34
+ "notes": (
35
+ "The paper uses 253 trading days in 2020. Response = S&P 500 index; "
36
+ "predictors = constituent stock prices (some companies contribute more "
37
+ "than one series)."
38
+ ),
39
+ }
40
+
41
+
42
+ def _sha256(path: Path) -> str:
43
+ digest = hashlib.sha256()
44
+ with path.open("rb") as handle:
45
+ for chunk in iter(lambda: handle.read(1 << 20), b""):
46
+ digest.update(chunk)
47
+ return digest.hexdigest()
48
+
49
+
50
+ def _download(url: str, dest: Path, *, timeout: int = 60) -> Path:
51
+ dest.parent.mkdir(parents=True, exist_ok=True)
52
+ if dest.exists():
53
+ return dest
54
+ tmp = dest.with_suffix(dest.suffix + ".part")
55
+ try:
56
+ with urllib.request.urlopen(url, timeout=timeout) as src, tmp.open("wb") as out:
57
+ out.write(src.read())
58
+ tmp.replace(dest)
59
+ except Exception as exc: # noqa: BLE001
60
+ if tmp.exists():
61
+ tmp.unlink()
62
+ raise RuntimeError(
63
+ f"Could not download {url}. Fetch the file by hand and place it at {dest}."
64
+ ) from exc
65
+ return dest
66
+
67
+
68
+ def fetch_hydraulic(
69
+ cache_dir: Path | None = None,
70
+ *,
71
+ downloader: Callable[[str, Path], Path] | None = None,
72
+ ) -> Path:
73
+ """Download the UCI hydraulic-systems archive into ``cache_dir``.
74
+
75
+ Returns the path to the zip. Parsing into the paper's ``(X, y)`` layout is
76
+ left to ``reproduction/run_real_data.py``, because the 17-sensor expansion
77
+ is study-specific.
78
+ """
79
+ root = cache_dir or DEFAULT_CACHE
80
+ dest = root / "hydraulic" / "hydraulic_systems.zip"
81
+ fetch = downloader or (lambda url, path: _download(url, path))
82
+ return fetch(HYDRAULIC_INFO["url"], dest)
83
+
84
+
85
+ def fetch_sp500_info() -> dict[str, str]:
86
+ """Return the documented S&P 500 source URLs. No automatic download.
87
+
88
+ The Kaggle source requires an account, so this function only documents
89
+ the locations. ``reproduction/run_real_data.py`` reads a user-supplied
90
+ directory of CSVs.
91
+ """
92
+ return dict(SP500_INFO)
93
+
94
+
95
+ def verify_checksum(path: Path, expected: str) -> bool:
96
+ """Compare a file to a published SHA-256 hex digest."""
97
+ return _sha256(path) == expected.lower()