multivariate-probit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- multivariate_probit-0.1.0/.gitignore +18 -0
- multivariate_probit-0.1.0/LICENSE +21 -0
- multivariate_probit-0.1.0/PKG-INFO +175 -0
- multivariate_probit-0.1.0/README.md +122 -0
- multivariate_probit-0.1.0/docs/api.md +154 -0
- multivariate_probit-0.1.0/docs/ifm.md +397 -0
- multivariate_probit-0.1.0/docs/implementation.md +140 -0
- multivariate_probit-0.1.0/docs/limitations.md +80 -0
- multivariate_probit-0.1.0/pyproject.toml +43 -0
- multivariate_probit-0.1.0/src/multivariate_probit/__init__.py +56 -0
- multivariate_probit-0.1.0/src/multivariate_probit/_corr.py +51 -0
- multivariate_probit-0.1.0/src/multivariate_probit/_mvn.py +155 -0
- multivariate_probit-0.1.0/src/multivariate_probit/ifm.py +177 -0
- multivariate_probit-0.1.0/src/multivariate_probit/inner.py +187 -0
- multivariate_probit-0.1.0/src/multivariate_probit/linear.py +118 -0
- multivariate_probit-0.1.0/src/multivariate_probit/model.py +365 -0
- multivariate_probit-0.1.0/src/multivariate_probit/results.py +72 -0
- multivariate_probit-0.1.0/tests/conftest.py +8 -0
- multivariate_probit-0.1.0/tests/test_correlation_projection.py +106 -0
- multivariate_probit-0.1.0/tests/test_linear.py +166 -0
- multivariate_probit-0.1.0/tests/test_xgboost.py +130 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mark Shipman
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: multivariate-probit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Multivariate probit models fitted by Inference Functions for Margins (IFM), with pluggable inner models.
|
|
5
|
+
Project-URL: Homepage, https://github.com/sign-of-fourier/multivariate-probit
|
|
6
|
+
Project-URL: Documentation, https://github.com/sign-of-fourier/multivariate-probit/blob/main/docs/ifm.md
|
|
7
|
+
Project-URL: Issues, https://github.com/sign-of-fourier/multivariate-probit/issues
|
|
8
|
+
Author: Mark Shipman
|
|
9
|
+
License: MIT License
|
|
10
|
+
|
|
11
|
+
Copyright (c) 2026 Mark Shipman
|
|
12
|
+
|
|
13
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
14
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
15
|
+
in the Software without restriction, including without limitation the rights
|
|
16
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
17
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
18
|
+
furnished to do so, subject to the following conditions:
|
|
19
|
+
|
|
20
|
+
The above copyright notice and this permission notice shall be included in all
|
|
21
|
+
copies or substantial portions of the Software.
|
|
22
|
+
|
|
23
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
24
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
25
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
26
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
27
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
28
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
29
|
+
SOFTWARE.
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Keywords: IFM,copula,correlated binary outcomes,multivariate probit,xgboost
|
|
32
|
+
Classifier: Development Status :: 3 - Alpha
|
|
33
|
+
Classifier: Intended Audience :: Science/Research
|
|
34
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
35
|
+
Classifier: Programming Language :: Python :: 3
|
|
36
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
37
|
+
Requires-Python: >=3.9
|
|
38
|
+
Requires-Dist: numpy>=1.22
|
|
39
|
+
Requires-Dist: scipy>=1.8
|
|
40
|
+
Provides-Extra: all
|
|
41
|
+
Requires-Dist: scikit-learn>=1.1; extra == 'all'
|
|
42
|
+
Requires-Dist: xgboost>=1.7; extra == 'all'
|
|
43
|
+
Provides-Extra: dev
|
|
44
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
45
|
+
Requires-Dist: scikit-learn>=1.1; extra == 'dev'
|
|
46
|
+
Requires-Dist: xgboost>=1.7; extra == 'dev'
|
|
47
|
+
Provides-Extra: sklearn
|
|
48
|
+
Requires-Dist: scikit-learn>=1.1; extra == 'sklearn'
|
|
49
|
+
Provides-Extra: xgboost
|
|
50
|
+
Requires-Dist: scikit-learn>=1.1; extra == 'xgboost'
|
|
51
|
+
Requires-Dist: xgboost>=1.7; extra == 'xgboost'
|
|
52
|
+
Description-Content-Type: text/markdown
|
|
53
|
+
|
|
54
|
+
# multivariate-probit
|
|
55
|
+
|
|
56
|
+
Multivariate probit models for correlated binary outcomes, with a pluggable
|
|
57
|
+
inner model.
|
|
58
|
+
|
|
59
|
+
## The model
|
|
60
|
+
|
|
61
|
+
Each outcome is a threshold on a latent Gaussian variable, and the outcomes are
|
|
62
|
+
tied together by the correlation of those latents:
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
Y_j = 1[ η_j(x) + e_j > 0 ], e ~ N(0, Σ), j = 1 … d
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
`η_j` is an arbitrary real-valued function of the features — linear by default,
|
|
69
|
+
or XGBoost, a random forest, or anything else that fits `(X, y)`. Σ is a
|
|
70
|
+
correlation matrix (unit diagonal) carrying the dependence between outcomes.
|
|
71
|
+
Marginally, `P(Y_j = 1 | x) = Φ(η_j(x))`.
|
|
72
|
+
|
|
73
|
+
Fitting is cross-fit two-stage IFM: every margin is fitted independently, then
|
|
74
|
+
Σ is estimated by maximum likelihood with those margins held fixed. That
|
|
75
|
+
separation is what lets the inner model be a black box. The algorithm, and the
|
|
76
|
+
alternatives that were tested and rejected, are in
|
|
77
|
+
**[docs/ifm.md](docs/ifm.md)**.
|
|
78
|
+
|
|
79
|
+
## Install
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
pip install multivariate-probit # core: numpy + scipy
|
|
83
|
+
pip install multivariate-probit[xgboost] # adds the "xgboost" preset
|
|
84
|
+
pip install multivariate-probit[all] # every preset
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Quickstart
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from multivariate_probit import MultivariateProbit
|
|
91
|
+
|
|
92
|
+
model = MultivariateProbit(inner="linear").fit(X, Y) # Y is (n, d), 0/1
|
|
93
|
+
|
|
94
|
+
proba = model.predict_proba(X)
|
|
95
|
+
proba.marginal # P(Y_j = 1 | x), shape (n, d)
|
|
96
|
+
proba.joint([1, 0, 1]) # P(Y = pattern | x), shape (n,)
|
|
97
|
+
proba.all() # P(every outcome = 1 | x)
|
|
98
|
+
proba.any(outcomes=[0, 2]) # P(at least one of these | x)
|
|
99
|
+
|
|
100
|
+
model.correlation_ # the fitted Σ, shape (d, d)
|
|
101
|
+
model.transform(X) # latent scores η, shape (n, d)
|
|
102
|
+
model.sample(X, n_samples=100) # simulated outcome patterns
|
|
103
|
+
model.score(X, Y) # mean joint log-likelihood
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`predict_proba` returns an object that behaves like the marginal-probability
|
|
107
|
+
array (`np.asarray(proba)`, indexing, `.shape`) and additionally answers the
|
|
108
|
+
joint questions Σ was estimated for. Marginal predictions do not involve Σ at
|
|
109
|
+
all; everything joint does.
|
|
110
|
+
|
|
111
|
+
## Everything here is a squashing function over a latent index
|
|
112
|
+
|
|
113
|
+
The inner model never sees a probability, and never sees another outcome's
|
|
114
|
+
labels. It produces an unbounded score η_j(x) on (-∞, ∞); Φ is the only
|
|
115
|
+
squashing function applied to it. Any estimator that emits a real-valued score,
|
|
116
|
+
or a probability that can be pushed back through Φ⁻¹, is a legal margin.
|
|
117
|
+
|
|
118
|
+
That is the whole abstraction, and it is why the inner model is swappable
|
|
119
|
+
without touching the estimation code.
|
|
120
|
+
|
|
121
|
+
## Inner models
|
|
122
|
+
|
|
123
|
+
| Preset | Estimator | Requires |
|
|
124
|
+
| --- | --- | --- |
|
|
125
|
+
| `"linear"` (default), `"probit"` | native probit via IRLS | — |
|
|
126
|
+
| `"xgboost"`, `"xgb"` | `XGBClassifier`, tuned for calibration | `xgboost` |
|
|
127
|
+
| `"rf"`, `"random_forest"` | `RandomForestClassifier` | `scikit-learn` |
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
MultivariateProbit(inner="xgboost") # a preset
|
|
131
|
+
MultivariateProbit(inner=XGBClassifier(max_depth=4)) # any sklearn-shaped model
|
|
132
|
+
MultivariateProbit(inner=["linear", "xgboost", "linear"]) # one per outcome
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
`available_inners()` lists the presets; `register_inner(name, factory)` adds
|
|
136
|
+
your own. See **[docs/api.md](docs/api.md)** for the inner-model contract.
|
|
137
|
+
|
|
138
|
+
## Two knobs that cost time
|
|
139
|
+
|
|
140
|
+
- **`dependence`** — `"joint"` (default) maximises the full d-variate
|
|
141
|
+
likelihood for Σ. `"pairwise"` maximises each pair's bivariate likelihood
|
|
142
|
+
instead: consistent, orders of magnitude cheaper, and the right choice once
|
|
143
|
+
you have more than a handful of outcomes.
|
|
144
|
+
- **`cv`** — `5` by default, cross-fitting the margins so Σ is never estimated
|
|
145
|
+
from in-sample predictions. This is not optional hygiene: in-sample margins
|
|
146
|
+
drive every fitted correlation to +1. `cv=None` skips it, which is defensible
|
|
147
|
+
for the linear default and reckless for anything that can overfit.
|
|
148
|
+
|
|
149
|
+
## Status
|
|
150
|
+
|
|
151
|
+
Alpha. The linear and XGBoost paths are covered by tests; the `rf` preset is
|
|
152
|
+
wired but untested. Standard errors are not computed — `correlation_` is a
|
|
153
|
+
point estimate. Known gaps are listed in
|
|
154
|
+
**[docs/limitations.md](docs/limitations.md)**.
|
|
155
|
+
|
|
156
|
+
## Documentation
|
|
157
|
+
|
|
158
|
+
- **[docs/ifm.md](docs/ifm.md)** — the estimation algorithm, and why IFM over
|
|
159
|
+
the alternatives
|
|
160
|
+
- **[docs/implementation.md](docs/implementation.md)** — what is hand-rolled,
|
|
161
|
+
what comes from SciPy, and why
|
|
162
|
+
- **[docs/api.md](docs/api.md)** — parameters, attributes, methods, extension
|
|
163
|
+
points
|
|
164
|
+
- **[docs/limitations.md](docs/limitations.md)** — known gaps and roadmap
|
|
165
|
+
|
|
166
|
+
## Development
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
pip install -e ".[dev]"
|
|
170
|
+
pytest
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
## License
|
|
174
|
+
|
|
175
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
# multivariate-probit
|
|
2
|
+
|
|
3
|
+
Multivariate probit models for correlated binary outcomes, with a pluggable
|
|
4
|
+
inner model.
|
|
5
|
+
|
|
6
|
+
## The model
|
|
7
|
+
|
|
8
|
+
Each outcome is a threshold on a latent Gaussian variable, and the outcomes are
|
|
9
|
+
tied together by the correlation of those latents:
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
Y_j = 1[ η_j(x) + e_j > 0 ], e ~ N(0, Σ), j = 1 … d
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
`η_j` is an arbitrary real-valued function of the features — linear by default,
|
|
16
|
+
or XGBoost, a random forest, or anything else that fits `(X, y)`. Σ is a
|
|
17
|
+
correlation matrix (unit diagonal) carrying the dependence between outcomes.
|
|
18
|
+
Marginally, `P(Y_j = 1 | x) = Φ(η_j(x))`.
|
|
19
|
+
|
|
20
|
+
Fitting is cross-fit two-stage IFM: every margin is fitted independently, then
|
|
21
|
+
Σ is estimated by maximum likelihood with those margins held fixed. That
|
|
22
|
+
separation is what lets the inner model be a black box. The algorithm, and the
|
|
23
|
+
alternatives that were tested and rejected, are in
|
|
24
|
+
**[docs/ifm.md](docs/ifm.md)**.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install multivariate-probit # core: numpy + scipy
|
|
30
|
+
pip install multivariate-probit[xgboost] # adds the "xgboost" preset
|
|
31
|
+
pip install multivariate-probit[all] # every preset
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Quickstart
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from multivariate_probit import MultivariateProbit
|
|
38
|
+
|
|
39
|
+
model = MultivariateProbit(inner="linear").fit(X, Y) # Y is (n, d), 0/1
|
|
40
|
+
|
|
41
|
+
proba = model.predict_proba(X)
|
|
42
|
+
proba.marginal # P(Y_j = 1 | x), shape (n, d)
|
|
43
|
+
proba.joint([1, 0, 1]) # P(Y = pattern | x), shape (n,)
|
|
44
|
+
proba.all() # P(every outcome = 1 | x)
|
|
45
|
+
proba.any(outcomes=[0, 2]) # P(at least one of these | x)
|
|
46
|
+
|
|
47
|
+
model.correlation_ # the fitted Σ, shape (d, d)
|
|
48
|
+
model.transform(X) # latent scores η, shape (n, d)
|
|
49
|
+
model.sample(X, n_samples=100) # simulated outcome patterns
|
|
50
|
+
model.score(X, Y) # mean joint log-likelihood
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
`predict_proba` returns an object that behaves like the marginal-probability
|
|
54
|
+
array (`np.asarray(proba)`, indexing, `.shape`) and additionally answers the
|
|
55
|
+
joint questions Σ was estimated for. Marginal predictions do not involve Σ at
|
|
56
|
+
all; everything joint does.
|
|
57
|
+
|
|
58
|
+
## Everything here is a squashing function over a latent index
|
|
59
|
+
|
|
60
|
+
The inner model never sees a probability, and never sees another outcome's
|
|
61
|
+
labels. It produces an unbounded score η_j(x) on (-∞, ∞); Φ is the only
|
|
62
|
+
squashing function applied to it. Any estimator that emits a real-valued score,
|
|
63
|
+
or a probability that can be pushed back through Φ⁻¹, is a legal margin.
|
|
64
|
+
|
|
65
|
+
That is the whole abstraction, and it is why the inner model is swappable
|
|
66
|
+
without touching the estimation code.
|
|
67
|
+
|
|
68
|
+
## Inner models
|
|
69
|
+
|
|
70
|
+
| Preset | Estimator | Requires |
|
|
71
|
+
| --- | --- | --- |
|
|
72
|
+
| `"linear"` (default), `"probit"` | native probit via IRLS | — |
|
|
73
|
+
| `"xgboost"`, `"xgb"` | `XGBClassifier`, tuned for calibration | `xgboost` |
|
|
74
|
+
| `"rf"`, `"random_forest"` | `RandomForestClassifier` | `scikit-learn` |
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
MultivariateProbit(inner="xgboost") # a preset
|
|
78
|
+
MultivariateProbit(inner=XGBClassifier(max_depth=4)) # any sklearn-shaped model
|
|
79
|
+
MultivariateProbit(inner=["linear", "xgboost", "linear"]) # one per outcome
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
`available_inners()` lists the presets; `register_inner(name, factory)` adds
|
|
83
|
+
your own. See **[docs/api.md](docs/api.md)** for the inner-model contract.
|
|
84
|
+
|
|
85
|
+
## Two knobs that cost time
|
|
86
|
+
|
|
87
|
+
- **`dependence`** — `"joint"` (default) maximises the full d-variate
|
|
88
|
+
likelihood for Σ. `"pairwise"` maximises each pair's bivariate likelihood
|
|
89
|
+
instead: consistent, orders of magnitude cheaper, and the right choice once
|
|
90
|
+
you have more than a handful of outcomes.
|
|
91
|
+
- **`cv`** — `5` by default, cross-fitting the margins so Σ is never estimated
|
|
92
|
+
from in-sample predictions. This is not optional hygiene: in-sample margins
|
|
93
|
+
drive every fitted correlation to +1. `cv=None` skips it, which is defensible
|
|
94
|
+
for the linear default and reckless for anything that can overfit.
|
|
95
|
+
|
|
96
|
+
## Status
|
|
97
|
+
|
|
98
|
+
Alpha. The linear and XGBoost paths are covered by tests; the `rf` preset is
|
|
99
|
+
wired but untested. Standard errors are not computed — `correlation_` is a
|
|
100
|
+
point estimate. Known gaps are listed in
|
|
101
|
+
**[docs/limitations.md](docs/limitations.md)**.
|
|
102
|
+
|
|
103
|
+
## Documentation
|
|
104
|
+
|
|
105
|
+
- **[docs/ifm.md](docs/ifm.md)** — the estimation algorithm, and why IFM over
|
|
106
|
+
the alternatives
|
|
107
|
+
- **[docs/implementation.md](docs/implementation.md)** — what is hand-rolled,
|
|
108
|
+
what comes from SciPy, and why
|
|
109
|
+
- **[docs/api.md](docs/api.md)** — parameters, attributes, methods, extension
|
|
110
|
+
points
|
|
111
|
+
- **[docs/limitations.md](docs/limitations.md)** — known gaps and roadmap
|
|
112
|
+
|
|
113
|
+
## Development
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
pip install -e ".[dev]"
|
|
117
|
+
pytest
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
## License
|
|
121
|
+
|
|
122
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
# API reference
|
|
2
|
+
|
|
3
|
+
For the model and a quickstart see [../README.md](../README.md); for the
|
|
4
|
+
estimation algorithm see [ifm.md](ifm.md).
|
|
5
|
+
|
|
6
|
+
## MultivariateProbit
|
|
7
|
+
|
|
8
|
+
```python
|
|
9
|
+
MultivariateProbit(
|
|
10
|
+
inner="linear",
|
|
11
|
+
inner_params=None,
|
|
12
|
+
dependence="joint",
|
|
13
|
+
cv=5,
|
|
14
|
+
n_quad=24,
|
|
15
|
+
optimizer="Nelder-Mead",
|
|
16
|
+
project_correlation=True,
|
|
17
|
+
random_state=None,
|
|
18
|
+
)
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
### Parameters
|
|
22
|
+
|
|
23
|
+
| Name | Default | Meaning |
|
|
24
|
+
| --- | --- | --- |
|
|
25
|
+
| `inner` | `"linear"` | Preset name, estimator instance, factory, or a list of length d giving one per outcome. Instances are deep-copied, so one instance can seed every margin. |
|
|
26
|
+
| `inner_params` | `None` | Keyword arguments forwarded to the preset factory. Ignored when `inner` is already an instance. |
|
|
27
|
+
| `dependence` | `"joint"` | `"joint"` maximises the full d-variate likelihood for Σ; `"pairwise"` maximises each pair's bivariate likelihood (composite likelihood, far cheaper). |
|
|
28
|
+
| `cv` | `5` | Folds used to cross-fit the latent indices that stage two consumes. `None` skips cross-fitting — see the warning below. |
|
|
29
|
+
| `n_quad` | `24` | Gauss-Legendre order for the orthant evaluator. Lower it if fitting with many outcomes gets slow. |
|
|
30
|
+
| `optimizer` | `"Nelder-Mead"` | Passed to `scipy.optimize.minimize` for `dependence="joint"`. Derivative-free by design. |
|
|
31
|
+
| `project_correlation` | `True` | Project a pairwise estimate onto the nearest positive-definite correlation matrix. No effect when `dependence="joint"`. |
|
|
32
|
+
| `random_state` | `None` | Controls the cross-fitting split and `sample`. |
|
|
33
|
+
|
|
34
|
+
> **`cv=None` is not a neutral speed-up.** In-sample margins drive every fitted
|
|
35
|
+
> correlation toward +1. It is defensible for the linear default and reckless
|
|
36
|
+
> for anything that can overfit. See
|
|
37
|
+
> [ifm.md](ifm.md#why-cross-fitting-is-required).
|
|
38
|
+
|
|
39
|
+
### Attributes
|
|
40
|
+
|
|
41
|
+
| Name | Shape | Meaning |
|
|
42
|
+
| --- | --- | --- |
|
|
43
|
+
| `inner_models_` | list, length d | The fitted margins, one per outcome. |
|
|
44
|
+
| `correlation_` | (d, d) | The fitted Σ. |
|
|
45
|
+
| `eta_` | (n, d) | The (cross-fitted) latent indices stage two was fitted on. |
|
|
46
|
+
| `nll_` | float | Negative log-likelihood at the end of the dependence fit. Joint and pairwise fits optimise different objectives, so the values are not comparable across settings. |
|
|
47
|
+
| `optimize_result_` | OptimizeResult or None | The SciPy result for `dependence="joint"`. |
|
|
48
|
+
| `n_outcomes_`, `n_features_in_` | int | |
|
|
49
|
+
|
|
50
|
+
### Methods
|
|
51
|
+
|
|
52
|
+
| Method | Returns | Notes |
|
|
53
|
+
| --- | --- | --- |
|
|
54
|
+
| `fit(X, Y, sample_weight=None)` | self | `Y` is (n, d) and strictly 0/1. Weights are forwarded to margins that accept them. |
|
|
55
|
+
| `decision_function(X)` / `transform(X)` | (n, d) | Latent indices η on (-∞, ∞). |
|
|
56
|
+
| `fit_transform(X, Y)` | (n, d) | |
|
|
57
|
+
| `predict_proba(X)` | `MultivariateProbitProba` | See below. |
|
|
58
|
+
| `predict_marginal_proba(X)` | (n, d) | Marginal probabilities as a plain array. |
|
|
59
|
+
| `predict(X, threshold=0.5)` | (n, d) | Per-outcome 0/1 at a marginal threshold. For a joint decision use `predict_proba(X).all()`. |
|
|
60
|
+
| `joint_proba(X, Y)` | (n,) | `P(Y = y | x)`. `Y` may be one pattern of length d, broadcast over rows, or an (n, d) array. |
|
|
61
|
+
| `joint_log_proba(X, Y)` | (n,) | |
|
|
62
|
+
| `score(X, Y, sample_weight=None)` | float | Mean joint log-likelihood. Higher is better. |
|
|
63
|
+
| `sample(X, n_samples=1, random_state=None)` | (n, d) or (n_samples, n, d) | Draws patterns from the fitted model. |
|
|
64
|
+
| `get_params` / `set_params` | | scikit-learn-style, without requiring scikit-learn. |
|
|
65
|
+
|
|
66
|
+
## MultivariateProbitProba
|
|
67
|
+
|
|
68
|
+
Returned by `predict_proba`. Behaves like the (n, d) array of marginal
|
|
69
|
+
probabilities — `np.asarray(proba)`, `proba[i, j]`, `proba.shape`, `len(proba)`
|
|
70
|
+
— and additionally answers the joint questions Σ was estimated for.
|
|
71
|
+
|
|
72
|
+
| Member | Returns | Meaning |
|
|
73
|
+
| --- | --- | --- |
|
|
74
|
+
| `.marginal` | (n, d) | `P(Y_j = 1 | x)`. Does not involve Σ. |
|
|
75
|
+
| `.eta` | (n, d) | The latent indices behind it. |
|
|
76
|
+
| `.corr` | (d, d) | The Σ used for joint queries. |
|
|
77
|
+
| `.joint(pattern)` | (n,) | `P(Y = pattern | x)` for a fully specified 0/1 pattern. |
|
|
78
|
+
| `.all(outcomes=None)` | (n,) | `P(every selected outcome = 1 | x)`. |
|
|
79
|
+
| `.any(outcomes=None)` | (n,) | `P(at least one selected outcome = 1 | x)`. |
|
|
80
|
+
| `.none(outcomes=None)` | (n,) | `P(no selected outcome = 1 | x)`. |
|
|
81
|
+
|
|
82
|
+
`outcomes` takes a list of column indices, so "do these three co-occur" is a
|
|
83
|
+
one-liner.
|
|
84
|
+
|
|
85
|
+
## Inner models
|
|
86
|
+
|
|
87
|
+
### The contract
|
|
88
|
+
|
|
89
|
+
An inner model is anything with `fit(X, y)` and `latent(X) -> (n,)`, where
|
|
90
|
+
`latent` returns a real-valued index on the probit scale.
|
|
91
|
+
|
|
92
|
+
Estimators that do not expose `latent` are wrapped automatically by
|
|
93
|
+
`ProbitCalibrated`, which supplies it:
|
|
94
|
+
|
|
95
|
+
- if the estimator has `predict_proba`, `latent(X)` is `Φ⁻¹(p̂)`, clipped away
|
|
96
|
+
from 0 and 1;
|
|
97
|
+
- otherwise, if it has `decision_function`, that score is used as the index
|
|
98
|
+
as-is — which assumes it is already probit-scaled. Check that assumption
|
|
99
|
+
before relying on it.
|
|
100
|
+
|
|
101
|
+
### Presets
|
|
102
|
+
|
|
103
|
+
| Name | Aliases | Estimator | Requires |
|
|
104
|
+
| --- | --- | --- | --- |
|
|
105
|
+
| `linear` | `probit` | `ProbitRegressor` (native IRLS) | — |
|
|
106
|
+
| `xgboost` | `xgb` | `XGBClassifier`, defaults tuned for calibration rather than ranking | `xgboost` |
|
|
107
|
+
| `rf` | `random_forest` | `RandomForestClassifier` | `scikit-learn` |
|
|
108
|
+
|
|
109
|
+
The registry is deliberately thin. Under IFM there is no per-family estimation
|
|
110
|
+
logic — the presets differ only in which pre-wired instance stage one fits — so
|
|
111
|
+
a preset is a factory function and nothing more.
|
|
112
|
+
|
|
113
|
+
### Registry functions
|
|
114
|
+
|
|
115
|
+
| Function | Purpose |
|
|
116
|
+
| --- | --- |
|
|
117
|
+
| `available_inners()` | Sorted preset names. |
|
|
118
|
+
| `make_inner(name, **kwargs)` | Instantiate a preset, forwarding kwargs to the estimator. |
|
|
119
|
+
| `register_inner(name, factory, overwrite=False)` | Add a preset. `factory(**kwargs)` returns a fresh estimator. |
|
|
120
|
+
| `as_inner(spec, **kwargs)` | Coerce a name, factory, class, or instance into an unfitted inner model. |
|
|
121
|
+
|
|
122
|
+
### Mixing models per outcome
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
MultivariateProbit(inner=["linear", "xgboost", "linear"])
|
|
126
|
+
MultivariateProbit(inner=["linear", make_inner("xgboost", max_depth=2)])
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## ProbitRegressor
|
|
130
|
+
|
|
131
|
+
The default margin, usable on its own as a binary classifier.
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
ProbitRegressor(alpha=1e-6, fit_intercept=True, max_iter=100, tol=1e-8)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
`alpha` is an L2 penalty on the slopes only, never the intercept. Exposes
|
|
138
|
+
`coef_`, `intercept_`, `n_iter_`, `classes_`, and `fit` / `latent` /
|
|
139
|
+
`decision_function` / `predict_proba` / `predict`.
|
|
140
|
+
|
|
141
|
+
## Lower-level functions
|
|
142
|
+
|
|
143
|
+
Exported for callers who want the pieces without the estimator.
|
|
144
|
+
|
|
145
|
+
| Function | Purpose |
|
|
146
|
+
| --- | --- |
|
|
147
|
+
| `bvn_cdf(a, b, rho)` | Φ_2, vectorised, correlations may vary by row. |
|
|
148
|
+
| `mvn_orthant(A, corr_stack)` | Φ_d with a per-row correlation stack. |
|
|
149
|
+
| `orthant_prob(upper, corr)` | Φ_d with one shared correlation matrix. |
|
|
150
|
+
| `pattern_prob(eta, Y, corr)` | `P(Y = y | x)` via the sign trick. |
|
|
151
|
+
| `pairwise_correlation(eta, Y, ...)` | Stage two, composite likelihood. |
|
|
152
|
+
| `joint_correlation(eta, Y, ...)` | Stage two, full likelihood. Returns `(corr, result)`. |
|
|
153
|
+
| `pair_log_likelihood(rho, ...)` | Bivariate probit log-likelihood in ρ. |
|
|
154
|
+
| `joint_log_likelihood(corr, eta, Y, ...)` | d-variate log-likelihood in Σ. |
|