scistest 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scistest-0.1.0/.github/workflows/publish.yml +40 -0
- scistest-0.1.0/.github/workflows/tests.yml +19 -0
- scistest-0.1.0/.gitignore +10 -0
- scistest-0.1.0/LICENSE +21 -0
- scistest-0.1.0/PKG-INFO +288 -0
- scistest-0.1.0/README.md +259 -0
- scistest-0.1.0/examples/demo.ipynb +354 -0
- scistest-0.1.0/examples/demo.py +74 -0
- scistest-0.1.0/pyproject.toml +58 -0
- scistest-0.1.0/scripts/run_simulation.py +155 -0
- scistest-0.1.0/src/scistest/__init__.py +37 -0
- scistest-0.1.0/src/scistest/_models.py +146 -0
- scistest-0.1.0/src/scistest/aggregation.py +361 -0
- scistest-0.1.0/src/scistest/config.py +73 -0
- scistest-0.1.0/src/scistest/core.py +682 -0
- scistest-0.1.0/src/scistest/py.typed +1 -0
- scistest-0.1.0/src/scistest/simulation.py +72 -0
- scistest-0.1.0/tests/test_aggregation.py +88 -0
- scistest-0.1.0/tests/test_core_helpers.py +98 -0
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
name: publish
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
permissions:
|
|
8
|
+
contents: read
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
build:
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v6
|
|
15
|
+
with:
|
|
16
|
+
persist-credentials: false
|
|
17
|
+
- uses: actions/setup-python@v6
|
|
18
|
+
with:
|
|
19
|
+
python-version: "3.x"
|
|
20
|
+
- run: python -m pip install --upgrade build
|
|
21
|
+
- run: python -m build
|
|
22
|
+
- uses: actions/upload-artifact@v5
|
|
23
|
+
with:
|
|
24
|
+
name: python-package-distributions
|
|
25
|
+
path: dist/
|
|
26
|
+
|
|
27
|
+
publish-to-pypi:
|
|
28
|
+
needs: build
|
|
29
|
+
runs-on: ubuntu-latest
|
|
30
|
+
environment:
|
|
31
|
+
name: pypi
|
|
32
|
+
url: https://pypi.org/p/scistest
|
|
33
|
+
permissions:
|
|
34
|
+
id-token: write
|
|
35
|
+
steps:
|
|
36
|
+
- uses: actions/download-artifact@v6
|
|
37
|
+
with:
|
|
38
|
+
name: python-package-distributions
|
|
39
|
+
path: dist/
|
|
40
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- uses: actions/checkout@v6
|
|
12
|
+
- uses: actions/setup-python@v6
|
|
13
|
+
with:
|
|
14
|
+
python-version: "3.11"
|
|
15
|
+
cache: pip
|
|
16
|
+
- run: python -m pip install --upgrade pip
|
|
17
|
+
- run: python -m pip install -e '.[dev]'
|
|
18
|
+
- run: pytest
|
|
19
|
+
- run: ruff check .
|
scistest-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Qixian Zhong and Rajen Shah
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
scistest-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: scistest
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Conditional independence testing for right-censored survival outcomes
|
|
5
|
+
Project-URL: Homepage, https://github.com/qxzhong/scistest
|
|
6
|
+
Project-URL: Repository, https://github.com/qxzhong/scistest
|
|
7
|
+
Project-URL: Issues, https://github.com/qxzhong/scistest/issues
|
|
8
|
+
Author-email: Qixian Zhong <qxzhong@xmu.edu.cn>, Rajen Shah <rds37@cam.ac.uk>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: conditional independence,hypothesis testing,survival analysis
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Requires-Dist: engression>=0.1.15
|
|
21
|
+
Requires-Dist: numpy>=1.24
|
|
22
|
+
Requires-Dist: scikit-learn>=1.3
|
|
23
|
+
Requires-Dist: scipy>=1.10
|
|
24
|
+
Requires-Dist: torch>=2.0
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
27
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# Conditional Independence Testing with Survival Data
|
|
31
|
+
|
|
32
|
+
`scistest` implements the conditional-independence test developed for
|
|
33
|
+
right-censored outcomes. It is based on the paper ``Conditional Independence Testing with Survival Data" by Qixian Zhong (Xiamen University) and Rajen Shah (University of Cambridge).
|
|
34
|
+
|
|
35
|
+
Given observations
|
|
36
|
+
|
|
37
|
+
$$
|
|
38
|
+
(X_i,Z_i,T_i,\Delta_i),\qquad
|
|
39
|
+
T_i=\min(U_i,C_i),\quad \Delta_i=\mathbf 1(U_i\le C_i),
|
|
40
|
+
$$
|
|
41
|
+
|
|
42
|
+
where $X$ and $Z$ are covariates, and $U$ and $C$ are event and censoring time, respectively. The package tests
|
|
43
|
+
|
|
44
|
+
$$
|
|
45
|
+
H_0: U\perp X\mid Z.
|
|
46
|
+
$$
|
|
47
|
+
|
|
48
|
+
The repository contains the single-split `scis_test` implementation, the
|
|
49
|
+
Guo--Shah (2025) rank-transformed subsampling aggregation, a small demo, a
|
|
50
|
+
simulation runner, and unit tests. The code is currently a research release
|
|
51
|
+
(`0.1.0`); freeze the version and package settings used for any reported
|
|
52
|
+
numerical experiment.
|
|
53
|
+
|
|
54
|
+
The package is maintained by Qixian Zhong and Rajen Shah and distributed
|
|
55
|
+
under the MIT License.
|
|
56
|
+
|
|
57
|
+
## Repository layout
|
|
58
|
+
|
|
59
|
+
```text
|
|
60
|
+
.
|
|
61
|
+
├── pyproject.toml
|
|
62
|
+
├── src/scistest/
|
|
63
|
+
│ ├── core.py # single-split scisTest and its p-value
|
|
64
|
+
│ ├── aggregation.py # Guo--Shah aggregated p-value
|
|
65
|
+
│ ├── config.py # typed hyperparameter configurations
|
|
66
|
+
│ └── simulation.py # reproducible example data generator
|
|
67
|
+
├── examples/
|
|
68
|
+
│ ├── demo.py
|
|
69
|
+
│ └── demo.ipynb
|
|
70
|
+
├── scripts/run_simulation.py
|
|
71
|
+
└── tests/
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Installation
|
|
75
|
+
|
|
76
|
+
From this directory, create an isolated environment and install the package:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
python -m venv .venv
|
|
80
|
+
source .venv/bin/activate
|
|
81
|
+
python -m pip install --upgrade pip
|
|
82
|
+
python -m pip install -e .
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
For development tools, use `python -m pip install -e '.[dev]'`.
|
|
86
|
+
|
|
87
|
+
## Single-split test
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from scistest import scis_test
|
|
91
|
+
|
|
92
|
+
result = scis_test(
|
|
93
|
+
x=X, # shape (n,) or (n, d_x)
|
|
94
|
+
z=Z, # shape (n,) or (n, d_z)
|
|
95
|
+
time=T, # observed min(event time, censoring time), shape (n,)
|
|
96
|
+
event=Delta, # 1=event observed, 0=right censored, shape (n,)
|
|
97
|
+
random_state=1,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
print(result.test_statistic)
|
|
101
|
+
print(result.p_value)
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
`scisTest` is provided as an alias for `scis_test`. The returned
|
|
105
|
+
`ScisTestResult` also includes `log_p_value`, the held-out score moments, sample
|
|
106
|
+
indices, and the individual orthogonalized scores. Use
|
|
107
|
+
`result.as_dict(include_arrays=False)` for JSON-friendly scalar output.
|
|
108
|
+
|
|
109
|
+
For each split, the package:
|
|
110
|
+
|
|
111
|
+
1. fits the null hazard $\lambda(t\mid Z)=\exp\{g(t,Z)\}$ on the hunting sample
|
|
112
|
+
using the penalized negative counting-process log-likelihood;
|
|
113
|
+
2. learns $\phi$, by default with a DNN, by minimizing exactly
|
|
114
|
+
$\sum_i V_i(\phi)^2-\sum_i V_i(\phi)$;
|
|
115
|
+
3. fits engression for the joint conditional distribution
|
|
116
|
+
$P_{(X,T)\mid Z}$ on the test sample;
|
|
117
|
+
4. constructs the orthogonalized held-out scores and reports the one-sided
|
|
118
|
+
standard-normal tail probability.
|
|
119
|
+
|
|
120
|
+
Preprocessing constants are estimated only on the hunting sample: `X` is
|
|
121
|
+
min--max scaled, `Z` is standardized, and observed time is divided by the
|
|
122
|
+
maximum hunting-sample time.
|
|
123
|
+
|
|
124
|
+
Hyperparameters are explicit immutable dataclasses:
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
from scistest import (
|
|
128
|
+
DirectionConfig,
|
|
129
|
+
GeneratorConfig,
|
|
130
|
+
HazardConfig,
|
|
131
|
+
SciTestConfig,
|
|
132
|
+
scis_test,
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
result = scis_test(
|
|
136
|
+
X, Z, T, Delta,
|
|
137
|
+
random_state=10,
|
|
138
|
+
hazard_config=HazardConfig(hidden_dims=(32,), epochs=100),
|
|
139
|
+
direction_config=DirectionConfig(hidden_dims=(32, 32), epochs=100),
|
|
140
|
+
generator_config=GeneratorConfig(num_layers=2, hidden_dim=100, epochs=100),
|
|
141
|
+
test_config=SciTestConfig(integration_points=100, generator_draws=100),
|
|
142
|
+
)
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
### Random-forest direction estimator
|
|
146
|
+
|
|
147
|
+
The DNN remains the default. To estimate \(\phi\) with a random forest instead,
|
|
148
|
+
pass `RandomForestDirectionConfig` as `direction_config`:
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
from scistest import RandomForestDirectionConfig, scis_test
|
|
152
|
+
|
|
153
|
+
result = scis_test(
|
|
154
|
+
X, Z, T, Delta,
|
|
155
|
+
random_state=10,
|
|
156
|
+
direction_config=RandomForestDirectionConfig(
|
|
157
|
+
n_estimators=100,
|
|
158
|
+
max_depth=6,
|
|
159
|
+
min_samples_leaf=5,
|
|
160
|
+
n_jobs=-1,
|
|
161
|
+
),
|
|
162
|
+
)
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
The forest partitions the `(time, Z, X)` feature space. Conditional on those
|
|
166
|
+
partitions, its terminal-node values are obtained by sparse least squares using
|
|
167
|
+
|
|
168
|
+
$$
|
|
169
|
+
\sum_i[V_i(\phi)^2-V_i(\phi)].
|
|
170
|
+
$$
|
|
171
|
+
|
|
172
|
+
Thus both direction estimators target the same criterion; only the function
|
|
173
|
+
class and optimization method differ. The hazard and conditional-generator
|
|
174
|
+
estimators are unchanged.
|
|
175
|
+
|
|
176
|
+
## Aggregated p-value with inputs `(K, J, L)`
|
|
177
|
+
|
|
178
|
+
The convenience interface refits the full scisTest procedure for all full-data
|
|
179
|
+
splits and calibration subsamples:
|
|
180
|
+
|
|
181
|
+
```python
|
|
182
|
+
from scistest import aggregated_scis_test
|
|
183
|
+
|
|
184
|
+
aggregated = aggregated_scis_test(
|
|
185
|
+
X, Z, T, Delta,
|
|
186
|
+
K=3, # number of disjoint subsamples per permutation
|
|
187
|
+
J=5, # independent permutations
|
|
188
|
+
L=6, # random splits per full sample or subsample
|
|
189
|
+
random_state=20260918,
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
print(aggregated.aggregated_p_value) # smoothed right-tail p-value
|
|
193
|
+
print(aggregated.empirical_p_value) # unsmoothed calibration fraction
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
The defaults are `K=3`, `J=5`, and `L=6`, so these three arguments may be
|
|
197
|
+
omitted.
|
|
198
|
+
|
|
199
|
+
Each permutation divides the data into `K` disjoint blocks of common size
|
|
200
|
+
|
|
201
|
+
$$
|
|
202
|
+
\texttt{subsample size}=\lfloor n/K\rfloor.
|
|
203
|
+
$$
|
|
204
|
+
|
|
205
|
+
The remaining `n % K` observations are unused in that permutation. Thus the
|
|
206
|
+
number of calibration rows is
|
|
207
|
+
|
|
208
|
+
$$
|
|
209
|
+
B=JK.
|
|
210
|
+
$$
|
|
211
|
+
|
|
212
|
+
All \(BL\) subsample statistics are pooled for the randomized rank transform
|
|
213
|
+
|
|
214
|
+
$$
|
|
215
|
+
\widetilde H_{b\ell}=\Phi^{-1}[\frac{R_{b\ell}-1/2}{BL}].
|
|
216
|
+
$$
|
|
217
|
+
|
|
218
|
+
The full-sample and row-wise calibration aggregates are arithmetic means over
|
|
219
|
+
the \(L\) split statistics. The primary result is the Gaussian-kernel-smoothed
|
|
220
|
+
right tail used by the authors' MultiSplit implementation. The result also
|
|
221
|
+
retains the strict empirical tail, whose resolution is `1 / B`. Because a
|
|
222
|
+
publication-scale run fits `(B + 1) * L` complete tests, it can be expensive.
|
|
223
|
+
|
|
224
|
+
For another asymptotically standard-normal split statistic, use the generic
|
|
225
|
+
callback API:
|
|
226
|
+
|
|
227
|
+
```python
|
|
228
|
+
from scistest import aggregate_p_value
|
|
229
|
+
|
|
230
|
+
def statistic(indices, split_seed):
|
|
231
|
+
# Subset the data, refit the complete base method using split_seed,
|
|
232
|
+
# and return one one-sided standard-normal statistic.
|
|
233
|
+
return my_statistic(indices, split_seed)
|
|
234
|
+
|
|
235
|
+
aggregated = aggregate_p_value(
|
|
236
|
+
n=len(T),
|
|
237
|
+
statistic=statistic,
|
|
238
|
+
K=3,
|
|
239
|
+
J=5,
|
|
240
|
+
L=6,
|
|
241
|
+
random_state=20260918,
|
|
242
|
+
)
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
If the full and subsample statistics have already been computed, call
|
|
246
|
+
`calibrate_aggregated_p_value(observed, subsampled, K, J, L, n=n)`. This is
|
|
247
|
+
useful for checkpointed cluster runs.
|
|
248
|
+
|
|
249
|
+
## Demo and simulation
|
|
250
|
+
|
|
251
|
+
Run a quick single-split example:
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
python examples/demo.py
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
For an interactive walkthrough of the DNN direction, random-forest direction,
|
|
258
|
+
and `K`-based aggregation interfaces, open `examples/demo.ipynb`.
|
|
259
|
+
|
|
260
|
+
Add `--aggregate` to demonstrate the complete aggregation workflow with small
|
|
261
|
+
settings. Run a reproducible null/alternative experiment with:
|
|
262
|
+
|
|
263
|
+
```bash
|
|
264
|
+
python scripts/run_simulation.py \
|
|
265
|
+
--repetitions 20 \
|
|
266
|
+
--sample-size 400 \
|
|
267
|
+
--scenario both \
|
|
268
|
+
--output results/simulation.csv
|
|
269
|
+
```
|
|
270
|
+
|
|
271
|
+
Aggregation can be enabled with `--aggregate`; the command-line defaults are
|
|
272
|
+
also `--K 3 --J 5 --L 6`. It is much more computationally demanding because
|
|
273
|
+
every calibration statistic refits all nuisance models.
|
|
274
|
+
|
|
275
|
+
## Tests
|
|
276
|
+
|
|
277
|
+
```bash
|
|
278
|
+
pytest
|
|
279
|
+
ruff check .
|
|
280
|
+
```
|
|
281
|
+
|
|
282
|
+
## Reference
|
|
283
|
+
Qixian Zhong and Rajen D. Shah (2026), “Conditional Independence Testing with Survival Data”.
|
|
284
|
+
|
|
285
|
+
Guo, F. Richard and Rajen D. Shah (2025),
|
|
286
|
+
“Rank-transformed subsampling: inference for multiple data splitting and
|
|
287
|
+
exchangeable p-values,” *Journal of the Royal Statistical Society Series B*,
|
|
288
|
+
87(1), 256–286.
|
scistest-0.1.0/README.md
ADDED
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
# Conditional Independence Testing with Survival Data
|
|
2
|
+
|
|
3
|
+
`scistest` implements the conditional-independence test developed for
|
|
4
|
+
right-censored outcomes. It is based on the paper ``Conditional Independence Testing with Survival Data" by Qixian Zhong (Xiamen University) and Rajen Shah (University of Cambridge).
|
|
5
|
+
|
|
6
|
+
Given observations
|
|
7
|
+
|
|
8
|
+
$$
|
|
9
|
+
(X_i,Z_i,T_i,\Delta_i),\qquad
|
|
10
|
+
T_i=\min(U_i,C_i),\quad \Delta_i=\mathbf 1(U_i\le C_i),
|
|
11
|
+
$$
|
|
12
|
+
|
|
13
|
+
where $X$ and $Z$ are covariates, and $U$ and $C$ are event and censoring time, respectively. The package tests
|
|
14
|
+
|
|
15
|
+
$$
|
|
16
|
+
H_0: U\perp X\mid Z.
|
|
17
|
+
$$
|
|
18
|
+
|
|
19
|
+
The repository contains the single-split `scis_test` implementation, the
|
|
20
|
+
Guo--Shah (2025) rank-transformed subsampling aggregation, a small demo, a
|
|
21
|
+
simulation runner, and unit tests. The code is currently a research release
|
|
22
|
+
(`0.1.0`); freeze the version and package settings used for any reported
|
|
23
|
+
numerical experiment.
|
|
24
|
+
|
|
25
|
+
The package is maintained by Qixian Zhong and Rajen Shah and distributed
|
|
26
|
+
under the MIT License.
|
|
27
|
+
|
|
28
|
+
## Repository layout
|
|
29
|
+
|
|
30
|
+
```text
|
|
31
|
+
.
|
|
32
|
+
├── pyproject.toml
|
|
33
|
+
├── src/scistest/
|
|
34
|
+
│ ├── core.py # single-split scisTest and its p-value
|
|
35
|
+
│ ├── aggregation.py # Guo--Shah aggregated p-value
|
|
36
|
+
│ ├── config.py # typed hyperparameter configurations
|
|
37
|
+
│ └── simulation.py # reproducible example data generator
|
|
38
|
+
├── examples/
|
|
39
|
+
│ ├── demo.py
|
|
40
|
+
│ └── demo.ipynb
|
|
41
|
+
├── scripts/run_simulation.py
|
|
42
|
+
└── tests/
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Installation
|
|
46
|
+
|
|
47
|
+
From this directory, create an isolated environment and install the package:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
python -m venv .venv
|
|
51
|
+
source .venv/bin/activate
|
|
52
|
+
python -m pip install --upgrade pip
|
|
53
|
+
python -m pip install -e .
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
For development tools, use `python -m pip install -e '.[dev]'`.
|
|
57
|
+
|
|
58
|
+
## Single-split test
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from scistest import scis_test
|
|
62
|
+
|
|
63
|
+
result = scis_test(
|
|
64
|
+
x=X, # shape (n,) or (n, d_x)
|
|
65
|
+
z=Z, # shape (n,) or (n, d_z)
|
|
66
|
+
time=T, # observed min(event time, censoring time), shape (n,)
|
|
67
|
+
event=Delta, # 1=event observed, 0=right censored, shape (n,)
|
|
68
|
+
random_state=1,
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
print(result.test_statistic)
|
|
72
|
+
print(result.p_value)
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
`scisTest` is provided as an alias for `scis_test`. The returned
|
|
76
|
+
`ScisTestResult` also includes `log_p_value`, the held-out score moments, sample
|
|
77
|
+
indices, and the individual orthogonalized scores. Use
|
|
78
|
+
`result.as_dict(include_arrays=False)` for JSON-friendly scalar output.
|
|
79
|
+
|
|
80
|
+
For each split, the package:
|
|
81
|
+
|
|
82
|
+
1. fits the null hazard $\lambda(t\mid Z)=\exp\{g(t,Z)\}$ on the hunting sample
|
|
83
|
+
using the penalized negative counting-process log-likelihood;
|
|
84
|
+
2. learns $\phi$, by default with a DNN, by minimizing exactly
|
|
85
|
+
$\sum_i V_i(\phi)^2-\sum_i V_i(\phi)$;
|
|
86
|
+
3. fits engression for the joint conditional distribution
|
|
87
|
+
$P_{(X,T)\mid Z}$ on the test sample;
|
|
88
|
+
4. constructs the orthogonalized held-out scores and reports the one-sided
|
|
89
|
+
standard-normal tail probability.
|
|
90
|
+
|
|
91
|
+
Preprocessing constants are estimated only on the hunting sample: `X` is
|
|
92
|
+
min--max scaled, `Z` is standardized, and observed time is divided by the
|
|
93
|
+
maximum hunting-sample time.
|
|
94
|
+
|
|
95
|
+
Hyperparameters are explicit immutable dataclasses:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from scistest import (
|
|
99
|
+
DirectionConfig,
|
|
100
|
+
GeneratorConfig,
|
|
101
|
+
HazardConfig,
|
|
102
|
+
SciTestConfig,
|
|
103
|
+
scis_test,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
result = scis_test(
|
|
107
|
+
X, Z, T, Delta,
|
|
108
|
+
random_state=10,
|
|
109
|
+
hazard_config=HazardConfig(hidden_dims=(32,), epochs=100),
|
|
110
|
+
direction_config=DirectionConfig(hidden_dims=(32, 32), epochs=100),
|
|
111
|
+
generator_config=GeneratorConfig(num_layers=2, hidden_dim=100, epochs=100),
|
|
112
|
+
test_config=SciTestConfig(integration_points=100, generator_draws=100),
|
|
113
|
+
)
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
### Random-forest direction estimator
|
|
117
|
+
|
|
118
|
+
The DNN remains the default. To estimate \(\phi\) with a random forest instead,
|
|
119
|
+
pass `RandomForestDirectionConfig` as `direction_config`:
|
|
120
|
+
|
|
121
|
+
```python
|
|
122
|
+
from scistest import RandomForestDirectionConfig, scis_test
|
|
123
|
+
|
|
124
|
+
result = scis_test(
|
|
125
|
+
X, Z, T, Delta,
|
|
126
|
+
random_state=10,
|
|
127
|
+
direction_config=RandomForestDirectionConfig(
|
|
128
|
+
n_estimators=100,
|
|
129
|
+
max_depth=6,
|
|
130
|
+
min_samples_leaf=5,
|
|
131
|
+
n_jobs=-1,
|
|
132
|
+
),
|
|
133
|
+
)
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The forest partitions the `(time, Z, X)` feature space. Conditional on those
|
|
137
|
+
partitions, its terminal-node values are obtained by sparse least squares using
|
|
138
|
+
|
|
139
|
+
$$
|
|
140
|
+
\sum_i[V_i(\phi)^2-V_i(\phi)].
|
|
141
|
+
$$
|
|
142
|
+
|
|
143
|
+
Thus both direction estimators target the same criterion; only the function
|
|
144
|
+
class and optimization method differ. The hazard and conditional-generator
|
|
145
|
+
estimators are unchanged.
|
|
146
|
+
|
|
147
|
+
## Aggregated p-value with inputs `(K, J, L)`
|
|
148
|
+
|
|
149
|
+
The convenience interface refits the full scisTest procedure for all full-data
|
|
150
|
+
splits and calibration subsamples:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
from scistest import aggregated_scis_test
|
|
154
|
+
|
|
155
|
+
aggregated = aggregated_scis_test(
|
|
156
|
+
X, Z, T, Delta,
|
|
157
|
+
K=3, # number of disjoint subsamples per permutation
|
|
158
|
+
J=5, # independent permutations
|
|
159
|
+
L=6, # random splits per full sample or subsample
|
|
160
|
+
random_state=20260918,
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
print(aggregated.aggregated_p_value) # smoothed right-tail p-value
|
|
164
|
+
print(aggregated.empirical_p_value) # unsmoothed calibration fraction
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
The defaults are `K=3`, `J=5`, and `L=6`, so these three arguments may be
|
|
168
|
+
omitted.
|
|
169
|
+
|
|
170
|
+
Each permutation divides the data into `K` disjoint blocks of common size
|
|
171
|
+
|
|
172
|
+
$$
|
|
173
|
+
\texttt{subsample size}=\lfloor n/K\rfloor.
|
|
174
|
+
$$
|
|
175
|
+
|
|
176
|
+
The remaining `n % K` observations are unused in that permutation. Thus the
|
|
177
|
+
number of calibration rows is
|
|
178
|
+
|
|
179
|
+
$$
|
|
180
|
+
B=JK.
|
|
181
|
+
$$
|
|
182
|
+
|
|
183
|
+
All \(BL\) subsample statistics are pooled for the randomized rank transform
|
|
184
|
+
|
|
185
|
+
$$
|
|
186
|
+
\widetilde H_{b\ell}=\Phi^{-1}[\frac{R_{b\ell}-1/2}{BL}].
|
|
187
|
+
$$
|
|
188
|
+
|
|
189
|
+
The full-sample and row-wise calibration aggregates are arithmetic means over
|
|
190
|
+
the \(L\) split statistics. The primary result is the Gaussian-kernel-smoothed
|
|
191
|
+
right tail used by the authors' MultiSplit implementation. The result also
|
|
192
|
+
retains the strict empirical tail, whose resolution is `1 / B`. Because a
|
|
193
|
+
publication-scale run fits `(B + 1) * L` complete tests, it can be expensive.
|
|
194
|
+
|
|
195
|
+
For another asymptotically standard-normal split statistic, use the generic
|
|
196
|
+
callback API:
|
|
197
|
+
|
|
198
|
+
```python
|
|
199
|
+
from scistest import aggregate_p_value
|
|
200
|
+
|
|
201
|
+
def statistic(indices, split_seed):
|
|
202
|
+
# Subset the data, refit the complete base method using split_seed,
|
|
203
|
+
# and return one one-sided standard-normal statistic.
|
|
204
|
+
return my_statistic(indices, split_seed)
|
|
205
|
+
|
|
206
|
+
aggregated = aggregate_p_value(
|
|
207
|
+
n=len(T),
|
|
208
|
+
statistic=statistic,
|
|
209
|
+
K=3,
|
|
210
|
+
J=5,
|
|
211
|
+
L=6,
|
|
212
|
+
random_state=20260918,
|
|
213
|
+
)
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
If the full and subsample statistics have already been computed, call
|
|
217
|
+
`calibrate_aggregated_p_value(observed, subsampled, K, J, L, n=n)`. This is
|
|
218
|
+
useful for checkpointed cluster runs.
|
|
219
|
+
|
|
220
|
+
## Demo and simulation
|
|
221
|
+
|
|
222
|
+
Run a quick single-split example:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
python examples/demo.py
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
For an interactive walkthrough of the DNN direction, random-forest direction,
|
|
229
|
+
and `K`-based aggregation interfaces, open `examples/demo.ipynb`.
|
|
230
|
+
|
|
231
|
+
Add `--aggregate` to demonstrate the complete aggregation workflow with small
|
|
232
|
+
settings. Run a reproducible null/alternative experiment with:
|
|
233
|
+
|
|
234
|
+
```bash
|
|
235
|
+
python scripts/run_simulation.py \
|
|
236
|
+
--repetitions 20 \
|
|
237
|
+
--sample-size 400 \
|
|
238
|
+
--scenario both \
|
|
239
|
+
--output results/simulation.csv
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
Aggregation can be enabled with `--aggregate`; the command-line defaults are
|
|
243
|
+
also `--K 3 --J 5 --L 6`. It is much more computationally demanding because
|
|
244
|
+
every calibration statistic refits all nuisance models.
|
|
245
|
+
|
|
246
|
+
## Tests
|
|
247
|
+
|
|
248
|
+
```bash
|
|
249
|
+
pytest
|
|
250
|
+
ruff check .
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
## Reference
|
|
254
|
+
Qixian Zhong and Rajen D. Shah (2026), “Conditional Independence Testing with Survival Data”.
|
|
255
|
+
|
|
256
|
+
Guo, F. Richard and Rajen D. Shah (2025),
|
|
257
|
+
“Rank-transformed subsampling: inference for multiple data splitting and
|
|
258
|
+
exchangeable p-values,” *Journal of the Royal Statistical Society Series B*,
|
|
259
|
+
87(1), 256–286.
|