falsesync 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- falsesync-0.1.2/LICENSE +21 -0
- falsesync-0.1.2/PKG-INFO +223 -0
- falsesync-0.1.2/README.md +183 -0
- falsesync-0.1.2/pyproject.toml +56 -0
- falsesync-0.1.2/setup.cfg +4 -0
- falsesync-0.1.2/src/falsesync/__init__.py +9 -0
- falsesync-0.1.2/src/falsesync/aggregation.py +85 -0
- falsesync-0.1.2/src/falsesync/alignment.py +37 -0
- falsesync-0.1.2/src/falsesync/changepoints.py +146 -0
- falsesync-0.1.2/src/falsesync/datasets.py +68 -0
- falsesync-0.1.2/src/falsesync/diagnostics.py +265 -0
- falsesync-0.1.2/src/falsesync/features.py +164 -0
- falsesync-0.1.2/src/falsesync/inference.py +90 -0
- falsesync-0.1.2/src/falsesync/model.py +221 -0
- falsesync-0.1.2/src/falsesync/plotting.py +59 -0
- falsesync-0.1.2/src/falsesync/regimes.py +124 -0
- falsesync-0.1.2/src/falsesync/simengine.py +277 -0
- falsesync-0.1.2/src/falsesync/simulation.py +110 -0
- falsesync-0.1.2/src/falsesync/transitions.py +56 -0
- falsesync-0.1.2/src/falsesync.egg-info/PKG-INFO +223 -0
- falsesync-0.1.2/src/falsesync.egg-info/SOURCES.txt +24 -0
- falsesync-0.1.2/src/falsesync.egg-info/dependency_links.txt +1 -0
- falsesync-0.1.2/src/falsesync.egg-info/requires.txt +16 -0
- falsesync-0.1.2/src/falsesync.egg-info/top_level.txt +1 -0
- falsesync-0.1.2/tests/test_phase3.py +116 -0
- falsesync-0.1.2/tests/test_theory.py +127 -0
falsesync-0.1.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Tatsuki Onishi
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
falsesync-0.1.2/PKG-INFO
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: falsesync
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: Aggregation-induced false synchrony: estimands, diagnostics, and simulation for change-point analysis on aggregated panels
|
|
5
|
+
Author: Tatsuki Onishi
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/bougtoir/falsesync
|
|
8
|
+
Project-URL: Source, https://github.com/bougtoir/falsesync
|
|
9
|
+
Project-URL: Issues, https://github.com/bougtoir/falsesync/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/bougtoir/falsesync/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: DOI, https://doi.org/10.5281/zenodo.23233536
|
|
12
|
+
Keywords: change-point analysis,aggregation,synchrony,panel time series,calibration
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: numpy<2.4,>=2.0
|
|
26
|
+
Requires-Dist: scipy>=1.11
|
|
27
|
+
Requires-Dist: pandas>=2.0
|
|
28
|
+
Requires-Dist: statsmodels>=0.14
|
|
29
|
+
Requires-Dist: matplotlib>=3.8
|
|
30
|
+
Requires-Dist: joblib>=1.3
|
|
31
|
+
Requires-Dist: pyyaml>=6
|
|
32
|
+
Requires-Dist: scikit-learn<1.8,>=1.5
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
35
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
36
|
+
Requires-Dist: black>=24; extra == "dev"
|
|
37
|
+
Provides-Extra: cp
|
|
38
|
+
Requires-Dist: ruptures>=1.1; extra == "cp"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# falsesync
|
|
42
|
+
|
|
43
|
+
Diagnostics for aggregation-induced false synchrony in change-point analysis.
|
|
44
|
+
|
|
45
|
+
## Scientific motivation
|
|
46
|
+
|
|
47
|
+
Many engineering and observational studies average a panel of unit-level
|
|
48
|
+
series (turbines, battery cells, sensors, regions) and then fit a change point
|
|
49
|
+
to the aggregate. A sharp aggregate breakpoint is easily read as evidence that
|
|
50
|
+
the units changed together. That reading is not justified in general: when
|
|
51
|
+
unit transitions occur at heterogeneous times \(\tau_i \sim F_\tau\), the
|
|
52
|
+
aggregate mean is a smoothed curve \(m(t) = a + A\,(g * F_\tau)(t)\) that can
|
|
53
|
+
still produce a well-defined, strong breakpoint. The fitted aggregate
|
|
54
|
+
breakpoint is a functional \(T_M(F_\tau, g, w, \text{window}, \text{noise})\)
|
|
55
|
+
that depends on the breakpoint operator \(M\), the observation window, the
|
|
56
|
+
weights, the observation process, and the noise, and it need not coincide with
|
|
57
|
+
the mean, median, or mode of the timing distribution.
|
|
58
|
+
|
|
59
|
+
## What falsesync does
|
|
60
|
+
|
|
61
|
+
- simulates unit-level transition panels with configurable timing laws,
|
|
62
|
+
transition shapes, weights, observation windows, trends, and noise;
|
|
63
|
+
- computes weighted aggregates and aggregate breakpoints under several
|
|
64
|
+
operators (least squares, maximum slope, CUSUM, binary segmentation) and
|
|
65
|
+
reports the spread across operators;
|
|
66
|
+
- estimates unit-level breakpoints with bootstrap uncertainty and a
|
|
67
|
+
floor-corrected timing dispersion
|
|
68
|
+
\(\mathrm{sd}_{\mathrm{corr}} = \sqrt{\max(\mathrm{sd}^2(\hat\tau_i) - \bar s^2, 0)}\);
|
|
69
|
+
- returns calibrated probabilities for five timing regimes
|
|
70
|
+
(`SYNCHRONOUS`, `NEAR_SYNCHRONOUS`, `DIFFUSE_ASYNCHRONOUS`, `CLUSTERED`,
|
|
71
|
+
`NO_TRANSITION`) from a classifier calibrated on a separate simulation grid;
|
|
72
|
+
- issues a false common-event warning when the aggregate break is strong but
|
|
73
|
+
the calibrated probability of synchrony is low, plus warnings for trend
|
|
74
|
+
confounding, operator dependence, and observation-process risks.
|
|
75
|
+
|
|
76
|
+
## What falsesync does not claim
|
|
77
|
+
|
|
78
|
+
- It is not a causal method and does not identify what caused any transition.
|
|
79
|
+
- A low synchrony probability is a warning that the aggregate break should not
|
|
80
|
+
be read as a common event; it is not a test that rejects synchrony.
|
|
81
|
+
- The regime probabilities are calibrated only for data-generating conditions
|
|
82
|
+
resembling the calibration grid (see `simulations/configs/`).
|
|
83
|
+
|
|
84
|
+
## Known limitations
|
|
85
|
+
|
|
86
|
+
- Informative observation processes (entry or dropout related to transition
|
|
87
|
+
timing) can create apparent synchrony; this failure mode is detectable in
|
|
88
|
+
simulation but not corrected by the package.
|
|
89
|
+
- Trend heterogeneity across units can be confounded with timing
|
|
90
|
+
heterogeneity.
|
|
91
|
+
- Unit-level break uncertainty inflates apparent timing dispersion; the
|
|
92
|
+
floor correction reduces but does not remove this effect.
|
|
93
|
+
- The aggregate breakpoint location is operator-dependent; different
|
|
94
|
+
operators can give materially different locations on the same aggregate.
|
|
95
|
+
- Without unit-level data, synchrony cannot be certified from the aggregate
|
|
96
|
+
alone.
|
|
97
|
+
|
|
98
|
+
## Installation
|
|
99
|
+
|
|
100
|
+
Python >= 3.10.
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
git clone https://github.com/bougtoir/falsesync.git
|
|
104
|
+
cd falsesync
|
|
105
|
+
pip install . # or: pip install -e ".[dev]" for tests
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Optional: `pip install ".[cp]"` adds `ruptures`-backed detectors. The
|
|
109
|
+
reported binary-segmentation results were computed with `ruptures` 1.1.10, so
|
|
110
|
+
FULL replication requires `pip install ".[dev,cp]"`.
|
|
111
|
+
|
|
112
|
+
## Minimal example
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
python examples/minimal_example.py
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
The example simulates 80 units with diffuse (normal, SD 1.1) transition times,
|
|
119
|
+
fits `FalseSynchronyModel` with the shipped calibrated classifier, and prints
|
|
120
|
+
the aggregate breakpoint, the corrected timing dispersion, the five regime
|
|
121
|
+
probabilities, and the warnings. In outline:
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
import joblib
|
|
125
|
+
import numpy as np
|
|
126
|
+
from falsesync import simulation
|
|
127
|
+
from falsesync.model import FalseSynchronyModel
|
|
128
|
+
|
|
129
|
+
rng = np.random.default_rng(1)
|
|
130
|
+
t = np.linspace(0, 10, 180)
|
|
131
|
+
taus = simulation.sample_taus(80, {"kind": "normal", "mu": 5.0, "sigma": 1.1}, rng)
|
|
132
|
+
panel = simulation.simulate_panel(t, taus, shape="logistic", amplitude=1.5,
|
|
133
|
+
width=0.3, noise_sd=0.2, rng=rng)
|
|
134
|
+
clf = joblib.load("simulations/calibration/classifier.joblib")
|
|
135
|
+
res = FalseSynchronyModel(classifier=clf).fit(t, panel.values)
|
|
136
|
+
print(res.aggregate_breakpoint, res.timing_dispersion["sd_corrected"])
|
|
137
|
+
print(res.regime_probabilities, res.interpretation_warning)
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Main workflow
|
|
141
|
+
|
|
142
|
+
1. `simulations/run_grid.py` simulates a configuration grid
|
|
143
|
+
(`simulations/configs/*.yaml`) and extracts diagnostic features.
|
|
144
|
+
2. `simulations/run_calibration.py` fits and temperature-calibrates the
|
|
145
|
+
regime classifier on the calibration grid
|
|
146
|
+
(`simulations/calibration/classifier.joblib`).
|
|
147
|
+
3. `simulations/run_evaluation.py` evaluates the frozen classifier on the
|
|
148
|
+
locked evaluation grid.
|
|
149
|
+
4. `simulations/run_stress.py`, `run_missingness_mitigation.py`, and
|
|
150
|
+
`run_near_sync.py` run the robustness and stress conditions.
|
|
151
|
+
5. `simulations/run_kelmarsh.py` and `simulations/run_nasa_battery.py` run the
|
|
152
|
+
two empirical demonstrations.
|
|
153
|
+
6. `simulations/make_phase3_figures.py` builds Figures 1-6.
|
|
154
|
+
|
|
155
|
+
## Reproducing the Technometrics results
|
|
156
|
+
|
|
157
|
+
| Mode | Command | Content | Runtime (1 CPU core) |
|
|
158
|
+
|-------|----------------------------------|-------------------------------------------------------------------------|----------------------|
|
|
159
|
+
| QUICK | `python replication/run_quick.py` | tests, minimal example, smoke grid, re-evaluation of the locked grid with the shipped classifier, regeneration of Figures 1-4, comparison with reference tables and figures | < 1 min |
|
|
160
|
+
| FULL | `python replication/run_full.py` | data download and checksum verification, calibration, locked evaluation, stress tests, both empirical analyses, all figures, comparison with reference tables; needs the `cp` extra | about 15 min |
|
|
161
|
+
|
|
162
|
+
Both scripts exit non-zero if any regenerated result table differs from the
|
|
163
|
+
shipped reference copy. `make all` / `make quick` run the same steps.
|
|
164
|
+
|
|
165
|
+
The reference figures were rendered with matplotlib 3.10. matplotlib 3.11
|
|
166
|
+
reproduces every result table exactly but renders Figures 1-4 with small
|
|
167
|
+
pixel differences, which the QUICK figure comparison reports as `DIFFERS`;
|
|
168
|
+
use `pip install "matplotlib<3.11"` for a pixel-level figure match.
|
|
169
|
+
|
|
170
|
+
**Locked evaluation.** `simulations/configs/evaluation_locked.yaml` (seeds
|
|
171
|
+
3000-3999) was fixed and committed before the evaluation was run and was not
|
|
172
|
+
edited afterwards; its SHA-256 is recorded in `simulations/configs/SHA256SUMS`.
|
|
173
|
+
The calibration grid (`calibration.yaml`) and the development grid
|
|
174
|
+
(`development.yaml`) use disjoint seed ranges. `METHOD_FREEZE.md` lists the
|
|
175
|
+
frozen thresholds and method choices.
|
|
176
|
+
|
|
177
|
+
**Traceability.** `manuscript_number_trace.csv` maps every number reported in
|
|
178
|
+
the manuscript to the result file and column that produces it;
|
|
179
|
+
`figure_manifest.csv` and `table_manifest.csv` map figures and tables to their
|
|
180
|
+
generating scripts and inputs.
|
|
181
|
+
|
|
182
|
+
## Data acquisition
|
|
183
|
+
|
|
184
|
+
No raw data are redistributed in this repository. `python fetch_data.py`
|
|
185
|
+
downloads the two public datasets from their original repositories, verifies
|
|
186
|
+
each file against the SHA-256 recorded in `data/acquisition_ledger.csv`, and
|
|
187
|
+
extracts the files used by the analyses into `data/raw/` (git-ignored). See
|
|
188
|
+
`data/README.md` for sources, licenses, and citations.
|
|
189
|
+
|
|
190
|
+
## Repository structure
|
|
191
|
+
|
|
192
|
+
```text
|
|
193
|
+
src/falsesync/ package source
|
|
194
|
+
tests/ unit tests (theory identities, workflow)
|
|
195
|
+
examples/ minimal example
|
|
196
|
+
simulations/ simulation, calibration, evaluation, empirical scripts
|
|
197
|
+
configs/ development, calibration, locked evaluation grids
|
|
198
|
+
calibration/ calibrated classifier artifact
|
|
199
|
+
results/ simulated grid outputs
|
|
200
|
+
replication/ QUICK and FULL replication entry points
|
|
201
|
+
proofs/ proofs of propositions P1-P8
|
|
202
|
+
outputs/ reference figures
|
|
203
|
+
data/ acquisition ledger and data documentation
|
|
204
|
+
*.csv reference result tables
|
|
205
|
+
math_specification.md, diagnostic_specification.md, METHOD_FREEZE.md,
|
|
206
|
+
counterexamples.md method documentation
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
## Citation
|
|
210
|
+
|
|
211
|
+
See `CITATION.cff`. Please cite the software and the accompanying manuscript:
|
|
212
|
+
T. Onishi, "Aggregation-Induced False Synchrony in Change-Point Analysis"
|
|
213
|
+
(manuscript submitted to *Technometrics*).
|
|
214
|
+
|
|
215
|
+
## License
|
|
216
|
+
|
|
217
|
+
MIT (see `LICENSE`). Data obtained through `fetch_data.py` remain under their
|
|
218
|
+
original licenses (see `data/README.md`).
|
|
219
|
+
|
|
220
|
+
## Manuscript status
|
|
221
|
+
|
|
222
|
+
Version 0.1.2 corresponds to the manuscript as submitted to *Technometrics*.
|
|
223
|
+
The manuscript has not been peer reviewed or accepted.
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# falsesync
|
|
2
|
+
|
|
3
|
+
Diagnostics for aggregation-induced false synchrony in change-point analysis.
|
|
4
|
+
|
|
5
|
+
## Scientific motivation
|
|
6
|
+
|
|
7
|
+
Many engineering and observational studies average a panel of unit-level
|
|
8
|
+
series (turbines, battery cells, sensors, regions) and then fit a change point
|
|
9
|
+
to the aggregate. A sharp aggregate breakpoint is easily read as evidence that
|
|
10
|
+
the units changed together. That reading is not justified in general: when
|
|
11
|
+
unit transitions occur at heterogeneous times \(\tau_i \sim F_\tau\), the
|
|
12
|
+
aggregate mean is a smoothed curve \(m(t) = a + A\,(g * F_\tau)(t)\) that can
|
|
13
|
+
still produce a well-defined, strong breakpoint. The fitted aggregate
|
|
14
|
+
breakpoint is a functional \(T_M(F_\tau, g, w, \text{window}, \text{noise})\)
|
|
15
|
+
that depends on the breakpoint operator \(M\), the observation window, the
|
|
16
|
+
weights, the observation process, and the noise, and it need not coincide with
|
|
17
|
+
the mean, median, or mode of the timing distribution.
|
|
18
|
+
|
|
19
|
+
## What falsesync does
|
|
20
|
+
|
|
21
|
+
- simulates unit-level transition panels with configurable timing laws,
|
|
22
|
+
transition shapes, weights, observation windows, trends, and noise;
|
|
23
|
+
- computes weighted aggregates and aggregate breakpoints under several
|
|
24
|
+
operators (least squares, maximum slope, CUSUM, binary segmentation) and
|
|
25
|
+
reports the spread across operators;
|
|
26
|
+
- estimates unit-level breakpoints with bootstrap uncertainty and a
|
|
27
|
+
floor-corrected timing dispersion
|
|
28
|
+
\(\mathrm{sd}_{\mathrm{corr}} = \sqrt{\max(\mathrm{sd}^2(\hat\tau_i) - \bar s^2, 0)}\);
|
|
29
|
+
- returns calibrated probabilities for five timing regimes
|
|
30
|
+
(`SYNCHRONOUS`, `NEAR_SYNCHRONOUS`, `DIFFUSE_ASYNCHRONOUS`, `CLUSTERED`,
|
|
31
|
+
`NO_TRANSITION`) from a classifier calibrated on a separate simulation grid;
|
|
32
|
+
- issues a false common-event warning when the aggregate break is strong but
|
|
33
|
+
the calibrated probability of synchrony is low, plus warnings for trend
|
|
34
|
+
confounding, operator dependence, and observation-process risks.
|
|
35
|
+
|
|
36
|
+
## What falsesync does not claim
|
|
37
|
+
|
|
38
|
+
- It is not a causal method and does not identify what caused any transition.
|
|
39
|
+
- A low synchrony probability is a warning that the aggregate break should not
|
|
40
|
+
be read as a common event; it is not a test that rejects synchrony.
|
|
41
|
+
- The regime probabilities are calibrated only for data-generating conditions
|
|
42
|
+
resembling the calibration grid (see `simulations/configs/`).
|
|
43
|
+
|
|
44
|
+
## Known limitations
|
|
45
|
+
|
|
46
|
+
- Informative observation processes (entry or dropout related to transition
|
|
47
|
+
timing) can create apparent synchrony; this failure mode is detectable in
|
|
48
|
+
simulation but not corrected by the package.
|
|
49
|
+
- Trend heterogeneity across units can be confounded with timing
|
|
50
|
+
heterogeneity.
|
|
51
|
+
- Unit-level break uncertainty inflates apparent timing dispersion; the
|
|
52
|
+
floor correction reduces but does not remove this effect.
|
|
53
|
+
- The aggregate breakpoint location is operator-dependent; different
|
|
54
|
+
operators can give materially different locations on the same aggregate.
|
|
55
|
+
- Without unit-level data, synchrony cannot be certified from the aggregate
|
|
56
|
+
alone.
|
|
57
|
+
|
|
58
|
+
## Installation
|
|
59
|
+
|
|
60
|
+
Python >= 3.10.
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
git clone https://github.com/bougtoir/falsesync.git
|
|
64
|
+
cd falsesync
|
|
65
|
+
pip install . # or: pip install -e ".[dev]" for tests
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Optional: `pip install ".[cp]"` adds `ruptures`-backed detectors. The
|
|
69
|
+
reported binary-segmentation results were computed with `ruptures` 1.1.10, so
|
|
70
|
+
FULL replication requires `pip install ".[dev,cp]"`.
|
|
71
|
+
|
|
72
|
+
## Minimal example
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
python examples/minimal_example.py
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
The example simulates 80 units with diffuse (normal, SD 1.1) transition times,
|
|
79
|
+
fits `FalseSynchronyModel` with the shipped calibrated classifier, and prints
|
|
80
|
+
the aggregate breakpoint, the corrected timing dispersion, the five regime
|
|
81
|
+
probabilities, and the warnings. In outline:
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
import joblib
|
|
85
|
+
import numpy as np
|
|
86
|
+
from falsesync import simulation
|
|
87
|
+
from falsesync.model import FalseSynchronyModel
|
|
88
|
+
|
|
89
|
+
rng = np.random.default_rng(1)
|
|
90
|
+
t = np.linspace(0, 10, 180)
|
|
91
|
+
taus = simulation.sample_taus(80, {"kind": "normal", "mu": 5.0, "sigma": 1.1}, rng)
|
|
92
|
+
panel = simulation.simulate_panel(t, taus, shape="logistic", amplitude=1.5,
|
|
93
|
+
width=0.3, noise_sd=0.2, rng=rng)
|
|
94
|
+
clf = joblib.load("simulations/calibration/classifier.joblib")
|
|
95
|
+
res = FalseSynchronyModel(classifier=clf).fit(t, panel.values)
|
|
96
|
+
print(res.aggregate_breakpoint, res.timing_dispersion["sd_corrected"])
|
|
97
|
+
print(res.regime_probabilities, res.interpretation_warning)
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## Main workflow
|
|
101
|
+
|
|
102
|
+
1. `simulations/run_grid.py` simulates a configuration grid
|
|
103
|
+
(`simulations/configs/*.yaml`) and extracts diagnostic features.
|
|
104
|
+
2. `simulations/run_calibration.py` fits and temperature-calibrates the
|
|
105
|
+
regime classifier on the calibration grid
|
|
106
|
+
(`simulations/calibration/classifier.joblib`).
|
|
107
|
+
3. `simulations/run_evaluation.py` evaluates the frozen classifier on the
|
|
108
|
+
locked evaluation grid.
|
|
109
|
+
4. `simulations/run_stress.py`, `run_missingness_mitigation.py`, and
|
|
110
|
+
`run_near_sync.py` run the robustness and stress conditions.
|
|
111
|
+
5. `simulations/run_kelmarsh.py` and `simulations/run_nasa_battery.py` run the
|
|
112
|
+
two empirical demonstrations.
|
|
113
|
+
6. `simulations/make_phase3_figures.py` builds Figures 1-6.
|
|
114
|
+
|
|
115
|
+
## Reproducing the Technometrics results
|
|
116
|
+
|
|
117
|
+
| Mode | Command | Content | Runtime (1 CPU core) |
|
|
118
|
+
|-------|----------------------------------|-------------------------------------------------------------------------|----------------------|
|
|
119
|
+
| QUICK | `python replication/run_quick.py` | tests, minimal example, smoke grid, re-evaluation of the locked grid with the shipped classifier, regeneration of Figures 1-4, comparison with reference tables and figures | < 1 min |
|
|
120
|
+
| FULL | `python replication/run_full.py` | data download and checksum verification, calibration, locked evaluation, stress tests, both empirical analyses, all figures, comparison with reference tables; needs the `cp` extra | about 15 min |
|
|
121
|
+
|
|
122
|
+
Both scripts exit non-zero if any regenerated result table differs from the
|
|
123
|
+
shipped reference copy. `make all` / `make quick` run the same steps.
|
|
124
|
+
|
|
125
|
+
The reference figures were rendered with matplotlib 3.10. matplotlib 3.11
|
|
126
|
+
reproduces every result table exactly but renders Figures 1-4 with small
|
|
127
|
+
pixel differences, which the QUICK figure comparison reports as `DIFFERS`;
|
|
128
|
+
use `pip install "matplotlib<3.11"` for a pixel-level figure match.
|
|
129
|
+
|
|
130
|
+
**Locked evaluation.** `simulations/configs/evaluation_locked.yaml` (seeds
|
|
131
|
+
3000-3999) was fixed and committed before the evaluation was run and was not
|
|
132
|
+
edited afterwards; its SHA-256 is recorded in `simulations/configs/SHA256SUMS`.
|
|
133
|
+
The calibration grid (`calibration.yaml`) and the development grid
|
|
134
|
+
(`development.yaml`) use disjoint seed ranges. `METHOD_FREEZE.md` lists the
|
|
135
|
+
frozen thresholds and method choices.
|
|
136
|
+
|
|
137
|
+
**Traceability.** `manuscript_number_trace.csv` maps every number reported in
|
|
138
|
+
the manuscript to the result file and column that produces it;
|
|
139
|
+
`figure_manifest.csv` and `table_manifest.csv` map figures and tables to their
|
|
140
|
+
generating scripts and inputs.
|
|
141
|
+
|
|
142
|
+
## Data acquisition
|
|
143
|
+
|
|
144
|
+
No raw data are redistributed in this repository. `python fetch_data.py`
|
|
145
|
+
downloads the two public datasets from their original repositories, verifies
|
|
146
|
+
each file against the SHA-256 recorded in `data/acquisition_ledger.csv`, and
|
|
147
|
+
extracts the files used by the analyses into `data/raw/` (git-ignored). See
|
|
148
|
+
`data/README.md` for sources, licenses, and citations.
|
|
149
|
+
|
|
150
|
+
## Repository structure
|
|
151
|
+
|
|
152
|
+
```text
|
|
153
|
+
src/falsesync/ package source
|
|
154
|
+
tests/ unit tests (theory identities, workflow)
|
|
155
|
+
examples/ minimal example
|
|
156
|
+
simulations/ simulation, calibration, evaluation, empirical scripts
|
|
157
|
+
configs/ development, calibration, locked evaluation grids
|
|
158
|
+
calibration/ calibrated classifier artifact
|
|
159
|
+
results/ simulated grid outputs
|
|
160
|
+
replication/ QUICK and FULL replication entry points
|
|
161
|
+
proofs/ proofs of propositions P1-P8
|
|
162
|
+
outputs/ reference figures
|
|
163
|
+
data/ acquisition ledger and data documentation
|
|
164
|
+
*.csv reference result tables
|
|
165
|
+
math_specification.md, diagnostic_specification.md, METHOD_FREEZE.md,
|
|
166
|
+
counterexamples.md method documentation
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
## Citation
|
|
170
|
+
|
|
171
|
+
See `CITATION.cff`. Please cite the software and the accompanying manuscript:
|
|
172
|
+
T. Onishi, "Aggregation-Induced False Synchrony in Change-Point Analysis"
|
|
173
|
+
(manuscript submitted to *Technometrics*).
|
|
174
|
+
|
|
175
|
+
## License
|
|
176
|
+
|
|
177
|
+
MIT (see `LICENSE`). Data obtained through `fetch_data.py` remain under their
|
|
178
|
+
original licenses (see `data/README.md`).
|
|
179
|
+
|
|
180
|
+
## Manuscript status
|
|
181
|
+
|
|
182
|
+
Version 0.1.2 corresponds to the manuscript as submitted to *Technometrics*.
|
|
183
|
+
The manuscript has not been peer reviewed or accepted.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "falsesync"
|
|
7
|
+
version = "0.1.2"
|
|
8
|
+
description = "Aggregation-induced false synchrony: estimands, diagnostics, and simulation for change-point analysis on aggregated panels"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Tatsuki Onishi" }]
|
|
14
|
+
keywords = ["change-point analysis", "aggregation", "synchrony", "panel time series", "calibration"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Science/Research",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Topic :: Scientific/Engineering :: Mathematics",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"numpy>=2.0,<2.4",
|
|
28
|
+
"scipy>=1.11",
|
|
29
|
+
"pandas>=2.0",
|
|
30
|
+
"statsmodels>=0.14",
|
|
31
|
+
"matplotlib>=3.8",
|
|
32
|
+
"joblib>=1.3",
|
|
33
|
+
"pyyaml>=6",
|
|
34
|
+
"scikit-learn>=1.5,<1.8",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
[project.urls]
|
|
38
|
+
Homepage = "https://github.com/bougtoir/falsesync"
|
|
39
|
+
Source = "https://github.com/bougtoir/falsesync"
|
|
40
|
+
Issues = "https://github.com/bougtoir/falsesync/issues"
|
|
41
|
+
Changelog = "https://github.com/bougtoir/falsesync/blob/main/CHANGELOG.md"
|
|
42
|
+
DOI = "https://doi.org/10.5281/zenodo.23233536"
|
|
43
|
+
|
|
44
|
+
[project.optional-dependencies]
|
|
45
|
+
dev = ["pytest>=8", "ruff>=0.5", "black>=24"]
|
|
46
|
+
cp = ["ruptures>=1.1"]
|
|
47
|
+
|
|
48
|
+
[tool.setuptools.packages.find]
|
|
49
|
+
where = ["src"]
|
|
50
|
+
|
|
51
|
+
[tool.ruff]
|
|
52
|
+
line-length = 100
|
|
53
|
+
target-version = "py310"
|
|
54
|
+
|
|
55
|
+
[tool.black]
|
|
56
|
+
line-length = 100
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""falsesync — aggregation-induced false synchrony diagnostics.
|
|
2
|
+
|
|
3
|
+
Core model: Y_i(t) = a_i + b_i(t) + A_i g_i((t - tau_i)/h_i) + eps_i(t).
|
|
4
|
+
The aggregate breakpoint is an operator-dependent functional
|
|
5
|
+
T_M(F_tau, g, w, ...) which in general equals no moment of F_tau.
|
|
6
|
+
See math_specification.md and proofs/ for P1-P8.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
__version__ = "0.1.2"
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Weighted aggregation and the population convolution m(t)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
from scipy.stats import norm
|
|
7
|
+
|
|
8
|
+
from .transitions import g
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def weighted_aggregate(
|
|
12
|
+
values: np.ndarray, weights: np.ndarray | None = None
|
|
13
|
+
) -> np.ndarray:
|
|
14
|
+
"""Row-weighted mean over units, respecting NaN observation windows.
|
|
15
|
+
|
|
16
|
+
values: (n_units, n_t); weights: (n_units,) constant over t, or
|
|
17
|
+
(n_units, n_t) for time-varying weights w_i(t).
|
|
18
|
+
Implements m_obs(t) = sum_i w_i 1{obs} Y_i / sum_i w_i 1{obs} (P5).
|
|
19
|
+
"""
|
|
20
|
+
v = np.asarray(values, dtype=float)
|
|
21
|
+
n = v.shape[0]
|
|
22
|
+
if weights is None:
|
|
23
|
+
w = np.ones((n, 1))
|
|
24
|
+
else:
|
|
25
|
+
w = np.asarray(weights, dtype=float)
|
|
26
|
+
if w.ndim == 1:
|
|
27
|
+
w = w[:, None]
|
|
28
|
+
elif w.shape != v.shape:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
f"2D weights must match values shape {v.shape}, got {w.shape}"
|
|
31
|
+
)
|
|
32
|
+
obs = np.isfinite(v)
|
|
33
|
+
wv = np.where(obs, v * w, 0.0)
|
|
34
|
+
denom = np.where(obs, w, 0.0).sum(axis=0)
|
|
35
|
+
with np.errstate(invalid="ignore", divide="ignore"):
|
|
36
|
+
return np.where(denom > 0, wv.sum(axis=0) / denom, np.nan)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def timing_cdf(tau_grid: np.ndarray, spec: dict) -> np.ndarray:
|
|
40
|
+
"""Population F_tau on a grid (used by the P1/P5 identities)."""
|
|
41
|
+
from .simulation import timing_density
|
|
42
|
+
|
|
43
|
+
tg = np.asarray(tau_grid, dtype=float)
|
|
44
|
+
f = timing_density(tg, spec)
|
|
45
|
+
c = np.concatenate([[0.0], np.cumsum((f[:-1] + f[1:]) / 2 * np.diff(tg))])
|
|
46
|
+
return np.clip(c / c[-1], 0, 1)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def population_m(
|
|
50
|
+
t: np.ndarray,
|
|
51
|
+
spec: dict,
|
|
52
|
+
shape: str = "logistic",
|
|
53
|
+
amplitude: float = 1.0,
|
|
54
|
+
baseline: float = 0.0,
|
|
55
|
+
width: float = 0.3,
|
|
56
|
+
tau_grid: np.ndarray | None = None,
|
|
57
|
+
) -> np.ndarray:
|
|
58
|
+
"""Population mean curve a + A * (g * f_tau)(t) via quadrature.
|
|
59
|
+
|
|
60
|
+
Closed forms used when available: step g -> F_tau(t) (P1);
|
|
61
|
+
probit + normal -> Phi((t-mu)/sqrt(1+sigma^2)) with unit width (P3a).
|
|
62
|
+
"""
|
|
63
|
+
t = np.asarray(t, dtype=float)
|
|
64
|
+
kind = spec.get("kind", "normal")
|
|
65
|
+
if shape == "step" and kind in ("normal", "mixture", "uniform"):
|
|
66
|
+
return baseline + amplitude * np.interp(
|
|
67
|
+
t, np.asarray(tau_grid if tau_grid is not None else t),
|
|
68
|
+
timing_cdf(np.asarray(tau_grid if tau_grid is not None else t), spec),
|
|
69
|
+
)
|
|
70
|
+
if shape == "probit" and kind == "normal" and width == 1.0:
|
|
71
|
+
return baseline + amplitude * norm.cdf(
|
|
72
|
+
(t - spec["mu"]) / np.sqrt(1 + spec["sigma"] ** 2)
|
|
73
|
+
)
|
|
74
|
+
tg = (
|
|
75
|
+
np.asarray(tau_grid, dtype=float)
|
|
76
|
+
if tau_grid is not None
|
|
77
|
+
else np.linspace(t.min() - 8 * width, t.max() + 8 * width, 4001)
|
|
78
|
+
)
|
|
79
|
+
from .simulation import timing_density
|
|
80
|
+
|
|
81
|
+
f = timing_density(tg, spec)
|
|
82
|
+
u = (t[:, None] - tg[None, :]) / width
|
|
83
|
+
trapz = getattr(np, "trapezoid", np.trapz)
|
|
84
|
+
conv = trapz(g(u, shape) * f[None, :], tg, axis=1)
|
|
85
|
+
return baseline + amplitude * conv
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Event-time realignment (alignment in unit-specific transition time)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def event_time_realign(
|
|
9
|
+
t: np.ndarray, values: np.ndarray, taus: np.ndarray
|
|
10
|
+
) -> np.ndarray:
|
|
11
|
+
"""Re-express each unit series in event time s = t - tau_i.
|
|
12
|
+
|
|
13
|
+
Returns an (n, 2k+1) array on a common event-time grid centered at 0;
|
|
14
|
+
NaN where a unit's observed window does not cover s.
|
|
15
|
+
"""
|
|
16
|
+
t = np.asarray(t, dtype=float)
|
|
17
|
+
taus = np.asarray(taus, dtype=float)
|
|
18
|
+
dt = float(np.median(np.diff(t)))
|
|
19
|
+
half = float(np.minimum(taus - t.min(), t.max() - taus).max())
|
|
20
|
+
k = int(half / dt)
|
|
21
|
+
s_grid = np.arange(-k, k + 1) * dt
|
|
22
|
+
out = np.full((values.shape[0], s_grid.size), np.nan)
|
|
23
|
+
for i in range(values.shape[0]):
|
|
24
|
+
src_t = s_grid + taus[i]
|
|
25
|
+
valid = (src_t >= t.min()) & (src_t <= t.max())
|
|
26
|
+
out[i, valid] = np.interp(src_t[valid], t, values[i])
|
|
27
|
+
return out
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def realign_summary(aligned: np.ndarray, s_dt: float) -> dict:
|
|
31
|
+
"""Aligned-panel summary for the diagnostic's event_aligned_summary."""
|
|
32
|
+
agg = np.nanmean(aligned, axis=0)
|
|
33
|
+
d = np.gradient(np.nan_to_num(agg, nan=np.nanmean(agg[np.isfinite(agg)])), s_dt)
|
|
34
|
+
return {
|
|
35
|
+
"aligned_max_slope": float(np.nanmax(np.abs(d))),
|
|
36
|
+
"coverage": float(np.isfinite(aligned).mean()),
|
|
37
|
+
}
|