falsesync 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Tatsuki Onishi
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,223 @@
1
+ Metadata-Version: 2.4
2
+ Name: falsesync
3
+ Version: 0.1.2
4
+ Summary: Aggregation-induced false synchrony: estimands, diagnostics, and simulation for change-point analysis on aggregated panels
5
+ Author: Tatsuki Onishi
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/bougtoir/falsesync
8
+ Project-URL: Source, https://github.com/bougtoir/falsesync
9
+ Project-URL: Issues, https://github.com/bougtoir/falsesync/issues
10
+ Project-URL: Changelog, https://github.com/bougtoir/falsesync/blob/main/CHANGELOG.md
11
+ Project-URL: DOI, https://doi.org/10.5281/zenodo.23233536
12
+ Keywords: change-point analysis,aggregation,synchrony,panel time series,calibration
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: numpy<2.4,>=2.0
26
+ Requires-Dist: scipy>=1.11
27
+ Requires-Dist: pandas>=2.0
28
+ Requires-Dist: statsmodels>=0.14
29
+ Requires-Dist: matplotlib>=3.8
30
+ Requires-Dist: joblib>=1.3
31
+ Requires-Dist: pyyaml>=6
32
+ Requires-Dist: scikit-learn<1.8,>=1.5
33
+ Provides-Extra: dev
34
+ Requires-Dist: pytest>=8; extra == "dev"
35
+ Requires-Dist: ruff>=0.5; extra == "dev"
36
+ Requires-Dist: black>=24; extra == "dev"
37
+ Provides-Extra: cp
38
+ Requires-Dist: ruptures>=1.1; extra == "cp"
39
+ Dynamic: license-file
40
+
41
+ # falsesync
42
+
43
+ Diagnostics for aggregation-induced false synchrony in change-point analysis.
44
+
45
+ ## Scientific motivation
46
+
47
+ Many engineering and observational studies average a panel of unit-level
48
+ series (turbines, battery cells, sensors, regions) and then fit a change point
49
+ to the aggregate. A sharp aggregate breakpoint is easily read as evidence that
50
+ the units changed together. That reading is not justified in general: when
51
+ unit transitions occur at heterogeneous times \(\tau_i \sim F_\tau\), the
52
+ aggregate mean is a smoothed curve \(m(t) = a + A\,(g * F_\tau)(t)\) that can
53
+ still produce a well-defined, strong breakpoint. The fitted aggregate
54
+ breakpoint is a functional \(T_M(F_\tau, g, w, \text{window}, \text{noise})\)
55
+ that depends on the breakpoint operator \(M\), the observation window, the
56
+ weights, the observation process, and the noise, and it need not coincide with
57
+ the mean, median, or mode of the timing distribution.
58
+
59
+ ## What falsesync does
60
+
61
+ - simulates unit-level transition panels with configurable timing laws,
62
+ transition shapes, weights, observation windows, trends, and noise;
63
+ - computes weighted aggregates and aggregate breakpoints under several
64
+ operators (least squares, maximum slope, CUSUM, binary segmentation) and
65
+ reports the spread across operators;
66
+ - estimates unit-level breakpoints with bootstrap uncertainty and a
67
+ floor-corrected timing dispersion
68
+ \(\mathrm{sd}_{\mathrm{corr}} = \sqrt{\max(\mathrm{sd}^2(\hat\tau_i) - \bar s^2, 0)}\);
69
+ - returns calibrated probabilities for five timing regimes
70
+ (`SYNCHRONOUS`, `NEAR_SYNCHRONOUS`, `DIFFUSE_ASYNCHRONOUS`, `CLUSTERED`,
71
+ `NO_TRANSITION`) from a classifier calibrated on a separate simulation grid;
72
+ - issues a false common-event warning when the aggregate break is strong but
73
+ the calibrated probability of synchrony is low, plus warnings for trend
74
+ confounding, operator dependence, and observation-process risks.
75
+
76
+ ## What falsesync does not claim
77
+
78
+ - It is not a causal method and does not identify what caused any transition.
79
+ - A low synchrony probability is a warning that the aggregate break should not
80
+ be read as a common event; it is not a test that rejects synchrony.
81
+ - The regime probabilities are calibrated only for data-generating conditions
82
+ resembling the calibration grid (see `simulations/configs/`).
83
+
84
+ ## Known limitations
85
+
86
+ - Informative observation processes (entry or dropout related to transition
87
+ timing) can create apparent synchrony; this failure mode is detectable in
88
+ simulation but not corrected by the package.
89
+ - Trend heterogeneity across units can be confounded with timing
90
+ heterogeneity.
91
+ - Unit-level break uncertainty inflates apparent timing dispersion; the
92
+ floor correction reduces but does not remove this effect.
93
+ - The aggregate breakpoint location is operator-dependent; different
94
+ operators can give materially different locations on the same aggregate.
95
+ - Without unit-level data, synchrony cannot be certified from the aggregate
96
+ alone.
97
+
98
+ ## Installation
99
+
100
+ Python >= 3.10.
101
+
102
+ ```bash
103
+ git clone https://github.com/bougtoir/falsesync.git
104
+ cd falsesync
105
+ pip install . # or: pip install -e ".[dev]" for tests
106
+ ```
107
+
108
+ Optional: `pip install ".[cp]"` adds `ruptures`-backed detectors. The
109
+ reported binary-segmentation results were computed with `ruptures` 1.1.10, so
110
+ FULL replication requires `pip install ".[dev,cp]"`.
111
+
112
+ ## Minimal example
113
+
114
+ ```bash
115
+ python examples/minimal_example.py
116
+ ```
117
+
118
+ The example simulates 80 units with diffuse (normal, SD 1.1) transition times,
119
+ fits `FalseSynchronyModel` with the shipped calibrated classifier, and prints
120
+ the aggregate breakpoint, the corrected timing dispersion, the five regime
121
+ probabilities, and the warnings. In outline:
122
+
123
+ ```python
124
+ import joblib
125
+ import numpy as np
126
+ from falsesync import simulation
127
+ from falsesync.model import FalseSynchronyModel
128
+
129
+ rng = np.random.default_rng(1)
130
+ t = np.linspace(0, 10, 180)
131
+ taus = simulation.sample_taus(80, {"kind": "normal", "mu": 5.0, "sigma": 1.1}, rng)
132
+ panel = simulation.simulate_panel(t, taus, shape="logistic", amplitude=1.5,
133
+ width=0.3, noise_sd=0.2, rng=rng)
134
+ clf = joblib.load("simulations/calibration/classifier.joblib")
135
+ res = FalseSynchronyModel(classifier=clf).fit(t, panel.values)
136
+ print(res.aggregate_breakpoint, res.timing_dispersion["sd_corrected"])
137
+ print(res.regime_probabilities, res.interpretation_warning)
138
+ ```
139
+
140
+ ## Main workflow
141
+
142
+ 1. `simulations/run_grid.py` simulates a configuration grid
143
+ (`simulations/configs/*.yaml`) and extracts diagnostic features.
144
+ 2. `simulations/run_calibration.py` fits and temperature-calibrates the
145
+ regime classifier on the calibration grid
146
+ (`simulations/calibration/classifier.joblib`).
147
+ 3. `simulations/run_evaluation.py` evaluates the frozen classifier on the
148
+ locked evaluation grid.
149
+ 4. `simulations/run_stress.py`, `run_missingness_mitigation.py`, and
150
+ `run_near_sync.py` run the robustness and stress conditions.
151
+ 5. `simulations/run_kelmarsh.py` and `simulations/run_nasa_battery.py` run the
152
+ two empirical demonstrations.
153
+ 6. `simulations/make_phase3_figures.py` builds Figures 1-6.
154
+
155
+ ## Reproducing the Technometrics results
156
+
157
+ | Mode | Command | Content | Runtime (1 CPU core) |
158
+ |-------|----------------------------------|-------------------------------------------------------------------------|----------------------|
159
+ | QUICK | `python replication/run_quick.py` | tests, minimal example, smoke grid, re-evaluation of the locked grid with the shipped classifier, regeneration of Figures 1-4, comparison with reference tables and figures | < 1 min |
160
+ | FULL | `python replication/run_full.py` | data download and checksum verification, calibration, locked evaluation, stress tests, both empirical analyses, all figures, comparison with reference tables; needs the `cp` extra | about 15 min |
161
+
162
+ Both scripts exit non-zero if any regenerated result table differs from the
163
+ shipped reference copy. `make all` / `make quick` run the same steps.
164
+
165
+ The reference figures were rendered with matplotlib 3.10. matplotlib 3.11
166
+ reproduces every result table exactly but renders Figures 1-4 with small
167
+ pixel differences, which the QUICK figure comparison reports as `DIFFERS`;
168
+ use `pip install "matplotlib<3.11"` for a pixel-level figure match.
169
+
170
+ **Locked evaluation.** `simulations/configs/evaluation_locked.yaml` (seeds
171
+ 3000-3999) was fixed and committed before the evaluation was run and was not
172
+ edited afterwards; its SHA-256 is recorded in `simulations/configs/SHA256SUMS`.
173
+ The calibration grid (`calibration.yaml`) and the development grid
174
+ (`development.yaml`) use disjoint seed ranges. `METHOD_FREEZE.md` lists the
175
+ frozen thresholds and method choices.
176
+
177
+ **Traceability.** `manuscript_number_trace.csv` maps every number reported in
178
+ the manuscript to the result file and column that produces it;
179
+ `figure_manifest.csv` and `table_manifest.csv` map figures and tables to their
180
+ generating scripts and inputs.
181
+
182
+ ## Data acquisition
183
+
184
+ No raw data are redistributed in this repository. `python fetch_data.py`
185
+ downloads the two public datasets from their original repositories, verifies
186
+ each file against the SHA-256 recorded in `data/acquisition_ledger.csv`, and
187
+ extracts the files used by the analyses into `data/raw/` (git-ignored). See
188
+ `data/README.md` for sources, licenses, and citations.
189
+
190
+ ## Repository structure
191
+
192
+ ```text
193
+ src/falsesync/ package source
194
+ tests/ unit tests (theory identities, workflow)
195
+ examples/ minimal example
196
+ simulations/ simulation, calibration, evaluation, empirical scripts
197
+ configs/ development, calibration, locked evaluation grids
198
+ calibration/ calibrated classifier artifact
199
+ results/ simulated grid outputs
200
+ replication/ QUICK and FULL replication entry points
201
+ proofs/ proofs of propositions P1-P8
202
+ outputs/ reference figures
203
+ data/ acquisition ledger and data documentation
204
+ *.csv reference result tables
205
+ math_specification.md, diagnostic_specification.md, METHOD_FREEZE.md,
206
+ counterexamples.md method documentation
207
+ ```
208
+
209
+ ## Citation
210
+
211
+ See `CITATION.cff`. Please cite the software and the accompanying manuscript:
212
+ T. Onishi, "Aggregation-Induced False Synchrony in Change-Point Analysis"
213
+ (manuscript submitted to *Technometrics*).
214
+
215
+ ## License
216
+
217
+ MIT (see `LICENSE`). Data obtained through `fetch_data.py` remain under their
218
+ original licenses (see `data/README.md`).
219
+
220
+ ## Manuscript status
221
+
222
+ Version 0.1.2 corresponds to the manuscript as submitted to *Technometrics*.
223
+ The manuscript has not been peer reviewed or accepted.
@@ -0,0 +1,183 @@
1
+ # falsesync
2
+
3
+ Diagnostics for aggregation-induced false synchrony in change-point analysis.
4
+
5
+ ## Scientific motivation
6
+
7
+ Many engineering and observational studies average a panel of unit-level
8
+ series (turbines, battery cells, sensors, regions) and then fit a change point
9
+ to the aggregate. A sharp aggregate breakpoint is easily read as evidence that
10
+ the units changed together. That reading is not justified in general: when
11
+ unit transitions occur at heterogeneous times \(\tau_i \sim F_\tau\), the
12
+ aggregate mean is a smoothed curve \(m(t) = a + A\,(g * F_\tau)(t)\) that can
13
+ still produce a well-defined, strong breakpoint. The fitted aggregate
14
+ breakpoint is a functional \(T_M(F_\tau, g, w, \text{window}, \text{noise})\)
15
+ that depends on the breakpoint operator \(M\), the observation window, the
16
+ weights, the observation process, and the noise, and it need not coincide with
17
+ the mean, median, or mode of the timing distribution.
18
+
19
+ ## What falsesync does
20
+
21
+ - simulates unit-level transition panels with configurable timing laws,
22
+ transition shapes, weights, observation windows, trends, and noise;
23
+ - computes weighted aggregates and aggregate breakpoints under several
24
+ operators (least squares, maximum slope, CUSUM, binary segmentation) and
25
+ reports the spread across operators;
26
+ - estimates unit-level breakpoints with bootstrap uncertainty and a
27
+ floor-corrected timing dispersion
28
+ \(\mathrm{sd}_{\mathrm{corr}} = \sqrt{\max(\mathrm{sd}^2(\hat\tau_i) - \bar s^2, 0)}\);
29
+ - returns calibrated probabilities for five timing regimes
30
+ (`SYNCHRONOUS`, `NEAR_SYNCHRONOUS`, `DIFFUSE_ASYNCHRONOUS`, `CLUSTERED`,
31
+ `NO_TRANSITION`) from a classifier calibrated on a separate simulation grid;
32
+ - issues a false common-event warning when the aggregate break is strong but
33
+ the calibrated probability of synchrony is low, plus warnings for trend
34
+ confounding, operator dependence, and observation-process risks.
35
+
36
+ ## What falsesync does not claim
37
+
38
+ - It is not a causal method and does not identify what caused any transition.
39
+ - A low synchrony probability is a warning that the aggregate break should not
40
+ be read as a common event; it is not a test that rejects synchrony.
41
+ - The regime probabilities are calibrated only for data-generating conditions
42
+ resembling the calibration grid (see `simulations/configs/`).
43
+
44
+ ## Known limitations
45
+
46
+ - Informative observation processes (entry or dropout related to transition
47
+ timing) can create apparent synchrony; this failure mode is detectable in
48
+ simulation but not corrected by the package.
49
+ - Trend heterogeneity across units can be confounded with timing
50
+ heterogeneity.
51
+ - Unit-level break uncertainty inflates apparent timing dispersion; the
52
+ floor correction reduces but does not remove this effect.
53
+ - The aggregate breakpoint location is operator-dependent; different
54
+ operators can give materially different locations on the same aggregate.
55
+ - Without unit-level data, synchrony cannot be certified from the aggregate
56
+ alone.
57
+
58
+ ## Installation
59
+
60
+ Python >= 3.10.
61
+
62
+ ```bash
63
+ git clone https://github.com/bougtoir/falsesync.git
64
+ cd falsesync
65
+ pip install . # or: pip install -e ".[dev]" for tests
66
+ ```
67
+
68
+ Optional: `pip install ".[cp]"` adds `ruptures`-backed detectors. The
69
+ reported binary-segmentation results were computed with `ruptures` 1.1.10, so
70
+ FULL replication requires `pip install ".[dev,cp]"`.
71
+
72
+ ## Minimal example
73
+
74
+ ```bash
75
+ python examples/minimal_example.py
76
+ ```
77
+
78
+ The example simulates 80 units with diffuse (normal, SD 1.1) transition times,
79
+ fits `FalseSynchronyModel` with the shipped calibrated classifier, and prints
80
+ the aggregate breakpoint, the corrected timing dispersion, the five regime
81
+ probabilities, and the warnings. In outline:
82
+
83
+ ```python
84
+ import joblib
85
+ import numpy as np
86
+ from falsesync import simulation
87
+ from falsesync.model import FalseSynchronyModel
88
+
89
+ rng = np.random.default_rng(1)
90
+ t = np.linspace(0, 10, 180)
91
+ taus = simulation.sample_taus(80, {"kind": "normal", "mu": 5.0, "sigma": 1.1}, rng)
92
+ panel = simulation.simulate_panel(t, taus, shape="logistic", amplitude=1.5,
93
+ width=0.3, noise_sd=0.2, rng=rng)
94
+ clf = joblib.load("simulations/calibration/classifier.joblib")
95
+ res = FalseSynchronyModel(classifier=clf).fit(t, panel.values)
96
+ print(res.aggregate_breakpoint, res.timing_dispersion["sd_corrected"])
97
+ print(res.regime_probabilities, res.interpretation_warning)
98
+ ```
99
+
100
+ ## Main workflow
101
+
102
+ 1. `simulations/run_grid.py` simulates a configuration grid
103
+ (`simulations/configs/*.yaml`) and extracts diagnostic features.
104
+ 2. `simulations/run_calibration.py` fits and temperature-calibrates the
105
+ regime classifier on the calibration grid
106
+ (`simulations/calibration/classifier.joblib`).
107
+ 3. `simulations/run_evaluation.py` evaluates the frozen classifier on the
108
+ locked evaluation grid.
109
+ 4. `simulations/run_stress.py`, `run_missingness_mitigation.py`, and
110
+ `run_near_sync.py` run the robustness and stress conditions.
111
+ 5. `simulations/run_kelmarsh.py` and `simulations/run_nasa_battery.py` run the
112
+ two empirical demonstrations.
113
+ 6. `simulations/make_phase3_figures.py` builds Figures 1-6.
114
+
115
+ ## Reproducing the Technometrics results
116
+
117
+ | Mode | Command | Content | Runtime (1 CPU core) |
118
+ |-------|----------------------------------|-------------------------------------------------------------------------|----------------------|
119
+ | QUICK | `python replication/run_quick.py` | tests, minimal example, smoke grid, re-evaluation of the locked grid with the shipped classifier, regeneration of Figures 1-4, comparison with reference tables and figures | < 1 min |
120
+ | FULL | `python replication/run_full.py` | data download and checksum verification, calibration, locked evaluation, stress tests, both empirical analyses, all figures, comparison with reference tables; needs the `cp` extra | about 15 min |
121
+
122
+ Both scripts exit non-zero if any regenerated result table differs from the
123
+ shipped reference copy. `make all` / `make quick` run the same steps.
124
+
125
+ The reference figures were rendered with matplotlib 3.10. matplotlib 3.11
126
+ reproduces every result table exactly but renders Figures 1-4 with small
127
+ pixel differences, which the QUICK figure comparison reports as `DIFFERS`;
128
+ use `pip install "matplotlib<3.11"` for a pixel-level figure match.
129
+
130
+ **Locked evaluation.** `simulations/configs/evaluation_locked.yaml` (seeds
131
+ 3000-3999) was fixed and committed before the evaluation was run and was not
132
+ edited afterwards; its SHA-256 is recorded in `simulations/configs/SHA256SUMS`.
133
+ The calibration grid (`calibration.yaml`) and the development grid
134
+ (`development.yaml`) use disjoint seed ranges. `METHOD_FREEZE.md` lists the
135
+ frozen thresholds and method choices.
136
+
137
+ **Traceability.** `manuscript_number_trace.csv` maps every number reported in
138
+ the manuscript to the result file and column that produces it;
139
+ `figure_manifest.csv` and `table_manifest.csv` map figures and tables to their
140
+ generating scripts and inputs.
141
+
142
+ ## Data acquisition
143
+
144
+ No raw data are redistributed in this repository. `python fetch_data.py`
145
+ downloads the two public datasets from their original repositories, verifies
146
+ each file against the SHA-256 recorded in `data/acquisition_ledger.csv`, and
147
+ extracts the files used by the analyses into `data/raw/` (git-ignored). See
148
+ `data/README.md` for sources, licenses, and citations.
149
+
150
+ ## Repository structure
151
+
152
+ ```text
153
+ src/falsesync/ package source
154
+ tests/ unit tests (theory identities, workflow)
155
+ examples/ minimal example
156
+ simulations/ simulation, calibration, evaluation, empirical scripts
157
+ configs/ development, calibration, locked evaluation grids
158
+ calibration/ calibrated classifier artifact
159
+ results/ simulated grid outputs
160
+ replication/ QUICK and FULL replication entry points
161
+ proofs/ proofs of propositions P1-P8
162
+ outputs/ reference figures
163
+ data/ acquisition ledger and data documentation
164
+ *.csv reference result tables
165
+ math_specification.md, diagnostic_specification.md, METHOD_FREEZE.md,
166
+ counterexamples.md method documentation
167
+ ```
168
+
169
+ ## Citation
170
+
171
+ See `CITATION.cff`. Please cite the software and the accompanying manuscript:
172
+ T. Onishi, "Aggregation-Induced False Synchrony in Change-Point Analysis"
173
+ (manuscript submitted to *Technometrics*).
174
+
175
+ ## License
176
+
177
+ MIT (see `LICENSE`). Data obtained through `fetch_data.py` remain under their
178
+ original licenses (see `data/README.md`).
179
+
180
+ ## Manuscript status
181
+
182
+ Version 0.1.2 corresponds to the manuscript as submitted to *Technometrics*.
183
+ The manuscript has not been peer reviewed or accepted.
@@ -0,0 +1,56 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "falsesync"
7
+ version = "0.1.2"
8
+ description = "Aggregation-induced false synchrony: estimands, diagnostics, and simulation for change-point analysis on aggregated panels"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Tatsuki Onishi" }]
14
+ keywords = ["change-point analysis", "aggregation", "synchrony", "panel time series", "calibration"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Science/Research",
18
+ "Operating System :: OS Independent",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3 :: Only",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Topic :: Scientific/Engineering :: Mathematics",
25
+ ]
26
+ dependencies = [
27
+ "numpy>=2.0,<2.4",
28
+ "scipy>=1.11",
29
+ "pandas>=2.0",
30
+ "statsmodels>=0.14",
31
+ "matplotlib>=3.8",
32
+ "joblib>=1.3",
33
+ "pyyaml>=6",
34
+ "scikit-learn>=1.5,<1.8",
35
+ ]
36
+
37
+ [project.urls]
38
+ Homepage = "https://github.com/bougtoir/falsesync"
39
+ Source = "https://github.com/bougtoir/falsesync"
40
+ Issues = "https://github.com/bougtoir/falsesync/issues"
41
+ Changelog = "https://github.com/bougtoir/falsesync/blob/main/CHANGELOG.md"
42
+ DOI = "https://doi.org/10.5281/zenodo.23233536"
43
+
44
+ [project.optional-dependencies]
45
+ dev = ["pytest>=8", "ruff>=0.5", "black>=24"]
46
+ cp = ["ruptures>=1.1"]
47
+
48
+ [tool.setuptools.packages.find]
49
+ where = ["src"]
50
+
51
+ [tool.ruff]
52
+ line-length = 100
53
+ target-version = "py310"
54
+
55
+ [tool.black]
56
+ line-length = 100
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,9 @@
1
+ """falsesync — aggregation-induced false synchrony diagnostics.
2
+
3
+ Core model: Y_i(t) = a_i + b_i(t) + A_i g_i((t - tau_i)/h_i) + eps_i(t).
4
+ The aggregate breakpoint is an operator-dependent functional
5
+ T_M(F_tau, g, w, ...) which in general equals no moment of F_tau.
6
+ See math_specification.md and proofs/ for P1-P8.
7
+ """
8
+
9
+ __version__ = "0.1.2"
@@ -0,0 +1,85 @@
1
+ """Weighted aggregation and the population convolution m(t)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import numpy as np
6
+ from scipy.stats import norm
7
+
8
+ from .transitions import g
9
+
10
+
11
+ def weighted_aggregate(
12
+ values: np.ndarray, weights: np.ndarray | None = None
13
+ ) -> np.ndarray:
14
+ """Row-weighted mean over units, respecting NaN observation windows.
15
+
16
+ values: (n_units, n_t); weights: (n_units,) constant over t, or
17
+ (n_units, n_t) for time-varying weights w_i(t).
18
+ Implements m_obs(t) = sum_i w_i 1{obs} Y_i / sum_i w_i 1{obs} (P5).
19
+ """
20
+ v = np.asarray(values, dtype=float)
21
+ n = v.shape[0]
22
+ if weights is None:
23
+ w = np.ones((n, 1))
24
+ else:
25
+ w = np.asarray(weights, dtype=float)
26
+ if w.ndim == 1:
27
+ w = w[:, None]
28
+ elif w.shape != v.shape:
29
+ raise ValueError(
30
+ f"2D weights must match values shape {v.shape}, got {w.shape}"
31
+ )
32
+ obs = np.isfinite(v)
33
+ wv = np.where(obs, v * w, 0.0)
34
+ denom = np.where(obs, w, 0.0).sum(axis=0)
35
+ with np.errstate(invalid="ignore", divide="ignore"):
36
+ return np.where(denom > 0, wv.sum(axis=0) / denom, np.nan)
37
+
38
+
39
+ def timing_cdf(tau_grid: np.ndarray, spec: dict) -> np.ndarray:
40
+ """Population F_tau on a grid (used by the P1/P5 identities)."""
41
+ from .simulation import timing_density
42
+
43
+ tg = np.asarray(tau_grid, dtype=float)
44
+ f = timing_density(tg, spec)
45
+ c = np.concatenate([[0.0], np.cumsum((f[:-1] + f[1:]) / 2 * np.diff(tg))])
46
+ return np.clip(c / c[-1], 0, 1)
47
+
48
+
49
+ def population_m(
50
+ t: np.ndarray,
51
+ spec: dict,
52
+ shape: str = "logistic",
53
+ amplitude: float = 1.0,
54
+ baseline: float = 0.0,
55
+ width: float = 0.3,
56
+ tau_grid: np.ndarray | None = None,
57
+ ) -> np.ndarray:
58
+ """Population mean curve a + A * (g * f_tau)(t) via quadrature.
59
+
60
+ Closed forms used when available: step g -> F_tau(t) (P1);
61
+ probit + normal -> Phi((t-mu)/sqrt(1+sigma^2)) with unit width (P3a).
62
+ """
63
+ t = np.asarray(t, dtype=float)
64
+ kind = spec.get("kind", "normal")
65
+ if shape == "step" and kind in ("normal", "mixture", "uniform"):
66
+ return baseline + amplitude * np.interp(
67
+ t, np.asarray(tau_grid if tau_grid is not None else t),
68
+ timing_cdf(np.asarray(tau_grid if tau_grid is not None else t), spec),
69
+ )
70
+ if shape == "probit" and kind == "normal" and width == 1.0:
71
+ return baseline + amplitude * norm.cdf(
72
+ (t - spec["mu"]) / np.sqrt(1 + spec["sigma"] ** 2)
73
+ )
74
+ tg = (
75
+ np.asarray(tau_grid, dtype=float)
76
+ if tau_grid is not None
77
+ else np.linspace(t.min() - 8 * width, t.max() + 8 * width, 4001)
78
+ )
79
+ from .simulation import timing_density
80
+
81
+ f = timing_density(tg, spec)
82
+ u = (t[:, None] - tg[None, :]) / width
83
+ trapz = getattr(np, "trapezoid", np.trapz)
84
+ conv = trapz(g(u, shape) * f[None, :], tg, axis=1)
85
+ return baseline + amplitude * conv
@@ -0,0 +1,37 @@
1
+ """Event-time realignment (alignment in unit-specific transition time)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import numpy as np
6
+
7
+
8
+ def event_time_realign(
9
+ t: np.ndarray, values: np.ndarray, taus: np.ndarray
10
+ ) -> np.ndarray:
11
+ """Re-express each unit series in event time s = t - tau_i.
12
+
13
+ Returns an (n, 2k+1) array on a common event-time grid centered at 0;
14
+ NaN where a unit's observed window does not cover s.
15
+ """
16
+ t = np.asarray(t, dtype=float)
17
+ taus = np.asarray(taus, dtype=float)
18
+ dt = float(np.median(np.diff(t)))
19
+ half = float(np.minimum(taus - t.min(), t.max() - taus).max())
20
+ k = int(half / dt)
21
+ s_grid = np.arange(-k, k + 1) * dt
22
+ out = np.full((values.shape[0], s_grid.size), np.nan)
23
+ for i in range(values.shape[0]):
24
+ src_t = s_grid + taus[i]
25
+ valid = (src_t >= t.min()) & (src_t <= t.max())
26
+ out[i, valid] = np.interp(src_t[valid], t, values[i])
27
+ return out
28
+
29
+
30
+ def realign_summary(aligned: np.ndarray, s_dt: float) -> dict:
31
+ """Aligned-panel summary for the diagnostic's event_aligned_summary."""
32
+ agg = np.nanmean(aligned, axis=0)
33
+ d = np.gradient(np.nan_to_num(agg, nan=np.nanmean(agg[np.isfinite(agg)])), s_dt)
34
+ return {
35
+ "aligned_max_slope": float(np.nanmax(np.abs(d))),
36
+ "coverage": float(np.isfinite(aligned).mean()),
37
+ }