falsesync 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
falsesync/simengine.py ADDED
@@ -0,0 +1,277 @@
1
+ """Phase-3 config-driven simulation engine (see simulation_design.md).
2
+
3
+ A SimConfig fully specifies one simulation cell; simulate() returns a Panel
4
+ plus ground-truth regime/timing metadata. All randomness flows through one
5
+ seeded numpy Generator.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+
12
+ import numpy as np
13
+ from scipy.stats import t as t_dist
14
+
15
+ from .aggregation import weighted_aggregate
16
+ from .simulation import Panel, sample_taus, simulate_panel
17
+ from .transitions import g
18
+
19
+ REGIMES = (
20
+ "synchronous",
21
+ "near_synchronous",
22
+ "diffuse",
23
+ "two_cluster",
24
+ "three_cluster",
25
+ "no_transition",
26
+ )
27
+
28
+ F_FAMILIES = (
29
+ "degenerate", "narrow_normal", "broad_normal", "uniform", "skewed",
30
+ "two_normal_mix", "three_normal_mix", "spike_diffuse", "heavy_tail",
31
+ )
32
+
33
+ NOISE_KINDS = ("iid", "t3", "heterosk", "ar1", "seasonal")
34
+ WEIGHT_KINDS = ("equal", "static", "timing_corr", "timevarying")
35
+ MISS_KINDS = ("complete", "mcar", "dropout", "late_entry", "informative")
36
+
37
+
38
+ @dataclass
39
+ class SimConfig:
40
+ n_units: int = 100
41
+ n_t: int = 120
42
+ t_span: tuple = (0.0, 10.0)
43
+ regime: str = "diffuse"
44
+ f_family: str = "broad_normal"
45
+ f_params: dict = field(default_factory=dict) # overrides, e.g. {"sigma":0.8}
46
+ center: float = 5.0 # central transition time
47
+ shape: str = "logistic" # step/logistic/probit/linear
48
+ amplitude: float = 1.0 # effect size (low 0.5, med 1.0, high 2.0 typical)
49
+ width: float = 0.4 # transition width h
50
+ noise_kind: str = "iid"
51
+ noise_sd: float = 0.15
52
+ noise_params: dict = field(default_factory=dict)
53
+ weight_kind: str = "equal"
54
+ weight_params: dict = field(default_factory=dict)
55
+ miss_kind: str = "complete"
56
+ miss_params: dict = field(default_factory=dict)
57
+ amp_hetero: float = 0.0 # sd of A_i / A
58
+ width_hetero: float = 0.0 # sd of h_i / h
59
+ base_hetero: float = 0.0 # sd of a_i
60
+ trend_kind: str = "none" # none/linear/nonlinear/local_drift/mixed
61
+ trend_scale: float = 0.0 # sd of trend slope per unit
62
+ noisevar_hetero: float = 0.0
63
+ seed: int = 0
64
+ cell_id: str = ""
65
+
66
+
67
+ def f_tau_spec(cfg: SimConfig) -> dict:
68
+ """Translate (regime, f_family, f_params) into a sample_taus spec."""
69
+ c = cfg.center
70
+ p = dict(cfg.f_params)
71
+ if cfg.regime == "no_transition":
72
+ return {"kind": "degenerate_at", "center": c} # unused
73
+ fam = cfg.f_family
74
+ if fam == "degenerate" or cfg.regime == "synchronous":
75
+ sd = p.get("sigma", 1e-6)
76
+ return {"kind": "normal", "mu": c, "sigma": sd}
77
+ if fam == "narrow_normal":
78
+ return {"kind": "normal", "mu": c, "sigma": p.get("sigma", 0.2)}
79
+ if fam == "broad_normal":
80
+ return {"kind": "normal", "mu": c, "sigma": p.get("sigma", 0.9)}
81
+ if fam == "uniform":
82
+ lo, hi = p.get("lo", c - 2.0), p.get("hi", c + 2.0)
83
+ return {"kind": "uniform", "lo": lo, "hi": hi}
84
+ if fam == "skewed":
85
+ return {"kind": "gamma", "shape": p.get("shape", 3.0),
86
+ "scale": p.get("scale", 0.5), "shift": c - 1.5}
87
+ if fam == "two_normal_mix":
88
+ d = p.get("delta", 2.5)
89
+ return {"kind": "mixture", "pi": p.get("pi", 0.5), "mu1": c - d / 2,
90
+ "mu2": c + d / 2, "sigma": p.get("sigma", 0.4)}
91
+ if fam == "three_normal_mix":
92
+ d = p.get("delta", 2.0)
93
+ return {"kind": "three_mix", "mu1": c - d, "mu2": c, "mu3": c + d,
94
+ "sigma": p.get("sigma", 0.35),
95
+ "pis": p.get("pis", [0.33, 0.34, 0.33])}
96
+ if fam == "spike_diffuse":
97
+ return {"kind": "spike", "center": c, "eps": p.get("eps", 0.4),
98
+ "lo": c - 2.5, "hi": c + 2.5}
99
+ if fam == "heavy_tail":
100
+ return {"kind": "student_t", "df": p.get("df", 3.0), "mu": c,
101
+ "sigma": p.get("sigma", 0.5)}
102
+ raise ValueError(f"unknown f_family {fam!r}")
103
+
104
+
105
+ def sample_taus_cfg(cfg: SimConfig, rng: np.random.Generator) -> np.ndarray:
106
+ spec = f_tau_spec(cfg)
107
+ kind = spec["kind"]
108
+ if kind == "degenerate_at":
109
+ return np.full(cfg.n_units, cfg.center)
110
+ if kind == "three_mix":
111
+ pis = np.asarray(spec["pis"], dtype=float)
112
+ comp = rng.choice(3, cfg.n_units, p=pis / pis.sum())
113
+ mus = np.array([spec["mu1"], spec["mu2"], spec["mu3"]])
114
+ return mus[comp] + rng.normal(0, spec["sigma"], cfg.n_units)
115
+ if kind == "student_t":
116
+ return spec["mu"] + spec["sigma"] * t_dist.rvs(spec["df"], size=cfg.n_units, random_state=rng)
117
+ return sample_taus(cfg.n_units, spec, rng)
118
+
119
+
120
+ def regime_truth(cfg: SimConfig) -> str:
121
+ if cfg.regime in REGIMES:
122
+ return cfg.regime
123
+ return cfg.regime
124
+
125
+
126
+ def make_weights(cfg: SimConfig, taus: np.ndarray, t: np.ndarray,
127
+ rng: np.random.Generator) -> np.ndarray:
128
+ """(n_units, n_t) observation-independent weight matrix."""
129
+ n = taus.size
130
+ kind = cfg.weight_kind
131
+ p = cfg.weight_params
132
+ if kind == "equal":
133
+ return np.ones((n, t.size))
134
+ if kind == "static":
135
+ w = rng.lognormal(0, p.get("sd", 0.8), n)
136
+ return np.tile(w / w.mean(), (t.size, 1)).T
137
+ if kind == "timing_corr":
138
+ kappa = p.get("kappa", 1.0)
139
+ w = np.exp(-kappa * (taus - taus.mean()) / max(taus.std(), 1e-9))
140
+ return np.tile(w / w.mean(), (t.size, 1)).T
141
+ if kind == "timevarying":
142
+ base = rng.lognormal(0, p.get("sd", 0.5), n)
143
+ ramp = np.exp(p.get("slope", 0.3) * (t[None, :] - t.mean()) / np.ptp(t) * taus[:, None])
144
+ w = base[:, None] * ramp
145
+ return w / w.mean(axis=0, keepdims=True)
146
+ raise ValueError(f"unknown weight_kind {kind!r}")
147
+
148
+
149
+ def make_windows(cfg: SimConfig, taus: np.ndarray, t: np.ndarray,
150
+ rng: np.random.Generator) -> np.ndarray:
151
+ """(n_units, 2) [L,U] windows; NaN marks unobserved in simulate."""
152
+ n = taus.size
153
+ lo, hi = t.min(), t.max()
154
+ win = np.tile([lo, hi], (n, 1)).astype(float)
155
+ kind = cfg.miss_kind
156
+ p = cfg.miss_params
157
+ if kind == "complete":
158
+ return win
159
+ if kind == "mcar":
160
+ rate = p.get("rate", 0.1)
161
+ # handled as point-wise missing in simulate()
162
+ return win
163
+ if kind == "dropout":
164
+ rate = p.get("rate", 0.3)
165
+ drop = rng.random(n) < rate
166
+ win[drop, 1] = np.sort(t)[rng.integers(len(t) // 3, len(t) - 1, drop.sum())]
167
+ return win
168
+ if kind == "late_entry":
169
+ rate = p.get("rate", 0.3)
170
+ late = rng.random(n) < rate
171
+ win[late, 0] = np.sort(t)[rng.integers(1, len(t) // 2, late.sum())]
172
+ return win
173
+ if kind == "informative":
174
+ kappa = p.get("kappa", 1.0)
175
+ # entry time correlated with tau_i: later-transitioning units enter later
176
+ win[:, 0] = np.clip(taus - p.get("lead", 1.0) + rng.normal(0, 0.3, n), lo, hi)
177
+ return win
178
+ raise ValueError(f"unknown miss_kind {kind!r}")
179
+
180
+
181
+ def add_trend(cfg: SimConfig, values: np.ndarray, t: np.ndarray,
182
+ rng: np.random.Generator) -> np.ndarray:
183
+ """Add b_i(t) heterogeneity to panel values (in-place safe copy)."""
184
+ v = values.copy()
185
+ n = v.shape[0]
186
+ s = cfg.trend_scale
187
+ if cfg.trend_kind == "none" or s == 0:
188
+ return v
189
+ tc = (t - t.mean()) / np.ptp(t)
190
+ if cfg.trend_kind == "linear":
191
+ slopes = rng.normal(0, s, n)
192
+ v += slopes[:, None] * tc[None, :]
193
+ elif cfg.trend_kind == "nonlinear":
194
+ for i in range(n):
195
+ knots = rng.normal(0, s, 3)
196
+ v[i] += knots[0] * tc + knots[1] * tc**2 * 4 + knots[2] * np.sin(2 * np.pi * tc)
197
+ elif cfg.trend_kind == "local_drift":
198
+ for i in range(n):
199
+ onset = rng.uniform(t.min(), t.max())
200
+ drift = np.where(t >= onset, rng.normal(0, s), 0.0)
201
+ v[i] += drift * (t - onset)
202
+ elif cfg.trend_kind == "mixed":
203
+ slopes = rng.normal(0, s, n)
204
+ wobble = rng.normal(0, s * 0.5, n)[:, None] * np.sin(4 * np.pi * tc)[None, :]
205
+ v += slopes[:, None] * tc[None, :] + wobble
206
+ else:
207
+ raise ValueError(f"unknown trend_kind {cfg.trend_kind!r}")
208
+ return v
209
+
210
+
211
+ def add_noise(cfg: SimConfig, values: np.ndarray, t: np.ndarray,
212
+ rng: np.random.Generator) -> np.ndarray:
213
+ n, nt = values.shape
214
+ sd = cfg.noise_sd
215
+ p = cfg.noise_params
216
+ if sd == 0:
217
+ return values
218
+ if cfg.noise_kind == "iid":
219
+ e = rng.normal(0, sd, (n, nt))
220
+ elif cfg.noise_kind == "t3":
221
+ e = sd * t_dist.rvs(3, size=(n, nt), random_state=rng)
222
+ elif cfg.noise_kind == "heterosk":
223
+ e = rng.normal(0, sd * (0.5 + t[None, :] / np.ptp(t)), (n, nt))
224
+ elif cfg.noise_kind == "ar1":
225
+ rho = p.get("rho", 0.7)
226
+ e = np.zeros((n, nt))
227
+ eps = rng.normal(0, sd, (n, nt))
228
+ e[:, 0] = eps[:, 0]
229
+ for k in range(1, nt):
230
+ e[:, k] = rho * e[:, k - 1] + eps[:, k]
231
+ elif cfg.noise_kind == "seasonal":
232
+ e = rng.normal(0, sd, (n, nt)) + sd * 0.8 * np.sin(2 * np.pi * p.get("period", 0.4) * t[None, :])
233
+ else:
234
+ raise ValueError(cfg.noise_kind)
235
+ if cfg.noisevar_hetero > 0:
236
+ e = e * rng.lognormal(0, cfg.noisevar_hetero, n)[:, None]
237
+ return values + e
238
+
239
+
240
+ def simulate(cfg: SimConfig) -> tuple[Panel, dict]:
241
+ """Full cell simulation -> (Panel, truth dict)."""
242
+ rng = np.random.default_rng(cfg.seed)
243
+ t = np.linspace(cfg.t_span[0], cfg.t_span[1], cfg.n_t)
244
+ taus = sample_taus_cfg(cfg, rng)
245
+ n = cfg.n_units
246
+ A = cfg.amplitude * rng.lognormal(0, cfg.amp_hetero, n)
247
+ h = cfg.width * rng.lognormal(0, cfg.width_hetero, n)
248
+ a = rng.normal(0, cfg.base_hetero, n)
249
+ if cfg.regime == "no_transition":
250
+ A = np.zeros(n)
251
+ u = (t[None, :] - taus[:, None]) / h[:, None]
252
+ values = a[:, None] + A[:, None] * g(u, cfg.shape)
253
+ values = add_trend(cfg, values, t, rng)
254
+ values = add_noise(cfg, values, t, rng)
255
+ win = make_windows(cfg, taus, t, rng)
256
+ obs = (t[None, :] >= win[:, 0:1]) & (t[None, :] <= win[:, 1:2])
257
+ values = np.where(obs, values, np.nan)
258
+ if cfg.miss_kind == "mcar":
259
+ drop = rng.random(values.shape) < cfg.miss_params.get("rate", 0.1)
260
+ values = np.where(drop, np.nan, values)
261
+ w = make_weights(cfg, taus, t, rng)
262
+ w_t = np.where(np.isfinite(values), w, np.nan)
263
+ agg = np.nansum(values * w_t, axis=0) / np.maximum(np.nansum(w_t, axis=0), 1e-12)
264
+ panel = Panel(
265
+ t=t, values=values, taus=taus, weights=w.mean(axis=1), windows=win,
266
+ shape=cfg.shape, amplitudes=A,
267
+ meta={"baseline": a, "width": h, "config": cfg},
268
+ )
269
+ truth = {
270
+ "regime": regime_truth(cfg),
271
+ "tau_sd": float(np.std(taus)),
272
+ "tau_iqr": float(np.quantile(taus, 0.75) - np.quantile(taus, 0.25)),
273
+ "taus": taus,
274
+ "aggregate": agg,
275
+ "cell_id": cfg.cell_id,
276
+ }
277
+ return panel, truth
@@ -0,0 +1,110 @@
1
+ """Timing simulation and panel generation (model ladder S1-S7)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+
7
+ import numpy as np
8
+ from scipy.stats import gamma as gamma_dist
9
+ from scipy.stats import norm
10
+
11
+ from .transitions import g
12
+
13
+
14
+ @dataclass
15
+ class Panel:
16
+ t: np.ndarray
17
+ values: np.ndarray # (n_units, n_t)
18
+ taus: np.ndarray
19
+ weights: np.ndarray # (n_units,) applied uniformly over t
20
+ windows: np.ndarray # (n_units, 2) [L_i, U_i]
21
+ shape: str
22
+ amplitudes: np.ndarray
23
+ meta: dict = field(default_factory=dict)
24
+
25
+
26
+ def sample_taus(n: int, spec: dict, rng: np.random.Generator) -> np.ndarray:
27
+ """Sample transition times from a timing spec.
28
+
29
+ kinds: normal {mu,sigma}, mixture {pi,mu1,mu2,sigma1,sigma2},
30
+ gamma {shape,scale,shift}, uniform {lo,hi}, spike {center,eps,lo,hi}.
31
+ """
32
+ kind = spec.get("kind", "normal")
33
+ if kind == "normal":
34
+ return rng.normal(spec["mu"], spec["sigma"], n)
35
+ if kind == "mixture":
36
+ pick = rng.random(n) < spec.get("pi", 0.5)
37
+ out = np.empty(n)
38
+ out[pick] = rng.normal(spec["mu1"], spec.get("sigma1", spec.get("sigma", 0.3)), pick.sum())
39
+ out[~pick] = rng.normal(
40
+ spec["mu2"], spec.get("sigma2", spec.get("sigma", 0.3)), (~pick).sum()
41
+ )
42
+ return out
43
+ if kind == "gamma":
44
+ return spec.get("shift", 0.0) + gamma_dist.rvs(
45
+ spec["shape"], scale=spec["scale"], size=n, random_state=rng
46
+ )
47
+ if kind == "uniform":
48
+ return rng.uniform(spec["lo"], spec["hi"], n)
49
+ if kind == "spike":
50
+ eps = spec.get("eps", 0.2)
51
+ out = np.where(rng.random(n) < 1 - eps, spec["center"], np.nan)
52
+ mask = np.isnan(out)
53
+ out[mask] = rng.uniform(spec["lo"], spec["hi"], mask.sum())
54
+ return out
55
+ raise ValueError(f"unknown tau spec kind {kind!r}")
56
+
57
+
58
+ def simulate_panel(
59
+ t: np.ndarray,
60
+ taus: np.ndarray,
61
+ shape: str = "logistic",
62
+ amplitude: float | np.ndarray = 1.0,
63
+ baseline: float | np.ndarray = 0.0,
64
+ width: float | np.ndarray = 0.3,
65
+ noise_sd: float = 0.0,
66
+ weights: np.ndarray | None = None,
67
+ windows: np.ndarray | None = None,
68
+ rng: np.random.Generator | None = None,
69
+ ) -> Panel:
70
+ """S2-style homogeneous-shape panel: Y_i(t) = a_i + A_i g((t-tau_i)/h_i) + eps."""
71
+ rng = rng or np.random.default_rng()
72
+ t = np.asarray(t, dtype=float)
73
+ taus = np.asarray(taus, dtype=float)
74
+ n = taus.size
75
+ A = np.broadcast_to(np.asarray(amplitude, dtype=float), (n,))
76
+ a = np.broadcast_to(np.asarray(baseline, dtype=float), (n,))
77
+ h = np.broadcast_to(np.asarray(width, dtype=float), (n,))
78
+ u = (t[None, :] - taus[:, None]) / h[:, None]
79
+ values = a[:, None] + A[:, None] * g(u, shape)
80
+ if noise_sd > 0:
81
+ values = values + rng.normal(0, noise_sd, values.shape)
82
+ w = np.ones(n) if weights is None else np.asarray(weights, dtype=float)
83
+ win = (
84
+ np.column_stack([np.full(n, t.min()), np.full(n, t.max())])
85
+ if windows is None
86
+ else np.asarray(windows, dtype=float)
87
+ )
88
+ obs = (t[None, :] >= win[:, 0:1]) & (t[None, :] <= win[:, 1:2])
89
+ values = np.where(obs, values, np.nan)
90
+ return Panel(t=t, values=values, taus=taus, weights=w, windows=win, shape=shape,
91
+ amplitudes=A, meta={"baseline": a, "width": h, "noise_sd": noise_sd})
92
+
93
+
94
+ def timing_density(tau_grid: np.ndarray, spec: dict) -> np.ndarray:
95
+ """Population density f_tau on a grid for closed-form checks."""
96
+ kind = spec.get("kind", "normal")
97
+ tg = np.asarray(tau_grid, dtype=float)
98
+ if kind == "normal":
99
+ return norm.pdf(tg, spec["mu"], spec["sigma"])
100
+ if kind == "mixture":
101
+ pi = spec.get("pi", 0.5)
102
+ s1 = spec.get("sigma1", spec.get("sigma", 0.3))
103
+ s2 = spec.get("sigma2", spec.get("sigma", 0.3))
104
+ return pi * norm.pdf(tg, spec["mu1"], s1) + (1 - pi) * norm.pdf(tg, spec["mu2"], s2)
105
+ if kind == "uniform":
106
+ return np.where((tg >= spec["lo"]) & (tg <= spec["hi"]),
107
+ 1.0 / (spec["hi"] - spec["lo"]), 0.0)
108
+ if kind == "gamma":
109
+ return gamma_dist.pdf(tg - spec.get("shift", 0.0), spec["shape"], scale=spec["scale"])
110
+ raise ValueError(f"no analytic density for kind {kind!r}")
@@ -0,0 +1,56 @@
1
+ """Unit transition shapes g(u), u = (t - tau)/h, in [0, 1].
2
+
3
+ All supported shapes satisfy the antisymmetry identity g(u) + g(-u) = 1
4
+ required by proposition P2, except `linear` which satisfies it on the
5
+ saturated extension used here.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import numpy as np
11
+ from scipy.stats import norm
12
+
13
+ SHAPES = ("step", "logistic", "probit", "linear")
14
+
15
+
16
+ def g(u: np.ndarray, shape: str = "logistic") -> np.ndarray:
17
+ """Transition function values. `u` is already standardized."""
18
+ u = np.asarray(u, dtype=float)
19
+ if shape == "step":
20
+ out = (u >= 0).astype(float)
21
+ out[u == 0] = 0.5 # antisymmetric convention; measure-zero for integrals
22
+ return out
23
+ if shape == "logistic":
24
+ return 1.0 / (1.0 + np.exp(-u))
25
+ if shape == "probit":
26
+ return norm.cdf(u)
27
+ if shape == "linear":
28
+ return np.clip(0.5 + u / 2.0, 0.0, 1.0)
29
+ raise ValueError(f"unknown transition shape {shape!r}; choose from {SHAPES}")
30
+
31
+
32
+ def g_prime(u: np.ndarray, shape: str = "logistic") -> np.ndarray:
33
+ """Derivative of the transition (density kernel used by P8 ill-posedness).
34
+
35
+ For `step` this is a Dirac delta conceptually; returns zeros except at the
36
+ discontinuity, where callers should use the P1 identity directly.
37
+ """
38
+ u = np.asarray(u, dtype=float)
39
+ if shape == "step":
40
+ out = np.full(u.shape, np.inf)
41
+ out[u != 0] = 0.0
42
+ return out
43
+ if shape == "logistic":
44
+ s = g(u, "logistic")
45
+ return s * (1.0 - s)
46
+ if shape == "probit":
47
+ return norm.pdf(u)
48
+ if shape == "linear":
49
+ return np.where(np.abs(u) < 1.0, 0.5, 0.0)
50
+ raise ValueError(f"unknown transition shape {shape!r}")
51
+
52
+
53
+ def is_antisymmetric(shape: str) -> bool:
54
+ """Whether g(u) + g(-u) = 1 holds for this shape (P2 condition)."""
55
+ u = np.linspace(-8, 8, 401)
56
+ return bool(np.allclose(g(u, shape) + g(-u, shape), 1.0, atol=1e-12))
@@ -0,0 +1,223 @@
1
+ Metadata-Version: 2.4
2
+ Name: falsesync
3
+ Version: 0.1.2
4
+ Summary: Aggregation-induced false synchrony: estimands, diagnostics, and simulation for change-point analysis on aggregated panels
5
+ Author: Tatsuki Onishi
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/bougtoir/falsesync
8
+ Project-URL: Source, https://github.com/bougtoir/falsesync
9
+ Project-URL: Issues, https://github.com/bougtoir/falsesync/issues
10
+ Project-URL: Changelog, https://github.com/bougtoir/falsesync/blob/main/CHANGELOG.md
11
+ Project-URL: DOI, https://doi.org/10.5281/zenodo.23233536
12
+ Keywords: change-point analysis,aggregation,synchrony,panel time series,calibration
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: numpy<2.4,>=2.0
26
+ Requires-Dist: scipy>=1.11
27
+ Requires-Dist: pandas>=2.0
28
+ Requires-Dist: statsmodels>=0.14
29
+ Requires-Dist: matplotlib>=3.8
30
+ Requires-Dist: joblib>=1.3
31
+ Requires-Dist: pyyaml>=6
32
+ Requires-Dist: scikit-learn<1.8,>=1.5
33
+ Provides-Extra: dev
34
+ Requires-Dist: pytest>=8; extra == "dev"
35
+ Requires-Dist: ruff>=0.5; extra == "dev"
36
+ Requires-Dist: black>=24; extra == "dev"
37
+ Provides-Extra: cp
38
+ Requires-Dist: ruptures>=1.1; extra == "cp"
39
+ Dynamic: license-file
40
+
41
+ # falsesync
42
+
43
+ Diagnostics for aggregation-induced false synchrony in change-point analysis.
44
+
45
+ ## Scientific motivation
46
+
47
+ Many engineering and observational studies average a panel of unit-level
48
+ series (turbines, battery cells, sensors, regions) and then fit a change point
49
+ to the aggregate. A sharp aggregate breakpoint is easily read as evidence that
50
+ the units changed together. That reading is not justified in general: when
51
+ unit transitions occur at heterogeneous times \(\tau_i \sim F_\tau\), the
52
+ aggregate mean is a smoothed curve \(m(t) = a + A\,(g * F_\tau)(t)\) that can
53
+ still produce a well-defined, strong breakpoint. The fitted aggregate
54
+ breakpoint is a functional \(T_M(F_\tau, g, w, \text{window}, \text{noise})\)
55
+ that depends on the breakpoint operator \(M\), the observation window, the
56
+ weights, the observation process, and the noise, and it need not coincide with
57
+ the mean, median, or mode of the timing distribution.
58
+
59
+ ## What falsesync does
60
+
61
+ - simulates unit-level transition panels with configurable timing laws,
62
+ transition shapes, weights, observation windows, trends, and noise;
63
+ - computes weighted aggregates and aggregate breakpoints under several
64
+ operators (least squares, maximum slope, CUSUM, binary segmentation) and
65
+ reports the spread across operators;
66
+ - estimates unit-level breakpoints with bootstrap uncertainty and a
67
+ floor-corrected timing dispersion
68
+ \(\mathrm{sd}_{\mathrm{corr}} = \sqrt{\max(\mathrm{sd}^2(\hat\tau_i) - \bar s^2, 0)}\);
69
+ - returns calibrated probabilities for five timing regimes
70
+ (`SYNCHRONOUS`, `NEAR_SYNCHRONOUS`, `DIFFUSE_ASYNCHRONOUS`, `CLUSTERED`,
71
+ `NO_TRANSITION`) from a classifier calibrated on a separate simulation grid;
72
+ - issues a false common-event warning when the aggregate break is strong but
73
+ the calibrated probability of synchrony is low, plus warnings for trend
74
+ confounding, operator dependence, and observation-process risks.
75
+
76
+ ## What falsesync does not claim
77
+
78
+ - It is not a causal method and does not identify what caused any transition.
79
+ - A low synchrony probability is a warning that the aggregate break should not
80
+ be read as a common event; it is not a test that rejects synchrony.
81
+ - The regime probabilities are calibrated only for data-generating conditions
82
+ resembling the calibration grid (see `simulations/configs/`).
83
+
84
+ ## Known limitations
85
+
86
+ - Informative observation processes (entry or dropout related to transition
87
+ timing) can create apparent synchrony; this failure mode is detectable in
88
+ simulation but not corrected by the package.
89
+ - Trend heterogeneity across units can be confounded with timing
90
+ heterogeneity.
91
+ - Unit-level break uncertainty inflates apparent timing dispersion; the
92
+ floor correction reduces but does not remove this effect.
93
+ - The aggregate breakpoint location is operator-dependent; different
94
+ operators can give materially different locations on the same aggregate.
95
+ - Without unit-level data, synchrony cannot be certified from the aggregate
96
+ alone.
97
+
98
+ ## Installation
99
+
100
+ Python >= 3.10.
101
+
102
+ ```bash
103
+ git clone https://github.com/bougtoir/falsesync.git
104
+ cd falsesync
105
+ pip install . # or: pip install -e ".[dev]" for tests
106
+ ```
107
+
108
+ Optional: `pip install ".[cp]"` adds `ruptures`-backed detectors. The
109
+ reported binary-segmentation results were computed with `ruptures` 1.1.10, so
110
+ FULL replication requires `pip install ".[dev,cp]"`.
111
+
112
+ ## Minimal example
113
+
114
+ ```bash
115
+ python examples/minimal_example.py
116
+ ```
117
+
118
+ The example simulates 80 units with diffuse (normal, SD 1.1) transition times,
119
+ fits `FalseSynchronyModel` with the shipped calibrated classifier, and prints
120
+ the aggregate breakpoint, the corrected timing dispersion, the five regime
121
+ probabilities, and the warnings. In outline:
122
+
123
+ ```python
124
+ import joblib
125
+ import numpy as np
126
+ from falsesync import simulation
127
+ from falsesync.model import FalseSynchronyModel
128
+
129
+ rng = np.random.default_rng(1)
130
+ t = np.linspace(0, 10, 180)
131
+ taus = simulation.sample_taus(80, {"kind": "normal", "mu": 5.0, "sigma": 1.1}, rng)
132
+ panel = simulation.simulate_panel(t, taus, shape="logistic", amplitude=1.5,
133
+ width=0.3, noise_sd=0.2, rng=rng)
134
+ clf = joblib.load("simulations/calibration/classifier.joblib")
135
+ res = FalseSynchronyModel(classifier=clf).fit(t, panel.values)
136
+ print(res.aggregate_breakpoint, res.timing_dispersion["sd_corrected"])
137
+ print(res.regime_probabilities, res.interpretation_warning)
138
+ ```
139
+
140
+ ## Main workflow
141
+
142
+ 1. `simulations/run_grid.py` simulates a configuration grid
143
+ (`simulations/configs/*.yaml`) and extracts diagnostic features.
144
+ 2. `simulations/run_calibration.py` fits and temperature-calibrates the
145
+ regime classifier on the calibration grid
146
+ (`simulations/calibration/classifier.joblib`).
147
+ 3. `simulations/run_evaluation.py` evaluates the frozen classifier on the
148
+ locked evaluation grid.
149
+ 4. `simulations/run_stress.py`, `run_missingness_mitigation.py`, and
150
+ `run_near_sync.py` run the robustness and stress conditions.
151
+ 5. `simulations/run_kelmarsh.py` and `simulations/run_nasa_battery.py` run the
152
+ two empirical demonstrations.
153
+ 6. `simulations/make_phase3_figures.py` builds Figures 1-6.
154
+
155
+ ## Reproducing the Technometrics results
156
+
157
+ | Mode | Command | Content | Runtime (1 CPU core) |
158
+ |-------|----------------------------------|-------------------------------------------------------------------------|----------------------|
159
+ | QUICK | `python replication/run_quick.py` | tests, minimal example, smoke grid, re-evaluation of the locked grid with the shipped classifier, regeneration of Figures 1-4, comparison with reference tables and figures | < 1 min |
160
+ | FULL | `python replication/run_full.py` | data download and checksum verification, calibration, locked evaluation, stress tests, both empirical analyses, all figures, comparison with reference tables; needs the `cp` extra | about 15 min |
161
+
162
+ Both scripts exit non-zero if any regenerated result table differs from the
163
+ shipped reference copy. `make all` / `make quick` run the same steps.
164
+
165
+ The reference figures were rendered with matplotlib 3.10. matplotlib 3.11
166
+ reproduces every result table exactly but renders Figures 1-4 with small
167
+ pixel differences, which the QUICK figure comparison reports as `DIFFERS`;
168
+ use `pip install "matplotlib<3.11"` for a pixel-level figure match.
169
+
170
+ **Locked evaluation.** `simulations/configs/evaluation_locked.yaml` (seeds
171
+ 3000-3999) was fixed and committed before the evaluation was run and was not
172
+ edited afterwards; its SHA-256 is recorded in `simulations/configs/SHA256SUMS`.
173
+ The calibration grid (`calibration.yaml`) and the development grid
174
+ (`development.yaml`) use disjoint seed ranges. `METHOD_FREEZE.md` lists the
175
+ frozen thresholds and method choices.
176
+
177
+ **Traceability.** `manuscript_number_trace.csv` maps every number reported in
178
+ the manuscript to the result file and column that produces it;
179
+ `figure_manifest.csv` and `table_manifest.csv` map figures and tables to their
180
+ generating scripts and inputs.
181
+
182
+ ## Data acquisition
183
+
184
+ No raw data are redistributed in this repository. `python fetch_data.py`
185
+ downloads the two public datasets from their original repositories, verifies
186
+ each file against the SHA-256 recorded in `data/acquisition_ledger.csv`, and
187
+ extracts the files used by the analyses into `data/raw/` (git-ignored). See
188
+ `data/README.md` for sources, licenses, and citations.
189
+
190
+ ## Repository structure
191
+
192
+ ```text
193
+ src/falsesync/ package source
194
+ tests/ unit tests (theory identities, workflow)
195
+ examples/ minimal example
196
+ simulations/ simulation, calibration, evaluation, empirical scripts
197
+ configs/ development, calibration, locked evaluation grids
198
+ calibration/ calibrated classifier artifact
199
+ results/ simulated grid outputs
200
+ replication/ QUICK and FULL replication entry points
201
+ proofs/ proofs of propositions P1-P8
202
+ outputs/ reference figures
203
+ data/ acquisition ledger and data documentation
204
+ *.csv reference result tables
205
+ math_specification.md, diagnostic_specification.md, METHOD_FREEZE.md,
206
+ counterexamples.md method documentation
207
+ ```
208
+
209
+ ## Citation
210
+
211
+ See `CITATION.cff`. Please cite the software and the accompanying manuscript:
212
+ T. Onishi, "Aggregation-Induced False Synchrony in Change-Point Analysis"
213
+ (manuscript submitted to *Technometrics*).
214
+
215
+ ## License
216
+
217
+ MIT (see `LICENSE`). Data obtained through `fetch_data.py` remain under their
218
+ original licenses (see `data/README.md`).
219
+
220
+ ## Manuscript status
221
+
222
+ Version 0.1.2 corresponds to the manuscript as submitted to *Technometrics*.
223
+ The manuscript has not been peer reviewed or accepted.