falsesync 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- falsesync/__init__.py +9 -0
- falsesync/aggregation.py +85 -0
- falsesync/alignment.py +37 -0
- falsesync/changepoints.py +146 -0
- falsesync/datasets.py +68 -0
- falsesync/diagnostics.py +265 -0
- falsesync/features.py +164 -0
- falsesync/inference.py +90 -0
- falsesync/model.py +221 -0
- falsesync/plotting.py +59 -0
- falsesync/regimes.py +124 -0
- falsesync/simengine.py +277 -0
- falsesync/simulation.py +110 -0
- falsesync/transitions.py +56 -0
- falsesync-0.1.2.dist-info/METADATA +223 -0
- falsesync-0.1.2.dist-info/RECORD +19 -0
- falsesync-0.1.2.dist-info/WHEEL +5 -0
- falsesync-0.1.2.dist-info/licenses/LICENSE +21 -0
- falsesync-0.1.2.dist-info/top_level.txt +1 -0
falsesync/simengine.py
ADDED
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
"""Phase-3 config-driven simulation engine (see simulation_design.md).
|
|
2
|
+
|
|
3
|
+
A SimConfig fully specifies one simulation cell; simulate() returns a Panel
|
|
4
|
+
plus ground-truth regime/timing metadata. All randomness flows through one
|
|
5
|
+
seeded numpy Generator.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
from scipy.stats import t as t_dist
|
|
14
|
+
|
|
15
|
+
from .aggregation import weighted_aggregate
|
|
16
|
+
from .simulation import Panel, sample_taus, simulate_panel
|
|
17
|
+
from .transitions import g
|
|
18
|
+
|
|
19
|
+
REGIMES = (
|
|
20
|
+
"synchronous",
|
|
21
|
+
"near_synchronous",
|
|
22
|
+
"diffuse",
|
|
23
|
+
"two_cluster",
|
|
24
|
+
"three_cluster",
|
|
25
|
+
"no_transition",
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
F_FAMILIES = (
|
|
29
|
+
"degenerate", "narrow_normal", "broad_normal", "uniform", "skewed",
|
|
30
|
+
"two_normal_mix", "three_normal_mix", "spike_diffuse", "heavy_tail",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
NOISE_KINDS = ("iid", "t3", "heterosk", "ar1", "seasonal")
|
|
34
|
+
WEIGHT_KINDS = ("equal", "static", "timing_corr", "timevarying")
|
|
35
|
+
MISS_KINDS = ("complete", "mcar", "dropout", "late_entry", "informative")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class SimConfig:
|
|
40
|
+
n_units: int = 100
|
|
41
|
+
n_t: int = 120
|
|
42
|
+
t_span: tuple = (0.0, 10.0)
|
|
43
|
+
regime: str = "diffuse"
|
|
44
|
+
f_family: str = "broad_normal"
|
|
45
|
+
f_params: dict = field(default_factory=dict) # overrides, e.g. {"sigma":0.8}
|
|
46
|
+
center: float = 5.0 # central transition time
|
|
47
|
+
shape: str = "logistic" # step/logistic/probit/linear
|
|
48
|
+
amplitude: float = 1.0 # effect size (low 0.5, med 1.0, high 2.0 typical)
|
|
49
|
+
width: float = 0.4 # transition width h
|
|
50
|
+
noise_kind: str = "iid"
|
|
51
|
+
noise_sd: float = 0.15
|
|
52
|
+
noise_params: dict = field(default_factory=dict)
|
|
53
|
+
weight_kind: str = "equal"
|
|
54
|
+
weight_params: dict = field(default_factory=dict)
|
|
55
|
+
miss_kind: str = "complete"
|
|
56
|
+
miss_params: dict = field(default_factory=dict)
|
|
57
|
+
amp_hetero: float = 0.0 # sd of A_i / A
|
|
58
|
+
width_hetero: float = 0.0 # sd of h_i / h
|
|
59
|
+
base_hetero: float = 0.0 # sd of a_i
|
|
60
|
+
trend_kind: str = "none" # none/linear/nonlinear/local_drift/mixed
|
|
61
|
+
trend_scale: float = 0.0 # sd of trend slope per unit
|
|
62
|
+
noisevar_hetero: float = 0.0
|
|
63
|
+
seed: int = 0
|
|
64
|
+
cell_id: str = ""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def f_tau_spec(cfg: SimConfig) -> dict:
|
|
68
|
+
"""Translate (regime, f_family, f_params) into a sample_taus spec."""
|
|
69
|
+
c = cfg.center
|
|
70
|
+
p = dict(cfg.f_params)
|
|
71
|
+
if cfg.regime == "no_transition":
|
|
72
|
+
return {"kind": "degenerate_at", "center": c} # unused
|
|
73
|
+
fam = cfg.f_family
|
|
74
|
+
if fam == "degenerate" or cfg.regime == "synchronous":
|
|
75
|
+
sd = p.get("sigma", 1e-6)
|
|
76
|
+
return {"kind": "normal", "mu": c, "sigma": sd}
|
|
77
|
+
if fam == "narrow_normal":
|
|
78
|
+
return {"kind": "normal", "mu": c, "sigma": p.get("sigma", 0.2)}
|
|
79
|
+
if fam == "broad_normal":
|
|
80
|
+
return {"kind": "normal", "mu": c, "sigma": p.get("sigma", 0.9)}
|
|
81
|
+
if fam == "uniform":
|
|
82
|
+
lo, hi = p.get("lo", c - 2.0), p.get("hi", c + 2.0)
|
|
83
|
+
return {"kind": "uniform", "lo": lo, "hi": hi}
|
|
84
|
+
if fam == "skewed":
|
|
85
|
+
return {"kind": "gamma", "shape": p.get("shape", 3.0),
|
|
86
|
+
"scale": p.get("scale", 0.5), "shift": c - 1.5}
|
|
87
|
+
if fam == "two_normal_mix":
|
|
88
|
+
d = p.get("delta", 2.5)
|
|
89
|
+
return {"kind": "mixture", "pi": p.get("pi", 0.5), "mu1": c - d / 2,
|
|
90
|
+
"mu2": c + d / 2, "sigma": p.get("sigma", 0.4)}
|
|
91
|
+
if fam == "three_normal_mix":
|
|
92
|
+
d = p.get("delta", 2.0)
|
|
93
|
+
return {"kind": "three_mix", "mu1": c - d, "mu2": c, "mu3": c + d,
|
|
94
|
+
"sigma": p.get("sigma", 0.35),
|
|
95
|
+
"pis": p.get("pis", [0.33, 0.34, 0.33])}
|
|
96
|
+
if fam == "spike_diffuse":
|
|
97
|
+
return {"kind": "spike", "center": c, "eps": p.get("eps", 0.4),
|
|
98
|
+
"lo": c - 2.5, "hi": c + 2.5}
|
|
99
|
+
if fam == "heavy_tail":
|
|
100
|
+
return {"kind": "student_t", "df": p.get("df", 3.0), "mu": c,
|
|
101
|
+
"sigma": p.get("sigma", 0.5)}
|
|
102
|
+
raise ValueError(f"unknown f_family {fam!r}")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def sample_taus_cfg(cfg: SimConfig, rng: np.random.Generator) -> np.ndarray:
|
|
106
|
+
spec = f_tau_spec(cfg)
|
|
107
|
+
kind = spec["kind"]
|
|
108
|
+
if kind == "degenerate_at":
|
|
109
|
+
return np.full(cfg.n_units, cfg.center)
|
|
110
|
+
if kind == "three_mix":
|
|
111
|
+
pis = np.asarray(spec["pis"], dtype=float)
|
|
112
|
+
comp = rng.choice(3, cfg.n_units, p=pis / pis.sum())
|
|
113
|
+
mus = np.array([spec["mu1"], spec["mu2"], spec["mu3"]])
|
|
114
|
+
return mus[comp] + rng.normal(0, spec["sigma"], cfg.n_units)
|
|
115
|
+
if kind == "student_t":
|
|
116
|
+
return spec["mu"] + spec["sigma"] * t_dist.rvs(spec["df"], size=cfg.n_units, random_state=rng)
|
|
117
|
+
return sample_taus(cfg.n_units, spec, rng)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def regime_truth(cfg: SimConfig) -> str:
|
|
121
|
+
if cfg.regime in REGIMES:
|
|
122
|
+
return cfg.regime
|
|
123
|
+
return cfg.regime
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def make_weights(cfg: SimConfig, taus: np.ndarray, t: np.ndarray,
|
|
127
|
+
rng: np.random.Generator) -> np.ndarray:
|
|
128
|
+
"""(n_units, n_t) observation-independent weight matrix."""
|
|
129
|
+
n = taus.size
|
|
130
|
+
kind = cfg.weight_kind
|
|
131
|
+
p = cfg.weight_params
|
|
132
|
+
if kind == "equal":
|
|
133
|
+
return np.ones((n, t.size))
|
|
134
|
+
if kind == "static":
|
|
135
|
+
w = rng.lognormal(0, p.get("sd", 0.8), n)
|
|
136
|
+
return np.tile(w / w.mean(), (t.size, 1)).T
|
|
137
|
+
if kind == "timing_corr":
|
|
138
|
+
kappa = p.get("kappa", 1.0)
|
|
139
|
+
w = np.exp(-kappa * (taus - taus.mean()) / max(taus.std(), 1e-9))
|
|
140
|
+
return np.tile(w / w.mean(), (t.size, 1)).T
|
|
141
|
+
if kind == "timevarying":
|
|
142
|
+
base = rng.lognormal(0, p.get("sd", 0.5), n)
|
|
143
|
+
ramp = np.exp(p.get("slope", 0.3) * (t[None, :] - t.mean()) / np.ptp(t) * taus[:, None])
|
|
144
|
+
w = base[:, None] * ramp
|
|
145
|
+
return w / w.mean(axis=0, keepdims=True)
|
|
146
|
+
raise ValueError(f"unknown weight_kind {kind!r}")
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def make_windows(cfg: SimConfig, taus: np.ndarray, t: np.ndarray,
|
|
150
|
+
rng: np.random.Generator) -> np.ndarray:
|
|
151
|
+
"""(n_units, 2) [L,U] windows; NaN marks unobserved in simulate."""
|
|
152
|
+
n = taus.size
|
|
153
|
+
lo, hi = t.min(), t.max()
|
|
154
|
+
win = np.tile([lo, hi], (n, 1)).astype(float)
|
|
155
|
+
kind = cfg.miss_kind
|
|
156
|
+
p = cfg.miss_params
|
|
157
|
+
if kind == "complete":
|
|
158
|
+
return win
|
|
159
|
+
if kind == "mcar":
|
|
160
|
+
rate = p.get("rate", 0.1)
|
|
161
|
+
# handled as point-wise missing in simulate()
|
|
162
|
+
return win
|
|
163
|
+
if kind == "dropout":
|
|
164
|
+
rate = p.get("rate", 0.3)
|
|
165
|
+
drop = rng.random(n) < rate
|
|
166
|
+
win[drop, 1] = np.sort(t)[rng.integers(len(t) // 3, len(t) - 1, drop.sum())]
|
|
167
|
+
return win
|
|
168
|
+
if kind == "late_entry":
|
|
169
|
+
rate = p.get("rate", 0.3)
|
|
170
|
+
late = rng.random(n) < rate
|
|
171
|
+
win[late, 0] = np.sort(t)[rng.integers(1, len(t) // 2, late.sum())]
|
|
172
|
+
return win
|
|
173
|
+
if kind == "informative":
|
|
174
|
+
kappa = p.get("kappa", 1.0)
|
|
175
|
+
# entry time correlated with tau_i: later-transitioning units enter later
|
|
176
|
+
win[:, 0] = np.clip(taus - p.get("lead", 1.0) + rng.normal(0, 0.3, n), lo, hi)
|
|
177
|
+
return win
|
|
178
|
+
raise ValueError(f"unknown miss_kind {kind!r}")
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def add_trend(cfg: SimConfig, values: np.ndarray, t: np.ndarray,
|
|
182
|
+
rng: np.random.Generator) -> np.ndarray:
|
|
183
|
+
"""Add b_i(t) heterogeneity to panel values (in-place safe copy)."""
|
|
184
|
+
v = values.copy()
|
|
185
|
+
n = v.shape[0]
|
|
186
|
+
s = cfg.trend_scale
|
|
187
|
+
if cfg.trend_kind == "none" or s == 0:
|
|
188
|
+
return v
|
|
189
|
+
tc = (t - t.mean()) / np.ptp(t)
|
|
190
|
+
if cfg.trend_kind == "linear":
|
|
191
|
+
slopes = rng.normal(0, s, n)
|
|
192
|
+
v += slopes[:, None] * tc[None, :]
|
|
193
|
+
elif cfg.trend_kind == "nonlinear":
|
|
194
|
+
for i in range(n):
|
|
195
|
+
knots = rng.normal(0, s, 3)
|
|
196
|
+
v[i] += knots[0] * tc + knots[1] * tc**2 * 4 + knots[2] * np.sin(2 * np.pi * tc)
|
|
197
|
+
elif cfg.trend_kind == "local_drift":
|
|
198
|
+
for i in range(n):
|
|
199
|
+
onset = rng.uniform(t.min(), t.max())
|
|
200
|
+
drift = np.where(t >= onset, rng.normal(0, s), 0.0)
|
|
201
|
+
v[i] += drift * (t - onset)
|
|
202
|
+
elif cfg.trend_kind == "mixed":
|
|
203
|
+
slopes = rng.normal(0, s, n)
|
|
204
|
+
wobble = rng.normal(0, s * 0.5, n)[:, None] * np.sin(4 * np.pi * tc)[None, :]
|
|
205
|
+
v += slopes[:, None] * tc[None, :] + wobble
|
|
206
|
+
else:
|
|
207
|
+
raise ValueError(f"unknown trend_kind {cfg.trend_kind!r}")
|
|
208
|
+
return v
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def add_noise(cfg: SimConfig, values: np.ndarray, t: np.ndarray,
|
|
212
|
+
rng: np.random.Generator) -> np.ndarray:
|
|
213
|
+
n, nt = values.shape
|
|
214
|
+
sd = cfg.noise_sd
|
|
215
|
+
p = cfg.noise_params
|
|
216
|
+
if sd == 0:
|
|
217
|
+
return values
|
|
218
|
+
if cfg.noise_kind == "iid":
|
|
219
|
+
e = rng.normal(0, sd, (n, nt))
|
|
220
|
+
elif cfg.noise_kind == "t3":
|
|
221
|
+
e = sd * t_dist.rvs(3, size=(n, nt), random_state=rng)
|
|
222
|
+
elif cfg.noise_kind == "heterosk":
|
|
223
|
+
e = rng.normal(0, sd * (0.5 + t[None, :] / np.ptp(t)), (n, nt))
|
|
224
|
+
elif cfg.noise_kind == "ar1":
|
|
225
|
+
rho = p.get("rho", 0.7)
|
|
226
|
+
e = np.zeros((n, nt))
|
|
227
|
+
eps = rng.normal(0, sd, (n, nt))
|
|
228
|
+
e[:, 0] = eps[:, 0]
|
|
229
|
+
for k in range(1, nt):
|
|
230
|
+
e[:, k] = rho * e[:, k - 1] + eps[:, k]
|
|
231
|
+
elif cfg.noise_kind == "seasonal":
|
|
232
|
+
e = rng.normal(0, sd, (n, nt)) + sd * 0.8 * np.sin(2 * np.pi * p.get("period", 0.4) * t[None, :])
|
|
233
|
+
else:
|
|
234
|
+
raise ValueError(cfg.noise_kind)
|
|
235
|
+
if cfg.noisevar_hetero > 0:
|
|
236
|
+
e = e * rng.lognormal(0, cfg.noisevar_hetero, n)[:, None]
|
|
237
|
+
return values + e
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def simulate(cfg: SimConfig) -> tuple[Panel, dict]:
|
|
241
|
+
"""Full cell simulation -> (Panel, truth dict)."""
|
|
242
|
+
rng = np.random.default_rng(cfg.seed)
|
|
243
|
+
t = np.linspace(cfg.t_span[0], cfg.t_span[1], cfg.n_t)
|
|
244
|
+
taus = sample_taus_cfg(cfg, rng)
|
|
245
|
+
n = cfg.n_units
|
|
246
|
+
A = cfg.amplitude * rng.lognormal(0, cfg.amp_hetero, n)
|
|
247
|
+
h = cfg.width * rng.lognormal(0, cfg.width_hetero, n)
|
|
248
|
+
a = rng.normal(0, cfg.base_hetero, n)
|
|
249
|
+
if cfg.regime == "no_transition":
|
|
250
|
+
A = np.zeros(n)
|
|
251
|
+
u = (t[None, :] - taus[:, None]) / h[:, None]
|
|
252
|
+
values = a[:, None] + A[:, None] * g(u, cfg.shape)
|
|
253
|
+
values = add_trend(cfg, values, t, rng)
|
|
254
|
+
values = add_noise(cfg, values, t, rng)
|
|
255
|
+
win = make_windows(cfg, taus, t, rng)
|
|
256
|
+
obs = (t[None, :] >= win[:, 0:1]) & (t[None, :] <= win[:, 1:2])
|
|
257
|
+
values = np.where(obs, values, np.nan)
|
|
258
|
+
if cfg.miss_kind == "mcar":
|
|
259
|
+
drop = rng.random(values.shape) < cfg.miss_params.get("rate", 0.1)
|
|
260
|
+
values = np.where(drop, np.nan, values)
|
|
261
|
+
w = make_weights(cfg, taus, t, rng)
|
|
262
|
+
w_t = np.where(np.isfinite(values), w, np.nan)
|
|
263
|
+
agg = np.nansum(values * w_t, axis=0) / np.maximum(np.nansum(w_t, axis=0), 1e-12)
|
|
264
|
+
panel = Panel(
|
|
265
|
+
t=t, values=values, taus=taus, weights=w.mean(axis=1), windows=win,
|
|
266
|
+
shape=cfg.shape, amplitudes=A,
|
|
267
|
+
meta={"baseline": a, "width": h, "config": cfg},
|
|
268
|
+
)
|
|
269
|
+
truth = {
|
|
270
|
+
"regime": regime_truth(cfg),
|
|
271
|
+
"tau_sd": float(np.std(taus)),
|
|
272
|
+
"tau_iqr": float(np.quantile(taus, 0.75) - np.quantile(taus, 0.25)),
|
|
273
|
+
"taus": taus,
|
|
274
|
+
"aggregate": agg,
|
|
275
|
+
"cell_id": cfg.cell_id,
|
|
276
|
+
}
|
|
277
|
+
return panel, truth
|
falsesync/simulation.py
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Timing simulation and panel generation (model ladder S1-S7)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from scipy.stats import gamma as gamma_dist
|
|
9
|
+
from scipy.stats import norm
|
|
10
|
+
|
|
11
|
+
from .transitions import g
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class Panel:
|
|
16
|
+
t: np.ndarray
|
|
17
|
+
values: np.ndarray # (n_units, n_t)
|
|
18
|
+
taus: np.ndarray
|
|
19
|
+
weights: np.ndarray # (n_units,) applied uniformly over t
|
|
20
|
+
windows: np.ndarray # (n_units, 2) [L_i, U_i]
|
|
21
|
+
shape: str
|
|
22
|
+
amplitudes: np.ndarray
|
|
23
|
+
meta: dict = field(default_factory=dict)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def sample_taus(n: int, spec: dict, rng: np.random.Generator) -> np.ndarray:
|
|
27
|
+
"""Sample transition times from a timing spec.
|
|
28
|
+
|
|
29
|
+
kinds: normal {mu,sigma}, mixture {pi,mu1,mu2,sigma1,sigma2},
|
|
30
|
+
gamma {shape,scale,shift}, uniform {lo,hi}, spike {center,eps,lo,hi}.
|
|
31
|
+
"""
|
|
32
|
+
kind = spec.get("kind", "normal")
|
|
33
|
+
if kind == "normal":
|
|
34
|
+
return rng.normal(spec["mu"], spec["sigma"], n)
|
|
35
|
+
if kind == "mixture":
|
|
36
|
+
pick = rng.random(n) < spec.get("pi", 0.5)
|
|
37
|
+
out = np.empty(n)
|
|
38
|
+
out[pick] = rng.normal(spec["mu1"], spec.get("sigma1", spec.get("sigma", 0.3)), pick.sum())
|
|
39
|
+
out[~pick] = rng.normal(
|
|
40
|
+
spec["mu2"], spec.get("sigma2", spec.get("sigma", 0.3)), (~pick).sum()
|
|
41
|
+
)
|
|
42
|
+
return out
|
|
43
|
+
if kind == "gamma":
|
|
44
|
+
return spec.get("shift", 0.0) + gamma_dist.rvs(
|
|
45
|
+
spec["shape"], scale=spec["scale"], size=n, random_state=rng
|
|
46
|
+
)
|
|
47
|
+
if kind == "uniform":
|
|
48
|
+
return rng.uniform(spec["lo"], spec["hi"], n)
|
|
49
|
+
if kind == "spike":
|
|
50
|
+
eps = spec.get("eps", 0.2)
|
|
51
|
+
out = np.where(rng.random(n) < 1 - eps, spec["center"], np.nan)
|
|
52
|
+
mask = np.isnan(out)
|
|
53
|
+
out[mask] = rng.uniform(spec["lo"], spec["hi"], mask.sum())
|
|
54
|
+
return out
|
|
55
|
+
raise ValueError(f"unknown tau spec kind {kind!r}")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def simulate_panel(
|
|
59
|
+
t: np.ndarray,
|
|
60
|
+
taus: np.ndarray,
|
|
61
|
+
shape: str = "logistic",
|
|
62
|
+
amplitude: float | np.ndarray = 1.0,
|
|
63
|
+
baseline: float | np.ndarray = 0.0,
|
|
64
|
+
width: float | np.ndarray = 0.3,
|
|
65
|
+
noise_sd: float = 0.0,
|
|
66
|
+
weights: np.ndarray | None = None,
|
|
67
|
+
windows: np.ndarray | None = None,
|
|
68
|
+
rng: np.random.Generator | None = None,
|
|
69
|
+
) -> Panel:
|
|
70
|
+
"""S2-style homogeneous-shape panel: Y_i(t) = a_i + A_i g((t-tau_i)/h_i) + eps."""
|
|
71
|
+
rng = rng or np.random.default_rng()
|
|
72
|
+
t = np.asarray(t, dtype=float)
|
|
73
|
+
taus = np.asarray(taus, dtype=float)
|
|
74
|
+
n = taus.size
|
|
75
|
+
A = np.broadcast_to(np.asarray(amplitude, dtype=float), (n,))
|
|
76
|
+
a = np.broadcast_to(np.asarray(baseline, dtype=float), (n,))
|
|
77
|
+
h = np.broadcast_to(np.asarray(width, dtype=float), (n,))
|
|
78
|
+
u = (t[None, :] - taus[:, None]) / h[:, None]
|
|
79
|
+
values = a[:, None] + A[:, None] * g(u, shape)
|
|
80
|
+
if noise_sd > 0:
|
|
81
|
+
values = values + rng.normal(0, noise_sd, values.shape)
|
|
82
|
+
w = np.ones(n) if weights is None else np.asarray(weights, dtype=float)
|
|
83
|
+
win = (
|
|
84
|
+
np.column_stack([np.full(n, t.min()), np.full(n, t.max())])
|
|
85
|
+
if windows is None
|
|
86
|
+
else np.asarray(windows, dtype=float)
|
|
87
|
+
)
|
|
88
|
+
obs = (t[None, :] >= win[:, 0:1]) & (t[None, :] <= win[:, 1:2])
|
|
89
|
+
values = np.where(obs, values, np.nan)
|
|
90
|
+
return Panel(t=t, values=values, taus=taus, weights=w, windows=win, shape=shape,
|
|
91
|
+
amplitudes=A, meta={"baseline": a, "width": h, "noise_sd": noise_sd})
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def timing_density(tau_grid: np.ndarray, spec: dict) -> np.ndarray:
|
|
95
|
+
"""Population density f_tau on a grid for closed-form checks."""
|
|
96
|
+
kind = spec.get("kind", "normal")
|
|
97
|
+
tg = np.asarray(tau_grid, dtype=float)
|
|
98
|
+
if kind == "normal":
|
|
99
|
+
return norm.pdf(tg, spec["mu"], spec["sigma"])
|
|
100
|
+
if kind == "mixture":
|
|
101
|
+
pi = spec.get("pi", 0.5)
|
|
102
|
+
s1 = spec.get("sigma1", spec.get("sigma", 0.3))
|
|
103
|
+
s2 = spec.get("sigma2", spec.get("sigma", 0.3))
|
|
104
|
+
return pi * norm.pdf(tg, spec["mu1"], s1) + (1 - pi) * norm.pdf(tg, spec["mu2"], s2)
|
|
105
|
+
if kind == "uniform":
|
|
106
|
+
return np.where((tg >= spec["lo"]) & (tg <= spec["hi"]),
|
|
107
|
+
1.0 / (spec["hi"] - spec["lo"]), 0.0)
|
|
108
|
+
if kind == "gamma":
|
|
109
|
+
return gamma_dist.pdf(tg - spec.get("shift", 0.0), spec["shape"], scale=spec["scale"])
|
|
110
|
+
raise ValueError(f"no analytic density for kind {kind!r}")
|
falsesync/transitions.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Unit transition shapes g(u), u = (t - tau)/h, in [0, 1].
|
|
2
|
+
|
|
3
|
+
All supported shapes satisfy the antisymmetry identity g(u) + g(-u) = 1
|
|
4
|
+
required by proposition P2, except `linear` which satisfies it on the
|
|
5
|
+
saturated extension used here.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
from scipy.stats import norm
|
|
12
|
+
|
|
13
|
+
SHAPES = ("step", "logistic", "probit", "linear")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def g(u: np.ndarray, shape: str = "logistic") -> np.ndarray:
|
|
17
|
+
"""Transition function values. `u` is already standardized."""
|
|
18
|
+
u = np.asarray(u, dtype=float)
|
|
19
|
+
if shape == "step":
|
|
20
|
+
out = (u >= 0).astype(float)
|
|
21
|
+
out[u == 0] = 0.5 # antisymmetric convention; measure-zero for integrals
|
|
22
|
+
return out
|
|
23
|
+
if shape == "logistic":
|
|
24
|
+
return 1.0 / (1.0 + np.exp(-u))
|
|
25
|
+
if shape == "probit":
|
|
26
|
+
return norm.cdf(u)
|
|
27
|
+
if shape == "linear":
|
|
28
|
+
return np.clip(0.5 + u / 2.0, 0.0, 1.0)
|
|
29
|
+
raise ValueError(f"unknown transition shape {shape!r}; choose from {SHAPES}")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def g_prime(u: np.ndarray, shape: str = "logistic") -> np.ndarray:
|
|
33
|
+
"""Derivative of the transition (density kernel used by P8 ill-posedness).
|
|
34
|
+
|
|
35
|
+
For `step` this is a Dirac delta conceptually; returns zeros except at the
|
|
36
|
+
discontinuity, where callers should use the P1 identity directly.
|
|
37
|
+
"""
|
|
38
|
+
u = np.asarray(u, dtype=float)
|
|
39
|
+
if shape == "step":
|
|
40
|
+
out = np.full(u.shape, np.inf)
|
|
41
|
+
out[u != 0] = 0.0
|
|
42
|
+
return out
|
|
43
|
+
if shape == "logistic":
|
|
44
|
+
s = g(u, "logistic")
|
|
45
|
+
return s * (1.0 - s)
|
|
46
|
+
if shape == "probit":
|
|
47
|
+
return norm.pdf(u)
|
|
48
|
+
if shape == "linear":
|
|
49
|
+
return np.where(np.abs(u) < 1.0, 0.5, 0.0)
|
|
50
|
+
raise ValueError(f"unknown transition shape {shape!r}")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def is_antisymmetric(shape: str) -> bool:
|
|
54
|
+
"""Whether g(u) + g(-u) = 1 holds for this shape (P2 condition)."""
|
|
55
|
+
u = np.linspace(-8, 8, 401)
|
|
56
|
+
return bool(np.allclose(g(u, shape) + g(-u, shape), 1.0, atol=1e-12))
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: falsesync
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: Aggregation-induced false synchrony: estimands, diagnostics, and simulation for change-point analysis on aggregated panels
|
|
5
|
+
Author: Tatsuki Onishi
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/bougtoir/falsesync
|
|
8
|
+
Project-URL: Source, https://github.com/bougtoir/falsesync
|
|
9
|
+
Project-URL: Issues, https://github.com/bougtoir/falsesync/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/bougtoir/falsesync/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: DOI, https://doi.org/10.5281/zenodo.23233536
|
|
12
|
+
Keywords: change-point analysis,aggregation,synchrony,panel time series,calibration
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: numpy<2.4,>=2.0
|
|
26
|
+
Requires-Dist: scipy>=1.11
|
|
27
|
+
Requires-Dist: pandas>=2.0
|
|
28
|
+
Requires-Dist: statsmodels>=0.14
|
|
29
|
+
Requires-Dist: matplotlib>=3.8
|
|
30
|
+
Requires-Dist: joblib>=1.3
|
|
31
|
+
Requires-Dist: pyyaml>=6
|
|
32
|
+
Requires-Dist: scikit-learn<1.8,>=1.5
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
35
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
36
|
+
Requires-Dist: black>=24; extra == "dev"
|
|
37
|
+
Provides-Extra: cp
|
|
38
|
+
Requires-Dist: ruptures>=1.1; extra == "cp"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# falsesync
|
|
42
|
+
|
|
43
|
+
Diagnostics for aggregation-induced false synchrony in change-point analysis.
|
|
44
|
+
|
|
45
|
+
## Scientific motivation
|
|
46
|
+
|
|
47
|
+
Many engineering and observational studies average a panel of unit-level
|
|
48
|
+
series (turbines, battery cells, sensors, regions) and then fit a change point
|
|
49
|
+
to the aggregate. A sharp aggregate breakpoint is easily read as evidence that
|
|
50
|
+
the units changed together. That reading is not justified in general: when
|
|
51
|
+
unit transitions occur at heterogeneous times \(\tau_i \sim F_\tau\), the
|
|
52
|
+
aggregate mean is a smoothed curve \(m(t) = a + A\,(g * F_\tau)(t)\) that can
|
|
53
|
+
still produce a well-defined, strong breakpoint. The fitted aggregate
|
|
54
|
+
breakpoint is a functional \(T_M(F_\tau, g, w, \text{window}, \text{noise})\)
|
|
55
|
+
that depends on the breakpoint operator \(M\), the observation window, the
|
|
56
|
+
weights, the observation process, and the noise, and it need not coincide with
|
|
57
|
+
the mean, median, or mode of the timing distribution.
|
|
58
|
+
|
|
59
|
+
## What falsesync does
|
|
60
|
+
|
|
61
|
+
- simulates unit-level transition panels with configurable timing laws,
|
|
62
|
+
transition shapes, weights, observation windows, trends, and noise;
|
|
63
|
+
- computes weighted aggregates and aggregate breakpoints under several
|
|
64
|
+
operators (least squares, maximum slope, CUSUM, binary segmentation) and
|
|
65
|
+
reports the spread across operators;
|
|
66
|
+
- estimates unit-level breakpoints with bootstrap uncertainty and a
|
|
67
|
+
floor-corrected timing dispersion
|
|
68
|
+
\(\mathrm{sd}_{\mathrm{corr}} = \sqrt{\max(\mathrm{sd}^2(\hat\tau_i) - \bar s^2, 0)}\);
|
|
69
|
+
- returns calibrated probabilities for five timing regimes
|
|
70
|
+
(`SYNCHRONOUS`, `NEAR_SYNCHRONOUS`, `DIFFUSE_ASYNCHRONOUS`, `CLUSTERED`,
|
|
71
|
+
`NO_TRANSITION`) from a classifier calibrated on a separate simulation grid;
|
|
72
|
+
- issues a false common-event warning when the aggregate break is strong but
|
|
73
|
+
the calibrated probability of synchrony is low, plus warnings for trend
|
|
74
|
+
confounding, operator dependence, and observation-process risks.
|
|
75
|
+
|
|
76
|
+
## What falsesync does not claim
|
|
77
|
+
|
|
78
|
+
- It is not a causal method and does not identify what caused any transition.
|
|
79
|
+
- A low synchrony probability is a warning that the aggregate break should not
|
|
80
|
+
be read as a common event; it is not a test that rejects synchrony.
|
|
81
|
+
- The regime probabilities are calibrated only for data-generating conditions
|
|
82
|
+
resembling the calibration grid (see `simulations/configs/`).
|
|
83
|
+
|
|
84
|
+
## Known limitations
|
|
85
|
+
|
|
86
|
+
- Informative observation processes (entry or dropout related to transition
|
|
87
|
+
timing) can create apparent synchrony; this failure mode is detectable in
|
|
88
|
+
simulation but not corrected by the package.
|
|
89
|
+
- Trend heterogeneity across units can be confounded with timing
|
|
90
|
+
heterogeneity.
|
|
91
|
+
- Unit-level break uncertainty inflates apparent timing dispersion; the
|
|
92
|
+
floor correction reduces but does not remove this effect.
|
|
93
|
+
- The aggregate breakpoint location is operator-dependent; different
|
|
94
|
+
operators can give materially different locations on the same aggregate.
|
|
95
|
+
- Without unit-level data, synchrony cannot be certified from the aggregate
|
|
96
|
+
alone.
|
|
97
|
+
|
|
98
|
+
## Installation
|
|
99
|
+
|
|
100
|
+
Python >= 3.10.
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
git clone https://github.com/bougtoir/falsesync.git
|
|
104
|
+
cd falsesync
|
|
105
|
+
pip install . # or: pip install -e ".[dev]" for tests
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Optional: `pip install ".[cp]"` adds `ruptures`-backed detectors. The
|
|
109
|
+
reported binary-segmentation results were computed with `ruptures` 1.1.10, so
|
|
110
|
+
FULL replication requires `pip install ".[dev,cp]"`.
|
|
111
|
+
|
|
112
|
+
## Minimal example
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
python examples/minimal_example.py
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
The example simulates 80 units with diffuse (normal, SD 1.1) transition times,
|
|
119
|
+
fits `FalseSynchronyModel` with the shipped calibrated classifier, and prints
|
|
120
|
+
the aggregate breakpoint, the corrected timing dispersion, the five regime
|
|
121
|
+
probabilities, and the warnings. In outline:
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
import joblib
|
|
125
|
+
import numpy as np
|
|
126
|
+
from falsesync import simulation
|
|
127
|
+
from falsesync.model import FalseSynchronyModel
|
|
128
|
+
|
|
129
|
+
rng = np.random.default_rng(1)
|
|
130
|
+
t = np.linspace(0, 10, 180)
|
|
131
|
+
taus = simulation.sample_taus(80, {"kind": "normal", "mu": 5.0, "sigma": 1.1}, rng)
|
|
132
|
+
panel = simulation.simulate_panel(t, taus, shape="logistic", amplitude=1.5,
|
|
133
|
+
width=0.3, noise_sd=0.2, rng=rng)
|
|
134
|
+
clf = joblib.load("simulations/calibration/classifier.joblib")
|
|
135
|
+
res = FalseSynchronyModel(classifier=clf).fit(t, panel.values)
|
|
136
|
+
print(res.aggregate_breakpoint, res.timing_dispersion["sd_corrected"])
|
|
137
|
+
print(res.regime_probabilities, res.interpretation_warning)
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Main workflow
|
|
141
|
+
|
|
142
|
+
1. `simulations/run_grid.py` simulates a configuration grid
|
|
143
|
+
(`simulations/configs/*.yaml`) and extracts diagnostic features.
|
|
144
|
+
2. `simulations/run_calibration.py` fits and temperature-calibrates the
|
|
145
|
+
regime classifier on the calibration grid
|
|
146
|
+
(`simulations/calibration/classifier.joblib`).
|
|
147
|
+
3. `simulations/run_evaluation.py` evaluates the frozen classifier on the
|
|
148
|
+
locked evaluation grid.
|
|
149
|
+
4. `simulations/run_stress.py`, `run_missingness_mitigation.py`, and
|
|
150
|
+
`run_near_sync.py` run the robustness and stress conditions.
|
|
151
|
+
5. `simulations/run_kelmarsh.py` and `simulations/run_nasa_battery.py` run the
|
|
152
|
+
two empirical demonstrations.
|
|
153
|
+
6. `simulations/make_phase3_figures.py` builds Figures 1-6.
|
|
154
|
+
|
|
155
|
+
## Reproducing the Technometrics results
|
|
156
|
+
|
|
157
|
+
| Mode | Command | Content | Runtime (1 CPU core) |
|
|
158
|
+
|-------|----------------------------------|-------------------------------------------------------------------------|----------------------|
|
|
159
|
+
| QUICK | `python replication/run_quick.py` | tests, minimal example, smoke grid, re-evaluation of the locked grid with the shipped classifier, regeneration of Figures 1-4, comparison with reference tables and figures | < 1 min |
|
|
160
|
+
| FULL | `python replication/run_full.py` | data download and checksum verification, calibration, locked evaluation, stress tests, both empirical analyses, all figures, comparison with reference tables; needs the `cp` extra | about 15 min |
|
|
161
|
+
|
|
162
|
+
Both scripts exit non-zero if any regenerated result table differs from the
|
|
163
|
+
shipped reference copy. `make all` / `make quick` run the same steps.
|
|
164
|
+
|
|
165
|
+
The reference figures were rendered with matplotlib 3.10. matplotlib 3.11
|
|
166
|
+
reproduces every result table exactly but renders Figures 1-4 with small
|
|
167
|
+
pixel differences, which the QUICK figure comparison reports as `DIFFERS`;
|
|
168
|
+
use `pip install "matplotlib<3.11"` for a pixel-level figure match.
|
|
169
|
+
|
|
170
|
+
**Locked evaluation.** `simulations/configs/evaluation_locked.yaml` (seeds
|
|
171
|
+
3000-3999) was fixed and committed before the evaluation was run and was not
|
|
172
|
+
edited afterwards; its SHA-256 is recorded in `simulations/configs/SHA256SUMS`.
|
|
173
|
+
The calibration grid (`calibration.yaml`) and the development grid
|
|
174
|
+
(`development.yaml`) use disjoint seed ranges. `METHOD_FREEZE.md` lists the
|
|
175
|
+
frozen thresholds and method choices.
|
|
176
|
+
|
|
177
|
+
**Traceability.** `manuscript_number_trace.csv` maps every number reported in
|
|
178
|
+
the manuscript to the result file and column that produces it;
|
|
179
|
+
`figure_manifest.csv` and `table_manifest.csv` map figures and tables to their
|
|
180
|
+
generating scripts and inputs.
|
|
181
|
+
|
|
182
|
+
## Data acquisition
|
|
183
|
+
|
|
184
|
+
No raw data are redistributed in this repository. `python fetch_data.py`
|
|
185
|
+
downloads the two public datasets from their original repositories, verifies
|
|
186
|
+
each file against the SHA-256 recorded in `data/acquisition_ledger.csv`, and
|
|
187
|
+
extracts the files used by the analyses into `data/raw/` (git-ignored). See
|
|
188
|
+
`data/README.md` for sources, licenses, and citations.
|
|
189
|
+
|
|
190
|
+
## Repository structure
|
|
191
|
+
|
|
192
|
+
```text
|
|
193
|
+
src/falsesync/ package source
|
|
194
|
+
tests/ unit tests (theory identities, workflow)
|
|
195
|
+
examples/ minimal example
|
|
196
|
+
simulations/ simulation, calibration, evaluation, empirical scripts
|
|
197
|
+
configs/ development, calibration, locked evaluation grids
|
|
198
|
+
calibration/ calibrated classifier artifact
|
|
199
|
+
results/ simulated grid outputs
|
|
200
|
+
replication/ QUICK and FULL replication entry points
|
|
201
|
+
proofs/ proofs of propositions P1-P8
|
|
202
|
+
outputs/ reference figures
|
|
203
|
+
data/ acquisition ledger and data documentation
|
|
204
|
+
*.csv reference result tables
|
|
205
|
+
math_specification.md, diagnostic_specification.md, METHOD_FREEZE.md,
|
|
206
|
+
counterexamples.md method documentation
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
## Citation
|
|
210
|
+
|
|
211
|
+
See `CITATION.cff`. Please cite the software and the accompanying manuscript:
|
|
212
|
+
T. Onishi, "Aggregation-Induced False Synchrony in Change-Point Analysis"
|
|
213
|
+
(manuscript submitted to *Technometrics*).
|
|
214
|
+
|
|
215
|
+
## License
|
|
216
|
+
|
|
217
|
+
MIT (see `LICENSE`). Data obtained through `fetch_data.py` remain under their
|
|
218
|
+
original licenses (see `data/README.md`).
|
|
219
|
+
|
|
220
|
+
## Manuscript status
|
|
221
|
+
|
|
222
|
+
Version 0.1.2 corresponds to the manuscript as submitted to *Technometrics*.
|
|
223
|
+
The manuscript has not been peer reviewed or accepted.
|