falsesync 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- falsesync/__init__.py +9 -0
- falsesync/aggregation.py +85 -0
- falsesync/alignment.py +37 -0
- falsesync/changepoints.py +146 -0
- falsesync/datasets.py +68 -0
- falsesync/diagnostics.py +265 -0
- falsesync/features.py +164 -0
- falsesync/inference.py +90 -0
- falsesync/model.py +221 -0
- falsesync/plotting.py +59 -0
- falsesync/regimes.py +124 -0
- falsesync/simengine.py +277 -0
- falsesync/simulation.py +110 -0
- falsesync/transitions.py +56 -0
- falsesync-0.1.2.dist-info/METADATA +223 -0
- falsesync-0.1.2.dist-info/RECORD +19 -0
- falsesync-0.1.2.dist-info/WHEEL +5 -0
- falsesync-0.1.2.dist-info/licenses/LICENSE +21 -0
- falsesync-0.1.2.dist-info/top_level.txt +1 -0
falsesync/__init__.py
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""falsesync — aggregation-induced false synchrony diagnostics.
|
|
2
|
+
|
|
3
|
+
Core model: Y_i(t) = a_i + b_i(t) + A_i g_i((t - tau_i)/h_i) + eps_i(t).
|
|
4
|
+
The aggregate breakpoint is an operator-dependent functional
|
|
5
|
+
T_M(F_tau, g, w, ...) which in general equals no moment of F_tau.
|
|
6
|
+
See math_specification.md and proofs/ for P1-P8.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
__version__ = "0.1.2"
|
falsesync/aggregation.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Weighted aggregation and the population convolution m(t)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
from scipy.stats import norm
|
|
7
|
+
|
|
8
|
+
from .transitions import g
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def weighted_aggregate(
|
|
12
|
+
values: np.ndarray, weights: np.ndarray | None = None
|
|
13
|
+
) -> np.ndarray:
|
|
14
|
+
"""Row-weighted mean over units, respecting NaN observation windows.
|
|
15
|
+
|
|
16
|
+
values: (n_units, n_t); weights: (n_units,) constant over t, or
|
|
17
|
+
(n_units, n_t) for time-varying weights w_i(t).
|
|
18
|
+
Implements m_obs(t) = sum_i w_i 1{obs} Y_i / sum_i w_i 1{obs} (P5).
|
|
19
|
+
"""
|
|
20
|
+
v = np.asarray(values, dtype=float)
|
|
21
|
+
n = v.shape[0]
|
|
22
|
+
if weights is None:
|
|
23
|
+
w = np.ones((n, 1))
|
|
24
|
+
else:
|
|
25
|
+
w = np.asarray(weights, dtype=float)
|
|
26
|
+
if w.ndim == 1:
|
|
27
|
+
w = w[:, None]
|
|
28
|
+
elif w.shape != v.shape:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
f"2D weights must match values shape {v.shape}, got {w.shape}"
|
|
31
|
+
)
|
|
32
|
+
obs = np.isfinite(v)
|
|
33
|
+
wv = np.where(obs, v * w, 0.0)
|
|
34
|
+
denom = np.where(obs, w, 0.0).sum(axis=0)
|
|
35
|
+
with np.errstate(invalid="ignore", divide="ignore"):
|
|
36
|
+
return np.where(denom > 0, wv.sum(axis=0) / denom, np.nan)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def timing_cdf(tau_grid: np.ndarray, spec: dict) -> np.ndarray:
|
|
40
|
+
"""Population F_tau on a grid (used by the P1/P5 identities)."""
|
|
41
|
+
from .simulation import timing_density
|
|
42
|
+
|
|
43
|
+
tg = np.asarray(tau_grid, dtype=float)
|
|
44
|
+
f = timing_density(tg, spec)
|
|
45
|
+
c = np.concatenate([[0.0], np.cumsum((f[:-1] + f[1:]) / 2 * np.diff(tg))])
|
|
46
|
+
return np.clip(c / c[-1], 0, 1)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def population_m(
|
|
50
|
+
t: np.ndarray,
|
|
51
|
+
spec: dict,
|
|
52
|
+
shape: str = "logistic",
|
|
53
|
+
amplitude: float = 1.0,
|
|
54
|
+
baseline: float = 0.0,
|
|
55
|
+
width: float = 0.3,
|
|
56
|
+
tau_grid: np.ndarray | None = None,
|
|
57
|
+
) -> np.ndarray:
|
|
58
|
+
"""Population mean curve a + A * (g * f_tau)(t) via quadrature.
|
|
59
|
+
|
|
60
|
+
Closed forms used when available: step g -> F_tau(t) (P1);
|
|
61
|
+
probit + normal -> Phi((t-mu)/sqrt(1+sigma^2)) with unit width (P3a).
|
|
62
|
+
"""
|
|
63
|
+
t = np.asarray(t, dtype=float)
|
|
64
|
+
kind = spec.get("kind", "normal")
|
|
65
|
+
if shape == "step" and kind in ("normal", "mixture", "uniform"):
|
|
66
|
+
return baseline + amplitude * np.interp(
|
|
67
|
+
t, np.asarray(tau_grid if tau_grid is not None else t),
|
|
68
|
+
timing_cdf(np.asarray(tau_grid if tau_grid is not None else t), spec),
|
|
69
|
+
)
|
|
70
|
+
if shape == "probit" and kind == "normal" and width == 1.0:
|
|
71
|
+
return baseline + amplitude * norm.cdf(
|
|
72
|
+
(t - spec["mu"]) / np.sqrt(1 + spec["sigma"] ** 2)
|
|
73
|
+
)
|
|
74
|
+
tg = (
|
|
75
|
+
np.asarray(tau_grid, dtype=float)
|
|
76
|
+
if tau_grid is not None
|
|
77
|
+
else np.linspace(t.min() - 8 * width, t.max() + 8 * width, 4001)
|
|
78
|
+
)
|
|
79
|
+
from .simulation import timing_density
|
|
80
|
+
|
|
81
|
+
f = timing_density(tg, spec)
|
|
82
|
+
u = (t[:, None] - tg[None, :]) / width
|
|
83
|
+
trapz = getattr(np, "trapezoid", np.trapz)
|
|
84
|
+
conv = trapz(g(u, shape) * f[None, :], tg, axis=1)
|
|
85
|
+
return baseline + amplitude * conv
|
falsesync/alignment.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Event-time realignment (alignment in unit-specific transition time)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def event_time_realign(
|
|
9
|
+
t: np.ndarray, values: np.ndarray, taus: np.ndarray
|
|
10
|
+
) -> np.ndarray:
|
|
11
|
+
"""Re-express each unit series in event time s = t - tau_i.
|
|
12
|
+
|
|
13
|
+
Returns an (n, 2k+1) array on a common event-time grid centered at 0;
|
|
14
|
+
NaN where a unit's observed window does not cover s.
|
|
15
|
+
"""
|
|
16
|
+
t = np.asarray(t, dtype=float)
|
|
17
|
+
taus = np.asarray(taus, dtype=float)
|
|
18
|
+
dt = float(np.median(np.diff(t)))
|
|
19
|
+
half = float(np.minimum(taus - t.min(), t.max() - taus).max())
|
|
20
|
+
k = int(half / dt)
|
|
21
|
+
s_grid = np.arange(-k, k + 1) * dt
|
|
22
|
+
out = np.full((values.shape[0], s_grid.size), np.nan)
|
|
23
|
+
for i in range(values.shape[0]):
|
|
24
|
+
src_t = s_grid + taus[i]
|
|
25
|
+
valid = (src_t >= t.min()) & (src_t <= t.max())
|
|
26
|
+
out[i, valid] = np.interp(src_t[valid], t, values[i])
|
|
27
|
+
return out
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def realign_summary(aligned: np.ndarray, s_dt: float) -> dict:
|
|
31
|
+
"""Aligned-panel summary for the diagnostic's event_aligned_summary."""
|
|
32
|
+
agg = np.nanmean(aligned, axis=0)
|
|
33
|
+
d = np.gradient(np.nan_to_num(agg, nan=np.nanmean(agg[np.isfinite(agg)])), s_dt)
|
|
34
|
+
return {
|
|
35
|
+
"aligned_max_slope": float(np.nanmax(np.abs(d))),
|
|
36
|
+
"coverage": float(np.isfinite(aligned).mean()),
|
|
37
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Breakpoint operators T_M on a univariate series.
|
|
2
|
+
|
|
3
|
+
Each operator returns a BreakResult; the `method` tag identifies which
|
|
4
|
+
estimand was computed (see breakpoint_operator_comparison.md). ruptures is
|
|
5
|
+
used when installed; every method has a graceful fallback.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class BreakResult:
|
|
17
|
+
location: float
|
|
18
|
+
index: int
|
|
19
|
+
strength: float # (mean_R - mean_L) / sigma_hat
|
|
20
|
+
method: str
|
|
21
|
+
extras: dict = field(default_factory=dict)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def ls_breakpoint(t: np.ndarray, y: np.ndarray) -> BreakResult:
|
|
25
|
+
"""Single least-squares break in the mean (piecewise-constant model)."""
|
|
26
|
+
t = np.asarray(t, dtype=float)
|
|
27
|
+
y = np.asarray(y, dtype=float)
|
|
28
|
+
ok = np.isfinite(y)
|
|
29
|
+
t, y = t[ok], y[ok]
|
|
30
|
+
n = y.size
|
|
31
|
+
cum1 = np.concatenate([[0.0], np.cumsum(y)])
|
|
32
|
+
cum2 = np.concatenate([[0.0], np.cumsum(y * y)])
|
|
33
|
+
|
|
34
|
+
def sse(c: int) -> float:
|
|
35
|
+
l_sum, l_sq = cum1[c], cum2[c]
|
|
36
|
+
r_sum, r_sq = cum1[n] - cum1[c], cum2[n] - cum2[c]
|
|
37
|
+
l_sse = l_sq - l_sum**2 / max(c, 1)
|
|
38
|
+
r_sse = r_sq - r_sum**2 / max(n - c, 1)
|
|
39
|
+
return l_sse + r_sse
|
|
40
|
+
|
|
41
|
+
cs = np.arange(2, n - 2)
|
|
42
|
+
vals = np.array([sse(c) for c in cs])
|
|
43
|
+
c_hat = cs[int(np.argmin(vals))]
|
|
44
|
+
mean_l, mean_r = y[:c_hat].mean(), y[c_hat:].mean()
|
|
45
|
+
sigma = np.sqrt(max(sse(c_hat) / (n - 2), 1e-12))
|
|
46
|
+
return BreakResult(
|
|
47
|
+
location=float(t[c_hat]),
|
|
48
|
+
index=int(c_hat),
|
|
49
|
+
strength=float((mean_r - mean_l) / sigma),
|
|
50
|
+
method="ls",
|
|
51
|
+
extras={"mean_l": float(mean_l), "mean_r": float(mean_r), "sigma": float(sigma)},
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def maxslope_breakpoint(t: np.ndarray, y: np.ndarray) -> BreakResult:
|
|
56
|
+
"""Breakpoint at argmax of the numerical derivative (T_ms)."""
|
|
57
|
+
t = np.asarray(t, dtype=float)
|
|
58
|
+
y = np.asarray(y, dtype=float)
|
|
59
|
+
ok = np.isfinite(y)
|
|
60
|
+
t, y = t[ok], y[ok]
|
|
61
|
+
d = np.gradient(y, t)
|
|
62
|
+
i = int(np.argmax(np.abs(d)))
|
|
63
|
+
return BreakResult(
|
|
64
|
+
location=float(t[i]), index=i, strength=float(d[i]), method="maxslope"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def cusum_breakpoint(t: np.ndarray, y: np.ndarray) -> BreakResult:
|
|
69
|
+
"""Single CUSUM-type breakpoint (argmax |cumulative deviation|)."""
|
|
70
|
+
t = np.asarray(t, dtype=float)
|
|
71
|
+
y = np.asarray(y, dtype=float)
|
|
72
|
+
ok = np.isfinite(y)
|
|
73
|
+
t, y = t[ok], y[ok]
|
|
74
|
+
s = np.cumsum(y - y.mean())
|
|
75
|
+
i = int(np.argmax(np.abs(s)))
|
|
76
|
+
res = ls_breakpoint(t, y)
|
|
77
|
+
return BreakResult(t[i], i, res.strength, "cusum", {"ls_location": res.location})
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def multibreak(t: np.ndarray, y: np.ndarray, n_bkps: int = 3) -> BreakResult:
|
|
81
|
+
"""Penalized multi-break via binary segmentation (LS fallback).
|
|
82
|
+
|
|
83
|
+
Uses ruptures.Binseg when installed; otherwise a native binary
|
|
84
|
+
segmentation on the LS objective (same estimand family at k=1).
|
|
85
|
+
"""
|
|
86
|
+
t = np.asarray(t, dtype=float)
|
|
87
|
+
y = np.asarray(y, dtype=float)
|
|
88
|
+
ok = np.isfinite(y)
|
|
89
|
+
t, y = t[ok], y[ok]
|
|
90
|
+
try:
|
|
91
|
+
import ruptures as rpt
|
|
92
|
+
|
|
93
|
+
bkps = rpt.Binseg(model="l2").fit(y).predict(n_bkps=n_bkps)
|
|
94
|
+
idx = [b for b in bkps if b < y.size]
|
|
95
|
+
except Exception:
|
|
96
|
+
idx = _native_binseg(y, n_bkps)
|
|
97
|
+
main = idx[0] if idx else ls_breakpoint(t, y).index
|
|
98
|
+
res = ls_breakpoint(t, y)
|
|
99
|
+
return BreakResult(
|
|
100
|
+
float(t[main]), main, res.strength, "binseg",
|
|
101
|
+
{"all_break_indices": idx, "all_break_locations": [float(t[i]) for i in idx]},
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _native_binseg(y: np.ndarray, n_bkps: int) -> list[int]:
|
|
106
|
+
"""Greedy binary segmentation on the LS objective."""
|
|
107
|
+
n = y.size
|
|
108
|
+
segments = [(0, n)]
|
|
109
|
+
breaks: list[int] = []
|
|
110
|
+
while len(breaks) < n_bkps:
|
|
111
|
+
best_gain, best = 0.0, None
|
|
112
|
+
for lo, hi in segments:
|
|
113
|
+
if hi - lo < 4:
|
|
114
|
+
continue
|
|
115
|
+
seg = y[lo:hi]
|
|
116
|
+
cum1 = np.concatenate([[0.0], np.cumsum(seg)])
|
|
117
|
+
cum2 = np.concatenate([[0.0], np.cumsum(seg * seg)])
|
|
118
|
+
base = cum2[-1] - cum1[-1] ** 2 / seg.size
|
|
119
|
+
for c in range(2, seg.size - 2):
|
|
120
|
+
l = cum2[c] - cum1[c] ** 2 / c
|
|
121
|
+
r = (cum2[-1] - cum2[c]) - (cum1[-1] - cum1[c]) ** 2 / (seg.size - c)
|
|
122
|
+
gain = base - l - r
|
|
123
|
+
if gain > best_gain:
|
|
124
|
+
best_gain, best = gain, (lo, hi, lo + c)
|
|
125
|
+
if best is None:
|
|
126
|
+
break
|
|
127
|
+
lo, hi, c = best
|
|
128
|
+
breaks.append(c)
|
|
129
|
+
segments.remove((lo, hi))
|
|
130
|
+
segments.extend([(lo, c), (c, hi)])
|
|
131
|
+
return sorted(breaks)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def aggregate_breakpoint(
|
|
135
|
+
t: np.ndarray, y: np.ndarray, method: str = "ls", **kwargs
|
|
136
|
+
) -> BreakResult:
|
|
137
|
+
"""Dispatch a breakpoint operator. Methods: ls, maxslope, cusum, binseg, wbs."""
|
|
138
|
+
if method == "ls":
|
|
139
|
+
return ls_breakpoint(t, y)
|
|
140
|
+
if method == "maxslope":
|
|
141
|
+
return maxslope_breakpoint(t, y)
|
|
142
|
+
if method == "cusum":
|
|
143
|
+
return cusum_breakpoint(t, y)
|
|
144
|
+
if method in ("binseg", "wbs", "pelt"):
|
|
145
|
+
return multibreak(t, y, **kwargs)
|
|
146
|
+
raise ValueError(f"unknown method {method!r}")
|
falsesync/datasets.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Loaders for acquired public engineering datasets.
|
|
2
|
+
|
|
3
|
+
Raw snapshots live under data/raw/ (see data/acquisition_ledger.csv).
|
|
4
|
+
Loaders never mutate raw files; they return tidy DataFrames.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
import pandas as pd
|
|
14
|
+
|
|
15
|
+
DATA_ROOT = Path(__file__).resolve().parents[2] / "data" / "raw"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class PanelData:
|
|
20
|
+
"""Tidy panel: unit_id, t, value (+ optional weight/channel columns)."""
|
|
21
|
+
|
|
22
|
+
frame: pd.DataFrame
|
|
23
|
+
source: str
|
|
24
|
+
notes: str = ""
|
|
25
|
+
|
|
26
|
+
def to_unit_matrix(self) -> tuple[np.ndarray, np.ndarray, list[str]]:
|
|
27
|
+
"""(t, values[n_units, n_t], unit_ids) wide form for diagnostics."""
|
|
28
|
+
df = self.frame
|
|
29
|
+
wide = df.pivot_table(index="unit_id", columns="t", values="value", aggfunc="mean")
|
|
30
|
+
t = wide.columns.to_numpy(dtype=float)
|
|
31
|
+
return t, wide.to_numpy(dtype=float), list(wide.index.astype(str))
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def load_nasa_battery_cycles(data_dir: Path | None = None) -> PanelData:
|
|
35
|
+
"""NASA PCoE Li-ion capacity-vs-cycle per battery unit (discharge capacity)."""
|
|
36
|
+
root = (data_dir or DATA_ROOT) / "nasa_battery"
|
|
37
|
+
rows = []
|
|
38
|
+
for f in sorted(root.rglob("*.csv")):
|
|
39
|
+
df = pd.read_csv(f)
|
|
40
|
+
cap_col = next((c for c in df.columns if c.lower() in ("capacity", "discharge_capacity")), None)
|
|
41
|
+
cyc_col = next((c for c in df.columns if "cycle" in c.lower()), None)
|
|
42
|
+
if cap_col is None:
|
|
43
|
+
continue
|
|
44
|
+
unit = f.parent.name + "/" + f.stem
|
|
45
|
+
t_col = cyc_col or df.columns[0]
|
|
46
|
+
for _, r in df.iterrows():
|
|
47
|
+
rows.append({"unit_id": unit, "t": r[t_col], "value": r[cap_col]})
|
|
48
|
+
return PanelData(pd.DataFrame(rows), "nasa_pcoe_battery")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def load_scada_power(data_dir: Path | None = None, dataset: str = "kelmarsh") -> PanelData:
|
|
52
|
+
"""Kelmarsh/Penmanshiel SCADA: power output per turbine over time."""
|
|
53
|
+
root = (data_dir or DATA_ROOT) / dataset
|
|
54
|
+
rows = []
|
|
55
|
+
for f in sorted(root.rglob("*.csv")):
|
|
56
|
+
df = pd.read_csv(f, low_memory=False)
|
|
57
|
+
t_col = next((c for c in df.columns if "time" in c.lower() or "date" in c.lower()), df.columns[0])
|
|
58
|
+
p_col = next((c for c in df.columns if "power" in c.lower()), None)
|
|
59
|
+
u_col = next((c for c in df.columns if "turbine" in c.lower() or "unit" in c.lower()), None)
|
|
60
|
+
if p_col is None:
|
|
61
|
+
continue
|
|
62
|
+
uvals = df[u_col] if u_col else pd.Series(f.stem, index=df.index)
|
|
63
|
+
tt = pd.to_datetime(df[t_col], errors="coerce")
|
|
64
|
+
t_num = (tt - tt.min()).dt.total_seconds() / 86400.0
|
|
65
|
+
for u, gdf in df.assign(_t=t_num, _u=uvals).groupby("_u"):
|
|
66
|
+
for _, r in gdf.iterrows():
|
|
67
|
+
rows.append({"unit_id": str(u), "t": r["_t"], "value": r[p_col]})
|
|
68
|
+
return PanelData(pd.DataFrame(rows), f"{dataset}_scada")
|
falsesync/diagnostics.py
ADDED
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
"""False-synchrony diagnostic (see diagnostic_specification.md)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
|
|
9
|
+
from .aggregation import weighted_aggregate
|
|
10
|
+
from .changepoints import BreakResult, aggregate_breakpoint
|
|
11
|
+
from .inference import deconvolve_timing, unit_break_uncertainties
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class FalseSynchronyResult:
|
|
16
|
+
aggregate_breakpoint: float
|
|
17
|
+
aggregate_break_strength: float
|
|
18
|
+
unit_breakpoints: np.ndarray
|
|
19
|
+
unit_breakpoint_uncertainty: np.ndarray
|
|
20
|
+
timing_distribution: dict # {"grid": t, "density": f_hat, "deconvolved": f_dec}
|
|
21
|
+
timing_dispersion: dict # {"sd": ..., "iqr": ..., "ci": (lo, hi)}
|
|
22
|
+
cluster_structure: dict # {"n_components": k, "centers": [...], "weights": [...]}
|
|
23
|
+
synchrony_score: float
|
|
24
|
+
regime_probabilities: dict # keys: SYNCHRONOUS/CLUSTERED/DIFFUSE_ASYNCHRONOUS/NO_TRANSITION
|
|
25
|
+
event_aligned_summary: dict
|
|
26
|
+
interpretation_warning: list[str]
|
|
27
|
+
method: str = "ls"
|
|
28
|
+
aggregate_break_result: BreakResult | None = None
|
|
29
|
+
components: dict = field(default_factory=dict)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _kde(samples: np.ndarray, grid: np.ndarray, bw: float) -> np.ndarray:
|
|
33
|
+
u = (grid[:, None] - samples[None, :]) / bw
|
|
34
|
+
k = np.exp(-0.5 * u * u) / np.sqrt(2 * np.pi)
|
|
35
|
+
return k.mean(axis=1) / bw
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _fit_unit_breaks(
|
|
39
|
+
t: np.ndarray, values: np.ndarray, method: str = "ls"
|
|
40
|
+
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
41
|
+
"""Per-unit LS break: (tau_hat, strength, mean_l->r jump)."""
|
|
42
|
+
n = values.shape[0]
|
|
43
|
+
taus = np.full(n, np.nan)
|
|
44
|
+
strengths = np.full(n, np.nan)
|
|
45
|
+
idxs = np.full(n, -1, dtype=int)
|
|
46
|
+
for i in range(n):
|
|
47
|
+
y = values[i]
|
|
48
|
+
if np.isfinite(y).sum() < 8:
|
|
49
|
+
continue
|
|
50
|
+
try:
|
|
51
|
+
res = aggregate_breakpoint(t, np.nan_to_num(y, nan=np.nanmean(y[np.isfinite(y)])), method)
|
|
52
|
+
taus[i] = res.location
|
|
53
|
+
strengths[i] = res.strength
|
|
54
|
+
idxs[i] = res.index
|
|
55
|
+
except Exception:
|
|
56
|
+
continue
|
|
57
|
+
return taus, strengths, idxs
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def false_synchrony_diagnostic(
|
|
61
|
+
t: np.ndarray,
|
|
62
|
+
aggregate_series: np.ndarray | None,
|
|
63
|
+
unit_values: np.ndarray | None = None,
|
|
64
|
+
weights: np.ndarray | None = None,
|
|
65
|
+
windows: np.ndarray | None = None,
|
|
66
|
+
method: str = "ls",
|
|
67
|
+
rng: np.random.Generator | None = None,
|
|
68
|
+
n_boot: int = 200,
|
|
69
|
+
) -> FalseSynchronyResult:
|
|
70
|
+
"""Main diagnostic entry point (spec §9/§10).
|
|
71
|
+
|
|
72
|
+
If unit_values is None, returns an aggregate-only result whose
|
|
73
|
+
interpretation_warning states synchrony cannot be certified (P6c).
|
|
74
|
+
"""
|
|
75
|
+
rng = rng or np.random.default_rng(0)
|
|
76
|
+
t = np.asarray(t, dtype=float)
|
|
77
|
+
warnings: list[str] = []
|
|
78
|
+
if aggregate_series is None and unit_values is None:
|
|
79
|
+
raise ValueError("need aggregate_series or unit_values")
|
|
80
|
+
if aggregate_series is None:
|
|
81
|
+
aggregate_series = weighted_aggregate(unit_values, weights)
|
|
82
|
+
agg_res = aggregate_breakpoint(t, aggregate_series, method)
|
|
83
|
+
agg_break, agg_strength = agg_res.location, agg_res.strength
|
|
84
|
+
|
|
85
|
+
if unit_values is None:
|
|
86
|
+
return FalseSynchronyResult(
|
|
87
|
+
aggregate_breakpoint=agg_break,
|
|
88
|
+
aggregate_break_strength=agg_strength,
|
|
89
|
+
unit_breakpoints=np.array([]),
|
|
90
|
+
unit_breakpoint_uncertainty=np.array([]),
|
|
91
|
+
timing_distribution={"grid": t, "density": np.full(t.size, np.nan)},
|
|
92
|
+
timing_dispersion={"sd": np.nan, "iqr": np.nan, "ci": (np.nan, np.nan)},
|
|
93
|
+
cluster_structure={"n_components": 0, "centers": [], "weights": []},
|
|
94
|
+
synchrony_score=np.nan,
|
|
95
|
+
regime_probabilities={
|
|
96
|
+
"SYNCHRONOUS": np.nan, "CLUSTERED": np.nan,
|
|
97
|
+
"DIFFUSE_ASYNCHRONOUS": np.nan, "NO_TRANSITION": np.nan,
|
|
98
|
+
},
|
|
99
|
+
event_aligned_summary={},
|
|
100
|
+
interpretation_warning=[
|
|
101
|
+
"aggregate-only input: synchrony cannot be certified from the "
|
|
102
|
+
"aggregate curve alone (P6c); unit-level series required.",
|
|
103
|
+
"regime probabilities are not estimable without unit data.",
|
|
104
|
+
],
|
|
105
|
+
method=method,
|
|
106
|
+
aggregate_break_result=agg_res,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
taus, strengths, _ = _fit_unit_breaks(t, unit_values, method)
|
|
110
|
+
valid = np.isfinite(taus)
|
|
111
|
+
taus_v = taus[valid]
|
|
112
|
+
if taus_v.size == 0:
|
|
113
|
+
warnings.append(
|
|
114
|
+
"no estimable unit breakpoints (insufficient observations or "
|
|
115
|
+
"failed fits): synchrony cannot be certified; unit-level "
|
|
116
|
+
"fields are not estimable."
|
|
117
|
+
)
|
|
118
|
+
return FalseSynchronyResult(
|
|
119
|
+
aggregate_breakpoint=agg_break,
|
|
120
|
+
aggregate_break_strength=agg_strength,
|
|
121
|
+
unit_breakpoints=taus,
|
|
122
|
+
unit_breakpoint_uncertainty=np.full(taus.size, np.nan),
|
|
123
|
+
timing_distribution={"grid": t, "density": np.full(t.size, np.nan)},
|
|
124
|
+
timing_dispersion={"sd": np.nan, "iqr": np.nan, "ci": (np.nan, np.nan)},
|
|
125
|
+
cluster_structure={"n_components": 0, "centers": [], "weights": []},
|
|
126
|
+
synchrony_score=np.nan,
|
|
127
|
+
regime_probabilities={
|
|
128
|
+
"SYNCHRONOUS": np.nan, "CLUSTERED": np.nan,
|
|
129
|
+
"DIFFUSE_ASYNCHRONOUS": np.nan, "NO_TRANSITION": np.nan,
|
|
130
|
+
},
|
|
131
|
+
event_aligned_summary={},
|
|
132
|
+
interpretation_warning=warnings,
|
|
133
|
+
method=method,
|
|
134
|
+
aggregate_break_result=agg_res,
|
|
135
|
+
)
|
|
136
|
+
w_v = weights[valid] if weights is not None else None
|
|
137
|
+
s_i = unit_break_uncertainties(t, unit_values[valid], taus_v, rng=rng, n_boot=60)
|
|
138
|
+
|
|
139
|
+
grid = np.linspace(t.min(), t.max(), 400)
|
|
140
|
+
bw = max(1.06 * np.std(taus_v) * taus_v.size ** -0.2, (t[1] - t[0]))
|
|
141
|
+
f_hat = _kde(taus_v, grid, bw)
|
|
142
|
+
f_dec = deconvolve_timing(taus_v, s_i, grid)
|
|
143
|
+
sd = float(np.std(taus_v))
|
|
144
|
+
q = np.quantile(taus_v, [0.25, 0.75])
|
|
145
|
+
iqr = float(q[1] - q[0])
|
|
146
|
+
|
|
147
|
+
boot_sd = [
|
|
148
|
+
float(np.std(rng.choice(taus_v, taus_v.size, replace=True)))
|
|
149
|
+
for _ in range(n_boot)
|
|
150
|
+
]
|
|
151
|
+
sd_ci = tuple(np.quantile(boot_sd, [0.025, 0.975]).tolist())
|
|
152
|
+
|
|
153
|
+
# s1: timing spread vs aggregate transition width (FWHM of m')
|
|
154
|
+
dm = np.gradient(np.nan_to_num(aggregate_series, nan=np.nanmean(aggregate_series[np.isfinite(aggregate_series)])), t)
|
|
155
|
+
half = np.max(dm) / 2
|
|
156
|
+
fwhm_idx = np.where(dm >= half)[0]
|
|
157
|
+
w_agg = float(t[fwhm_idx[-1]] - t[fwhm_idx[0]]) if fwhm_idx.size > 1 else np.inf
|
|
158
|
+
s1_width = sd / w_agg if np.isfinite(w_agg) and w_agg > 0 else np.nan
|
|
159
|
+
|
|
160
|
+
# s2: mass concentration of deconvolved density (peak share vs uniform)
|
|
161
|
+
f_dec_n = np.clip(f_dec, 0, None)
|
|
162
|
+
f_dec_n = f_dec_n / np.trapz(f_dec_n, grid) if np.trapz(f_dec_n, grid) > 0 else f_dec_n
|
|
163
|
+
peak_share = float(np.max(f_dec_n) * (grid[1] - grid[0]) * 20)
|
|
164
|
+
s2_shape = min(peak_share, 1.0)
|
|
165
|
+
|
|
166
|
+
# s3: unit break alignment — fraction of units with a detectable break
|
|
167
|
+
detectable = np.isfinite(strengths) & (np.abs(strengths) > 2.0)
|
|
168
|
+
s3_alignment = float(detectable[valid].mean()) if valid.any() else np.nan
|
|
169
|
+
|
|
170
|
+
# s4: weight-timing correlation + window-timing correlation
|
|
171
|
+
s4_parts = []
|
|
172
|
+
if w_v is not None:
|
|
173
|
+
c_w = float(np.corrcoef(w_v, taus_v)[0, 1])
|
|
174
|
+
s4_parts.append(("weight_timing_corr", c_w))
|
|
175
|
+
if abs(c_w) > 0.3:
|
|
176
|
+
warnings.append(
|
|
177
|
+
f"weight-timing correlation {c_w:+.2f}: aggregate breakpoint is "
|
|
178
|
+
"tilted toward heavily weighted units' timing (P4)."
|
|
179
|
+
)
|
|
180
|
+
if windows is not None:
|
|
181
|
+
wv = np.asarray(windows)[valid]
|
|
182
|
+
span = wv[:, 1] - wv[:, 0]
|
|
183
|
+
c_win = float(np.corrcoef(span, taus_v)[0, 1])
|
|
184
|
+
s4_parts.append(("window_timing_corr", c_win))
|
|
185
|
+
if abs(c_win) > 0.3:
|
|
186
|
+
warnings.append(
|
|
187
|
+
f"window-timing correlation {c_win:+.2f}: observed aggregate is "
|
|
188
|
+
"a convolution against a time-varying observed timing law (P5)."
|
|
189
|
+
)
|
|
190
|
+
s4_weighting = float(np.mean([abs(c) for _, c in s4_parts])) if s4_parts else 0.0
|
|
191
|
+
|
|
192
|
+
# regime probabilities (simple distance-based classifier; calibrated in Phase 3)
|
|
193
|
+
noise_floor = float(np.median(s_i)) if s_i.size else np.nan
|
|
194
|
+
excess = sd**2 - noise_floor**2 if np.isfinite(noise_floor) else sd**2
|
|
195
|
+
p_sync = float(np.clip(1 - sd / max(2 * noise_floor, 1e-9), 0, 1)) if np.isfinite(noise_floor) else np.nan
|
|
196
|
+
# crude mixture check via dip in kde between two modes
|
|
197
|
+
modes = _local_maxima(f_dec_n)
|
|
198
|
+
p_cluster = float(len(modes) >= 2) * 0.7 + (0.3 if excess > noise_floor**2 else 0.0)
|
|
199
|
+
p_diffuse = float(np.clip(sd / (0.5 * (t.max() - t.min())), 0, 1))
|
|
200
|
+
p_notrans = float(np.clip(1 - abs(agg_strength) / 5, 0, 1)) if np.isfinite(agg_strength) else np.nan
|
|
201
|
+
probs = np.array([max(p_sync, 0.01), p_cluster, p_diffuse * 0.5, p_notrans * 0.2])
|
|
202
|
+
probs = probs / probs.sum()
|
|
203
|
+
regime_probabilities = dict(zip(
|
|
204
|
+
["SYNCHRONOUS", "CLUSTERED", "DIFFUSE_ASYNCHRONOUS", "NO_TRANSITION"],
|
|
205
|
+
[float(p) for p in probs],
|
|
206
|
+
))
|
|
207
|
+
|
|
208
|
+
components = {
|
|
209
|
+
"s1_width": s1_width,
|
|
210
|
+
"s2_shape": s2_shape,
|
|
211
|
+
"s3_alignment": s3_alignment,
|
|
212
|
+
"s4_weighting": s4_weighting,
|
|
213
|
+
}
|
|
214
|
+
finite = [v for v in components.values() if np.isfinite(v)]
|
|
215
|
+
synchrony_score = float(np.clip(1 - np.mean(finite), 0, 1)) if finite else np.nan
|
|
216
|
+
|
|
217
|
+
if excess <= 0:
|
|
218
|
+
warnings.append(
|
|
219
|
+
"unit break-time spread is within the unit-level estimation-noise "
|
|
220
|
+
"floor; observed dispersion may be entirely measurement (CE-UNITERR)."
|
|
221
|
+
)
|
|
222
|
+
warnings.append(
|
|
223
|
+
f"LS breakpoint {agg_break:.3g} is window-dependent (P7): "
|
|
224
|
+
"do not interpret as mean/median/mode of F_tau without the FOC check."
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
from .alignment import event_time_realign
|
|
228
|
+
|
|
229
|
+
aligned = event_time_realign(t, unit_values[valid], taus_v)
|
|
230
|
+
event_aligned_summary = {
|
|
231
|
+
"n_aligned": int(valid.sum()),
|
|
232
|
+
"aligned_agg_sharpness": float(np.nanmax(np.abs(np.gradient(
|
|
233
|
+
np.nanmean(aligned, axis=0)[np.isfinite(np.nanmean(aligned, axis=0))]
|
|
234
|
+
)))) if valid.any() else np.nan,
|
|
235
|
+
"tau_spread_before": sd,
|
|
236
|
+
"est_noise_floor": noise_floor,
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
centers = [float(grid[m]) for m in modes]
|
|
240
|
+
cluster_structure = {
|
|
241
|
+
"n_components": len(modes),
|
|
242
|
+
"centers": centers,
|
|
243
|
+
"weights": [float(f_dec_n[m]) for m in modes],
|
|
244
|
+
}
|
|
245
|
+
return FalseSynchronyResult(
|
|
246
|
+
aggregate_breakpoint=agg_break,
|
|
247
|
+
aggregate_break_strength=agg_strength,
|
|
248
|
+
unit_breakpoints=taus,
|
|
249
|
+
unit_breakpoint_uncertainty=s_i,
|
|
250
|
+
timing_distribution={"grid": grid, "density": f_hat, "deconvolved": f_dec_n},
|
|
251
|
+
timing_dispersion={"sd": sd, "iqr": iqr, "ci": sd_ci},
|
|
252
|
+
cluster_structure=cluster_structure,
|
|
253
|
+
synchrony_score=synchrony_score,
|
|
254
|
+
regime_probabilities=regime_probabilities,
|
|
255
|
+
event_aligned_summary=event_aligned_summary,
|
|
256
|
+
interpretation_warning=warnings,
|
|
257
|
+
method=method,
|
|
258
|
+
aggregate_break_result=agg_res,
|
|
259
|
+
components=components,
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _local_maxima(f: np.ndarray) -> list[int]:
|
|
264
|
+
out = [i for i in range(1, len(f) - 1) if f[i] >= f[i - 1] and f[i] >= f[i + 1]]
|
|
265
|
+
return [i for i in out if f[i] > 0.1 * f.max()] if f.size else []
|