fluxplot 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. fluxplot/__init__.py +115 -0
  2. fluxplot/_fieldmap.py +97 -0
  3. fluxplot/_mesh_reduce.py +54 -0
  4. fluxplot/_scene3d_size.py +95 -0
  5. fluxplot/_viewer/THIRD-PARTY.txt +23 -0
  6. fluxplot/_viewer/flux-model3d-viewer.min.js +4221 -0
  7. fluxplot/_viewer/stamp.json +4 -0
  8. fluxplot/api.py +1196 -0
  9. fluxplot/autotag.py +164 -0
  10. fluxplot/base.mplstyle +0 -0
  11. fluxplot/brackets.py +242 -0
  12. fluxplot/canonical_json.py +23 -0
  13. fluxplot/capture.py +150 -0
  14. fluxplot/colorcheck.py +285 -0
  15. fluxplot/colors.py +727 -0
  16. fluxplot/colorscale.py +477 -0
  17. fluxplot/data.py +178 -0
  18. fluxplot/definitions/colormaps.json +1639 -0
  19. fluxplot/definitions/flexoki.tokens.json +2571 -0
  20. fluxplot/definitions/palettes.json +2547 -0
  21. fluxplot/descriptors.py +87 -0
  22. fluxplot/fields.py +611 -0
  23. fluxplot/fits.py +240 -0
  24. fluxplot/glb.py +84 -0
  25. fluxplot/ids.py +173 -0
  26. fluxplot/images.py +362 -0
  27. fluxplot/integrity.py +27 -0
  28. fluxplot/manifest.py +788 -0
  29. fluxplot/mesh3d.py +376 -0
  30. fluxplot/panels.py +284 -0
  31. fluxplot/postprocess.py +638 -0
  32. fluxplot/presets.py +66 -0
  33. fluxplot/provenance.py +177 -0
  34. fluxplot/raster.py +295 -0
  35. fluxplot/recipe.py +178 -0
  36. fluxplot/render.py +66 -0
  37. fluxplot/roles.py +147 -0
  38. fluxplot/scene3d.py +386 -0
  39. fluxplot/scene3d_manifest.py +112 -0
  40. fluxplot/scene3d_viewer.py +633 -0
  41. fluxplot/schemas/.gitkeep +0 -0
  42. fluxplot/schemas/manifest.schema.json +2479 -0
  43. fluxplot/schemas/recipe.schema.json +179 -0
  44. fluxplot/schemas/scene3d.schema.json +461 -0
  45. fluxplot/seaborn_adapters.py +323 -0
  46. fluxplot/signature_fluxplots/__init__.py +18 -0
  47. fluxplot/signature_fluxplots/_colour.py +412 -0
  48. fluxplot/signature_fluxplots/fluxbox.py +433 -0
  49. fluxplot/signature_fluxplots/glowbar.py +769 -0
  50. fluxplot/signature_fluxplots/hexmatrix.py +927 -0
  51. fluxplot/stats/__init__.py +63 -0
  52. fluxplot/stats/_common.py +196 -0
  53. fluxplot/stats/multi_group.py +443 -0
  54. fluxplot/stats/paired.py +209 -0
  55. fluxplot/stats/two_group.py +149 -0
  56. fluxplot/style.py +469 -0
  57. fluxplot/surface.py +487 -0
  58. fluxplot/surface3d.py +197 -0
  59. fluxplot/tagger.py +561 -0
  60. fluxplot/version.py +19 -0
  61. fluxplot-0.1.0.dist-info/METADATA +1199 -0
  62. fluxplot-0.1.0.dist-info/RECORD +65 -0
  63. fluxplot-0.1.0.dist-info/WHEEL +4 -0
  64. fluxplot-0.1.0.dist-info/licenses/LICENSE +21 -0
  65. fluxplot-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +472 -0
@@ -0,0 +1,443 @@
1
+ """Tests for three or more groups — the omnibus question ("do the groups differ at all?") and the
2
+ post-hoc pairwise comparisons that follow it — plus :func:`pairwise`, which runs any two-group
3
+ test of this package over every pair of a family and corrects the p-values.
4
+
5
+ Each omnibus test returns one row keyed by :data:`REPORT_COLUMNS` (``groups`` lists every group,
6
+ ``n_total`` their combined size, ``dof`` / ``dof_error`` the numerator and denominator dof of an F
7
+ test); each post-hoc test returns one row per pair (``groups = [a, b]``, signs follow ``a - b``),
8
+ ready for :func:`fluxplot.brackets`, which draws them over a plot and records which test each star
9
+ came from.
10
+
11
+ Repeated-measures designs take a long table (``table``, a pandas / polars DataFrame or a dict of
12
+ columns) with a ``subject`` column, a ``within`` (condition) column and a ``dv`` column; every
13
+ subject must have every condition exactly once.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import itertools
18
+ from typing import Any, Callable, Dict, Iterable, List, Optional, Sequence, Tuple, Union
19
+
20
+ import numpy as np
21
+ from scipy import optimize as _opt
22
+ from scipy import stats as _sp
23
+
24
+ from ._common import REPORT_COLUMNS, as_sample, bh, holm, report_row
25
+ from .two_group import cliffs_delta, hedges_unpooled
26
+
27
+ __all__ = [
28
+ "anova_oneway", "welch_anova", "kruskal_epsilon", "rm_anova", "friedman_kendall",
29
+ "tukey_hsd", "games_howell", "dunn", "pairwise", "REPORT_COLUMNS",
30
+ ]
31
+
32
+ _Z = _sp.norm.ppf(0.975)
33
+ _BOOT_SEED = 0
34
+
35
+
36
+ # ---------------------------------------------------------------------------------------------
37
+ # inputs
38
+ # ---------------------------------------------------------------------------------------------
39
+ def _groups(groups: Sequence[Any], names: Optional[Sequence[str]], who: str) -> Tuple[List[np.ndarray], List[str]]:
40
+ if len(groups) == 1 and isinstance(groups[0], dict): # a {name: sample} mapping
41
+ names = list(groups[0]) if names is None else list(names)
42
+ groups = list(groups[0].values())
43
+ if len(groups) < 2:
44
+ raise ValueError(f"{who} needs at least 2 groups, got {len(groups)}")
45
+ if names is None:
46
+ names = [f"group{i + 1}" for i in range(len(groups))]
47
+ names = [str(n) for n in names]
48
+ if len(names) != len(groups):
49
+ raise ValueError(f"{who}: names has {len(names)} entries for {len(groups)} groups")
50
+ return [as_sample(g, n) for g, n in zip(groups, names)], names
51
+
52
+
53
+ def _column(table: Any, name: str) -> list:
54
+ try:
55
+ col = table[name]
56
+ except Exception as exc: # KeyError (pandas/dict), ColumnNotFoundError (polars), …
57
+ raise KeyError(f"{name!r} is not a column of the table") from exc
58
+ for attr in ("to_list", "tolist"):
59
+ if hasattr(col, attr):
60
+ return list(getattr(col, attr)())
61
+ return list(col)
62
+
63
+
64
+ def _matrix(table: Any, subject: str, within: str, dv: str, who: str) -> Tuple[np.ndarray, List[str], List[Any]]:
65
+ """The complete ``subjects × conditions`` matrix of a long table: conditions in order of first
66
+ appearance, subjects likewise; every cell filled exactly once."""
67
+ subs, conds, vals = _column(table, subject), _column(table, within), _column(table, dv)
68
+ if not (len(subs) == len(conds) == len(vals)):
69
+ raise ValueError(f"{who}: the columns differ in length")
70
+ sub_order = list(dict.fromkeys(subs))
71
+ cond_order = list(dict.fromkeys(conds))
72
+ n, k = len(sub_order), len(cond_order)
73
+ if k < 2:
74
+ raise ValueError(f"{who} needs at least 2 conditions in {within!r}, got {k}")
75
+ if n < 2:
76
+ raise ValueError(f"{who} needs at least 2 subjects in {subject!r}, got {n}")
77
+ m = np.full((n, k), np.nan)
78
+ si, ci = {s: i for i, s in enumerate(sub_order)}, {c: j for j, c in enumerate(cond_order)}
79
+ seen = set()
80
+ for s, c, v in zip(subs, conds, vals):
81
+ cell = (si[s], ci[c])
82
+ if cell in seen:
83
+ raise ValueError(f"{who}: subject {s!r} has more than one row for {within}={c!r}")
84
+ seen.add(cell)
85
+ m[cell] = float(v) if v is not None else np.nan
86
+ if not np.all(np.isfinite(m)):
87
+ missing = [(sub_order[i], cond_order[j]) for i, j in zip(*np.where(~np.isfinite(m)))]
88
+ raise ValueError(f"{who}: incomplete or non-finite cells for {missing[:3]}{'…' if len(missing) > 3 else ''}; "
89
+ "every subject needs a finite value for every condition")
90
+ return m, [str(c) for c in cond_order], sub_order
91
+
92
+
93
+ # ---------------------------------------------------------------------------------------------
94
+ # noncentral-F confidence intervals for variance-explained effect sizes (Steiger 2004)
95
+ # ---------------------------------------------------------------------------------------------
96
+ def _ncf_ncp_for(f_obs: float, df1: float, df2: float, target: float) -> float:
97
+ """The noncentrality ``lam >= 0`` with ``ncf.cdf(f_obs, df1, df2, lam) == target``, or 0 when
98
+ even the central distribution puts less than ``target`` below ``f_obs``."""
99
+ f = lambda lam: _sp.ncf.cdf(f_obs, df1, df2, lam) - target # noqa: E731
100
+ if f(0.0) <= 0:
101
+ return 0.0
102
+ hi = max(1.0, f_obs * df1)
103
+ while f(hi) > 0:
104
+ hi *= 2
105
+ return _opt.brentq(f, 0.0, hi, xtol=1e-10)
106
+
107
+
108
+ def _variance_explained_ci(f_obs: float, df1: float, df2: float) -> Tuple[float, float]:
109
+ """95% CI for the population proportion of variance explained (``eta^2`` / ``omega^2`` /
110
+ partial ``eta^2`` estimate the same parameter) from the noncentral F distribution of the
111
+ observed F: the bounds are ``lam / (lam + df1 + df2 + 1)`` for the noncentralities bracketing
112
+ ``f_obs`` at the 2.5% and 97.5% points (Steiger 2004; Kelley 2007). The lower bound is 0 when
113
+ ``p >= 0.025``."""
114
+ lo = _ncf_ncp_for(f_obs, df1, df2, 0.975)
115
+ hi = _ncf_ncp_for(f_obs, df1, df2, 0.025)
116
+ n_eff = df1 + df2 + 1
117
+ return lo / (lo + n_eff), hi / (hi + n_eff)
118
+
119
+
120
+ def _bootstrap_ci(statistic: Callable[[np.random.Generator], float], n_boot: int) -> Tuple[float, float]:
121
+ """A seeded percentile-bootstrap 95% interval (deterministic across runs)."""
122
+ rng = np.random.default_rng(_BOOT_SEED)
123
+ draws = np.array([statistic(rng) for _ in range(int(n_boot))])
124
+ draws = draws[np.isfinite(draws)]
125
+ if draws.size == 0:
126
+ return float("nan"), float("nan")
127
+ lo, hi = np.percentile(draws, [2.5, 97.5])
128
+ return float(lo), float(hi)
129
+
130
+
131
+ # ---------------------------------------------------------------------------------------------
132
+ # omnibus tests, independent groups
133
+ # ---------------------------------------------------------------------------------------------
134
+ def _sums_of_squares(groups: List[np.ndarray]):
135
+ allv = np.concatenate(groups)
136
+ grand = allv.mean()
137
+ ss_b = sum(g.size * (g.mean() - grand) ** 2 for g in groups)
138
+ ss_w = sum(((g - g.mean()) ** 2).sum() for g in groups)
139
+ return ss_b, ss_w, allv.size
140
+
141
+
142
+ def anova_oneway(*groups: Any, names: Optional[Sequence[str]] = None, effect: str = "eta2") -> Dict[str, Any]:
143
+ """Classic one-way ANOVA (equal variances assumed) with η² or ω² and its 95% CI.
144
+
145
+ ``F = MS_between / MS_within`` on ``(k - 1, N - k)`` dof (scipy's ``f_oneway``). The effect
146
+ size is η² = SS_between / SS_total (``effect="eta2"``, the default) or the less biased
147
+ ω² = (SS_between − (k − 1) MS_within) / (SS_total + MS_within) (``effect="omega2"``). Both
148
+ estimate the population proportion of variance explained, so both get the same 95% CI: the
149
+ noncentral-F interval of Steiger (2004), whose lower bound is 0 whenever ``p >= 0.025``.
150
+
151
+ Parameters
152
+ ----------
153
+ *groups
154
+ Two or more samples, or a single ``{name: sample}`` dict.
155
+ names
156
+ Group names (recorded in ``groups``); default ``group1, group2, …``.
157
+ effect
158
+ ``"eta2"`` or ``"omega2"``.
159
+ """
160
+ gs, names = _groups(groups, names, "anova_oneway")
161
+ if effect not in ("eta2", "omega2"):
162
+ raise ValueError(f"effect must be 'eta2' or 'omega2', got {effect!r}")
163
+ k = len(gs)
164
+ ss_b, ss_w, n = _sums_of_squares(gs)
165
+ df1, df2 = k - 1, n - k
166
+ if ss_w == 0:
167
+ raise ValueError("every group has zero within-group variance; F is undefined")
168
+ res = _sp.f_oneway(*gs)
169
+ ms_w = ss_w / df2
170
+ eta2 = ss_b / (ss_b + ss_w)
171
+ omega2 = (ss_b - df1 * ms_w) / (ss_b + ss_w + ms_w)
172
+ lo, hi = _variance_explained_ci(res.statistic, df1, df2)
173
+ value, method = (eta2, "Eta squared") if effect == "eta2" else (omega2, "Omega squared")
174
+ return report_row("One-way ANOVA", res.statistic, res.pvalue, df1, method, value, lo, hi,
175
+ n=[g.size for g in gs], groups=names, dof_error=df2)
176
+
177
+
178
+ def welch_anova(*groups: Any, names: Optional[Sequence[str]] = None) -> Dict[str, Any]:
179
+ """Welch's ANOVA (unequal variances) with ω² and its 95% CI.
180
+
181
+ Welch's (1951) F weights each group by ``n_i / var_i``; the denominator dof are
182
+ ``(k² − 1) / (3 Σ (1 − w_i / W)² / (n_i − 1))``. The effect size is
183
+ ω² = (k − 1)(F − 1) / ((k − 1)(F − 1) + N) (Kirk's formula; identical to the classic ω² when
184
+ the variances are equal), with the noncentral-F 95% CI on Welch's dof.
185
+ """
186
+ gs, names = _groups(groups, names, "welch_anova")
187
+ k = len(gs)
188
+ n = np.array([g.size for g in gs], dtype=float)
189
+ m = np.array([g.mean() for g in gs])
190
+ v = np.array([g.var(ddof=1) for g in gs])
191
+ if np.any(v == 0):
192
+ raise ValueError("a group has zero variance; Welch's weights are undefined")
193
+ w = n / v
194
+ big_w = w.sum()
195
+ grand = (w * m).sum() / big_w
196
+ lam = ((1 - w / big_w) ** 2 / (n - 1)).sum()
197
+ f_stat = ((w * (m - grand) ** 2).sum() / (k - 1)) / (1 + 2 * (k - 2) / (k**2 - 1) * lam)
198
+ df1 = k - 1
199
+ df2 = (k**2 - 1) / (3 * lam)
200
+ p = _sp.f.sf(f_stat, df1, df2)
201
+ total = int(n.sum())
202
+ omega2 = df1 * (f_stat - 1) / (df1 * (f_stat - 1) + total)
203
+ lo, hi = _variance_explained_ci(f_stat, df1, df2)
204
+ return report_row("Welch's ANOVA", f_stat, p, df1, "Omega squared", omega2, lo, hi,
205
+ n=[g.size for g in gs], groups=names, dof_error=df2)
206
+
207
+
208
+ def kruskal_epsilon(*groups: Any, names: Optional[Sequence[str]] = None, n_boot: int = 2000) -> Dict[str, Any]:
209
+ """Kruskal–Wallis H test with the ε² effect size and a bootstrap 95% CI.
210
+
211
+ ``H`` is scipy's tie-corrected statistic on ``k − 1`` dof. ε² = H / (N − 1) (Kelley 1935; the
212
+ rank analogue of η², in ``[0, 1]``). Its 95% CI is a seeded percentile bootstrap of ε² over
213
+ ``n_boot`` within-group resamples — deterministic across runs.
214
+ """
215
+ gs, names = _groups(groups, names, "kruskal_epsilon")
216
+ res = _sp.kruskal(*gs)
217
+ total = sum(g.size for g in gs)
218
+ eps2 = res.statistic / (total - 1)
219
+
220
+ def draw(rng):
221
+ boot = [rng.choice(g, g.size, replace=True) for g in gs]
222
+ try:
223
+ return _sp.kruskal(*boot).statistic / (total - 1)
224
+ except ValueError: # every resampled value identical
225
+ return np.nan
226
+
227
+ lo, hi = _bootstrap_ci(draw, n_boot)
228
+ return report_row("Kruskal–Wallis H test", res.statistic, res.pvalue, len(gs) - 1,
229
+ "Epsilon squared", eps2, lo, hi, n=[g.size for g in gs], groups=names)
230
+
231
+
232
+ # ---------------------------------------------------------------------------------------------
233
+ # omnibus tests, repeated measures
234
+ # ---------------------------------------------------------------------------------------------
235
+ def rm_anova(table: Any, subject: str, within: str, dv: str) -> Dict[str, Any]:
236
+ """One-way repeated-measures ANOVA with the Greenhouse–Geisser correction and partial η².
237
+
238
+ The within-subject decomposition ``SS_total = SS_subjects + SS_conditions + SS_error`` gives
239
+ ``F = MS_conditions / MS_error`` on ``(k − 1, (k − 1)(n − 1))`` dof. Sphericity is not assumed:
240
+ both dof are multiplied by the Greenhouse–Geisser ε (from the double-centred covariance matrix
241
+ of the conditions, clamped to ``[1 / (k − 1), 1]``) before the p-value is taken, and the row
242
+ reports those corrected dof. Partial η² = SS_conditions / (SS_conditions + SS_error), with the
243
+ noncentral-F 95% CI on the corrected dof.
244
+
245
+ ``groups`` lists the conditions; ``n_total`` is the number of subjects.
246
+ """
247
+ m, conds, subs = _matrix(table, subject, within, dv, "rm_anova")
248
+ n, k = m.shape
249
+ grand = m.mean()
250
+ ss_total = ((m - grand) ** 2).sum()
251
+ ss_subj = k * ((m.mean(axis=1) - grand) ** 2).sum()
252
+ ss_cond = n * ((m.mean(axis=0) - grand) ** 2).sum()
253
+ ss_err = ss_total - ss_subj - ss_cond
254
+ df1, df2 = k - 1, (k - 1) * (n - 1)
255
+ if ss_err <= 0:
256
+ raise ValueError("rm_anova: the error sum of squares is zero; F is undefined")
257
+ f_stat = (ss_cond / df1) / (ss_err / df2)
258
+ # Greenhouse–Geisser epsilon from the double-centred covariance matrix
259
+ cov = np.cov(m, rowvar=False, ddof=1)
260
+ centre = np.eye(k) - np.ones((k, k)) / k
261
+ dc = centre @ cov @ centre
262
+ eps = np.trace(dc) ** 2 / ((k - 1) * (dc**2).sum())
263
+ eps = float(min(1.0, max(eps, 1.0 / (k - 1))))
264
+ gdf1, gdf2 = eps * df1, eps * df2
265
+ p = _sp.f.sf(f_stat, gdf1, gdf2)
266
+ eta_p = ss_cond / (ss_cond + ss_err)
267
+ lo, hi = _variance_explained_ci(f_stat, gdf1, gdf2)
268
+ return report_row("Repeated-measures ANOVA (Greenhouse–Geisser)", f_stat, p, gdf1,
269
+ "Partial eta squared", eta_p, lo, hi, n=[n] * k, n_total=n, groups=conds, dof_error=gdf2)
270
+
271
+
272
+ def friedman_kendall(table: Any, subject: str, within: str, dv: str, n_boot: int = 2000) -> Dict[str, Any]:
273
+ """Friedman's test with Kendall's W and a bootstrap 95% CI.
274
+
275
+ scipy's ``friedmanchisquare`` on the ``subjects × conditions`` matrix gives the χ² statistic on
276
+ ``k − 1`` dof. Kendall's W = χ² / (n (k − 1)) is the agreement between subjects about the
277
+ ordering of the conditions, in ``[0, 1]``. Its 95% CI is a seeded percentile bootstrap over
278
+ subjects (rows resampled with replacement, ``n_boot`` draws).
279
+
280
+ ``groups`` lists the conditions; ``n_total`` is the number of subjects.
281
+ """
282
+ m, conds, subs = _matrix(table, subject, within, dv, "friedman_kendall")
283
+ n, k = m.shape
284
+ if k < 3:
285
+ raise ValueError("friedman_kendall needs at least 3 conditions (use a paired test for 2)")
286
+ res = _sp.friedmanchisquare(*m.T)
287
+ w = res.statistic / (n * (k - 1))
288
+
289
+ def draw(rng):
290
+ rows = m[rng.integers(0, n, n)]
291
+ try:
292
+ return _sp.friedmanchisquare(*rows.T).statistic / (n * (k - 1))
293
+ except ValueError:
294
+ return np.nan
295
+
296
+ lo, hi = _bootstrap_ci(draw, n_boot)
297
+ return report_row("Friedman test", res.statistic, res.pvalue, k - 1, "Kendall's W", w, lo, hi,
298
+ n=[n] * k, n_total=n, groups=conds)
299
+
300
+
301
+ # ---------------------------------------------------------------------------------------------
302
+ # post-hoc pairwise comparisons
303
+ # ---------------------------------------------------------------------------------------------
304
+ def _hedges_pooled(a: np.ndarray, b: np.ndarray) -> Tuple[float, float, float]:
305
+ """Hedges' g with the pooled SD (equal variances) and its large-sample 95% CI (Hedges & Olkin
306
+ 1985): ``(g, lo, hi)``."""
307
+ n1, n2 = a.size, b.size
308
+ sp = np.sqrt(((n1 - 1) * a.var(ddof=1) + (n2 - 1) * b.var(ddof=1)) / (n1 + n2 - 2))
309
+ if sp == 0:
310
+ raise ValueError("both samples have zero variance; the effect size is undefined")
311
+ d = (a.mean() - b.mean()) / sp
312
+ j = 1 - 3 / (4 * (n1 + n2 - 2) - 1)
313
+ g = j * d
314
+ se = np.sqrt((n1 + n2) / (n1 * n2) + g**2 / (2 * (n1 + n2)))
315
+ return float(g), float(g - _Z * se), float(g + _Z * se)
316
+
317
+
318
+ def _pairs(k: int) -> List[Tuple[int, int]]:
319
+ return list(itertools.combinations(range(k), 2))
320
+
321
+
322
+ def tukey_hsd(*groups: Any, names: Optional[Sequence[str]] = None) -> List[Dict[str, Any]]:
323
+ """Tukey's honestly-significant-difference test over every pair, with Hedges' g (pooled SD).
324
+
325
+ scipy's ``tukey_hsd`` gives each pair's mean difference ``a − b`` and its p-value from the
326
+ studentized range on ``N − k`` dof. That p-value already controls the family-wise error rate,
327
+ so ``p_corrected_holm`` and ``p_corrected_bh`` equal it: do **not** correct these rows again.
328
+ The effect size is Hedges' g with the pooled SD (Tukey assumes equal variances), with the
329
+ Hedges–Olkin large-sample 95% CI.
330
+ """
331
+ gs, names = _groups(groups, names, "tukey_hsd")
332
+ res = _sp.tukey_hsd(*gs)
333
+ total = sum(g.size for g in gs)
334
+ rows = []
335
+ for i, j in _pairs(len(gs)):
336
+ g, lo, hi = _hedges_pooled(gs[i], gs[j])
337
+ rows.append(report_row("Tukey HSD", res.statistic[i, j], res.pvalue[i, j], total - len(gs),
338
+ "Hedges' g (pooled SD)", g, lo, hi, n=(gs[i].size, gs[j].size),
339
+ groups=(names[i], names[j])))
340
+ return rows
341
+
342
+
343
+ def games_howell(*groups: Any, names: Optional[Sequence[str]] = None) -> List[Dict[str, Any]]:
344
+ """The Games–Howell test over every pair (unequal variances and sizes), with Hedges' g
345
+ (non-pooled SD).
346
+
347
+ Each pair's ``t = (mean_a − mean_b) / sqrt(var_a / n_a + var_b / n_b)`` on Welch–Satterthwaite
348
+ dof is referred to the studentized range distribution of ``k`` groups
349
+ (``p = P(Q_{k, df} > |t| sqrt 2)``), which controls the family-wise error rate: as for
350
+ :func:`tukey_hsd`, the corrected columns equal ``p-value``. The effect size and its Bonett CI
351
+ are those of :func:`fluxplot.stats.welch_hedges`.
352
+ """
353
+ gs, names = _groups(groups, names, "games_howell")
354
+ k = len(gs)
355
+ rows = []
356
+ for i, j in _pairs(k):
357
+ a, b = gs[i], gs[j]
358
+ va, vb = a.var(ddof=1) / a.size, b.var(ddof=1) / b.size
359
+ if va + vb == 0:
360
+ raise ValueError(f"groups {names[i]!r} and {names[j]!r} have zero variance")
361
+ t = (a.mean() - b.mean()) / np.sqrt(va + vb)
362
+ df = (va + vb) ** 2 / (va**2 / (a.size - 1) + vb**2 / (b.size - 1))
363
+ p = _sp.studentized_range.sf(abs(t) * np.sqrt(2), k, df)
364
+ g, lo, hi = hedges_unpooled(a, b)
365
+ rows.append(report_row("Games–Howell test", t, p, df, "Hedges' g (non-pooled SD)", g, lo, hi,
366
+ n=(a.size, b.size), groups=(names[i], names[j])))
367
+ return rows
368
+
369
+
370
+ def dunn(*groups: Any, names: Optional[Sequence[str]] = None, adjust: Union[str, Sequence[str]] = "holm") -> List[Dict[str, Any]]:
371
+ """Dunn's (1964) rank-sum test over every pair after Kruskal–Wallis, with Cliff's delta.
372
+
373
+ The pooled ranks give each pair ``z = (R̄_a − R̄_b) / sqrt((N (N + 1) / 12 − T) (1 / n_a + 1 / n_b))``
374
+ with the tie term ``T = Σ (t³ − t) / (12 (N − 1))``; ``p-value`` is the raw two-sided normal
375
+ p, and ``adjust`` (``"holm"``, ``"bh"``, or both) fills the matching corrected column across
376
+ the pairs. The effect size is Cliff's delta with Newcombe's 95% CI, as in
377
+ :func:`fluxplot.stats.mann_whitney_cliff`.
378
+ """
379
+ gs, names = _groups(groups, names, "dunn")
380
+ pooled = np.concatenate(gs)
381
+ ranks = _sp.rankdata(pooled)
382
+ total = pooled.size
383
+ _, counts = np.unique(pooled, return_counts=True)
384
+ tie_term = float(((counts**3 - counts).sum()) / (12 * (total - 1)))
385
+ bounds = np.cumsum([0] + [g.size for g in gs])
386
+ mean_ranks = [ranks[bounds[i]:bounds[i + 1]].mean() for i in range(len(gs))]
387
+ rows = []
388
+ for i, j in _pairs(len(gs)):
389
+ a, b = gs[i], gs[j]
390
+ se = np.sqrt((total * (total + 1) / 12 - tie_term) * (1 / a.size + 1 / b.size))
391
+ z = (mean_ranks[i] - mean_ranks[j]) / se
392
+ p = 2 * _sp.norm.sf(abs(z))
393
+ delta, lo, hi = cliffs_delta(a, b)
394
+ rows.append(report_row("Dunn's test", z, p, None, "Cliff's delta", delta, lo, hi,
395
+ n=(a.size, b.size), groups=(names[i], names[j])))
396
+ return _adjust(rows, adjust)
397
+
398
+
399
+ def _adjust(rows: List[dict], adjust: Union[str, Sequence[str], None]) -> List[dict]:
400
+ if adjust is None:
401
+ return rows
402
+ kinds = (adjust,) if isinstance(adjust, str) else tuple(adjust)
403
+ for kind in kinds:
404
+ if kind == "holm":
405
+ rows = holm(rows)
406
+ elif kind == "bh":
407
+ rows = bh(rows)
408
+ else:
409
+ raise ValueError(f"adjust must be 'holm', 'bh' or a sequence of them, got {kind!r}")
410
+ return rows
411
+
412
+
413
+ def pairwise(test: Callable[..., dict], groups: Dict[str, Any], pairs: Optional[Iterable[Tuple[str, str]]] = None,
414
+ adjust: Union[str, Sequence[str], None] = "holm") -> List[Dict[str, Any]]:
415
+ """Run a two-group test of this package over pairs of named groups and correct the p-values.
416
+
417
+ >>> rows = fp.stats.pairwise(fp.stats.welch_hedges, {"ctl": ctl, "drug": drug, "sham": sham})
418
+ >>> fp.brackets(ax, rows, positions=gb.positions)
419
+
420
+ Parameters
421
+ ----------
422
+ test
423
+ ``welch_hedges``, ``mann_whitney_cliff``, ``paired_t_hedges`` or ``wilcoxon_rank_biserial``
424
+ (any callable ``test(a, b, names=(a_name, b_name)) -> row``).
425
+ groups
426
+ ``{name: sample}``; the order fixes the default pairs and the sign convention (``a − b``
427
+ with ``a`` the earlier group).
428
+ pairs
429
+ The ``(a, b)`` name pairs to compare; default every pair in ``groups`` order.
430
+ adjust
431
+ ``"holm"`` (default), ``"bh"``, a sequence of both, or ``None``: the correction(s) filled
432
+ in across the family (``p_corrected_holm`` / ``p_corrected_bh``).
433
+ """
434
+ names = list(groups)
435
+ if pairs is None:
436
+ pairs = list(itertools.combinations(names, 2))
437
+ rows = []
438
+ for a, b in pairs:
439
+ for name in (a, b):
440
+ if name not in groups:
441
+ raise KeyError(f"pairwise: {name!r} is not one of the groups {names}")
442
+ rows.append(test(groups[a], groups[b], names=(a, b)))
443
+ return _adjust(rows, adjust)
@@ -0,0 +1,209 @@
1
+ """Tests for paired samples: matched observations where ``a[i]`` and ``b[i]`` belong together
2
+ (the same animal before/after, the same cell under two conditions).
3
+
4
+ Each function returns one row as a dict keyed by :data:`REPORT_COLUMNS`, like the independent-sample
5
+ tests in :mod:`fluxplot.stats.two_group`. Differences are always taken as ``a - b``.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from typing import Any, Dict, Tuple
10
+
11
+ import numpy as np
12
+ from scipy import optimize as _opt
13
+ from scipy import special as _sc
14
+ from scipy import stats as _sp
15
+
16
+ from ._common import REPORT_COLUMNS, as_pairs, check_alternative, report_row
17
+
18
+ __all__ = ["paired_t_hedges", "wilcoxon_rank_biserial", "REPORT_COLUMNS"]
19
+
20
+ _Z = _sp.norm.ppf(0.975)
21
+
22
+
23
+ def _nct_ncp_for(t_obs: float, df: float, target: float) -> float:
24
+ """The noncentrality ``ncp`` with ``nct.cdf(t_obs, df, ncp) == target``.
25
+
26
+ The cdf falls monotonically as ``ncp`` grows, so the root is bracketed by stepping outward from
27
+ ``t_obs`` (where the cdf is near 0.5) in doubling steps, then solved with Brent's method.
28
+ """
29
+ f = lambda ncp: _sp.nct.cdf(t_obs, df, ncp) - target # noqa: E731
30
+ step = 1.0
31
+ if target > 0.5: # the root lies below t_obs
32
+ lo, hi = t_obs - step, t_obs
33
+ while f(lo) < 0:
34
+ step *= 2
35
+ lo, hi = t_obs - step, lo
36
+ else: # the root lies above t_obs
37
+ lo, hi = t_obs, t_obs + step
38
+ while f(hi) > 0:
39
+ step *= 2
40
+ lo, hi = hi, t_obs + step
41
+ return _opt.brentq(f, lo, hi, xtol=1e-12)
42
+
43
+
44
+ def paired_t_hedges(a: Any, b: Any, *, alternative: str = "two-sided", names=("a", "b")) -> Dict[str, Any]:
45
+ """Paired t-test of ``a`` vs ``b`` with Hedges' g_z and its 95% CI.
46
+
47
+ For matched observations, asking whether the mean of the paired differences ``a - b`` is
48
+ different from zero. The effect size is ``d_z = mean(a - b) / sd(a - b)`` (standardized by the
49
+ SD of the paired differences) times the small-sample correction ``J = 1 - 3 / (4 (n - 1) - 1)``.
50
+ The 95% CI is exact under normality: the noncentral-t interval for the population ``d_z``,
51
+ inverting the noncentral t distribution of the observed t with ``n - 1`` degrees of freedom.
52
+ It is not multiplied by ``J``: that would break its exactness (at n = 6 and ``d_z = 2``
53
+ coverage would fall to about 90%).
54
+
55
+ Signs follow ``a - b``: a positive statistic and effect size mean ``a`` is larger on average.
56
+
57
+ Parameters
58
+ ----------
59
+ a, b
60
+ The paired samples, equal length, ``a[i]`` matched with ``b[i]``; at least 2 pairs of finite
61
+ values.
62
+ alternative
63
+ ``"two-sided"`` (default), ``"less"`` or ``"greater"`` for the mean difference ``a - b``.
64
+ The CI is always two-sided.
65
+ names
66
+ The names of the two conditions, recorded in the row's ``groups``.
67
+
68
+ Returns
69
+ -------
70
+ dict
71
+ One reporting row keyed by :data:`REPORT_COLUMNS`: the t statistic, the p-value,
72
+ ``dof = n - 1``, Hedges' g_z and its 95% CI. ``n_a = n_b = n_total = n``, the number of pairs.
73
+ """
74
+ a, b = as_pairs(a, b)
75
+ diff = a - b
76
+ n = diff.size
77
+ sd = diff.std(ddof=1)
78
+ if sd == 0:
79
+ raise ValueError("the paired differences have zero variance; the effect size is undefined")
80
+
81
+ t = _sp.ttest_rel(a, b, alternative=check_alternative(alternative))
82
+ df = n - 1
83
+ d_z = diff.mean() / sd
84
+ j = 1 - 3 / (4 * df - 1)
85
+ t_obs = d_z * np.sqrt(n) # identical to t.statistic
86
+ lo = _nct_ncp_for(t_obs, df, 0.975) / np.sqrt(n)
87
+ hi = _nct_ncp_for(t_obs, df, 0.025) / np.sqrt(n)
88
+ return report_row("Paired t-test", t.statistic, t.pvalue, df, "Hedges' g_z", j * d_z, lo, hi,
89
+ n=(n, n), n_total=n, groups=names, alternative=alternative)
90
+
91
+
92
+ def _bvn_cdf_same_sign(h: float, k: float, rho: float) -> float:
93
+ """Standard bivariate-normal ``P(X < h, Y < k)`` with correlation ``rho``, for ``h * k > 0``
94
+ or ``h = k = 0`` (all this module needs), via Owen's (1956) T-function identity."""
95
+ if h == 0 and k == 0:
96
+ return 0.25 + np.arcsin(rho) / (2 * np.pi)
97
+ s = np.sqrt(1 - rho**2)
98
+ return (0.5 * (_sp.norm.cdf(h) + _sp.norm.cdf(k))
99
+ - _sc.owens_t(h, (k - rho * h) / (h * s)) - _sc.owens_t(k, (h - rho * k) / (k * s)))
100
+
101
+
102
+ def _rank_biserial_moments(delta: float, n: int) -> Tuple[float, float]:
103
+ """Mean and variance of the matched-pairs rank-biserial ``r`` from ``n`` differences
104
+ ``D ~ Normal(delta, 1)``.
105
+
106
+ ``W+ = sum_{i<=j} 1[D_i + D_j > 0]`` (Walsh averages), so with ``p1 = P(D > 0)``,
107
+ ``p2 = P(D1 + D2 > 0)``, ``p3 = P(D1 + D2 > 0, D1 + D3 > 0)`` and ``p4 = P(D1 > 0, D1 + D2 > 0)``:
108
+ ``E W+ = n p1 + C(n, 2) p2`` and
109
+ ``Var W+ = n p1 (1 - p1) + C(n, 2) p2 (1 - p2) + n (n-1) (n-2) (p3 - p2^2)
110
+ + 2 n (n-1) (p4 - p1 p2)`` — at ``delta = 0`` this is the familiar ``n (n+1) (2n+1) / 24``.
111
+ """
112
+ r2 = np.sqrt(2)
113
+ p1 = _sp.norm.cdf(delta)
114
+ p2 = _sp.norm.cdf(r2 * delta)
115
+ p3 = _bvn_cdf_same_sign(r2 * delta, r2 * delta, 0.5)
116
+ p4 = _bvn_cdf_same_sign(delta, r2 * delta, 1 / r2)
117
+ total = n * (n + 1) / 2
118
+ mean_w = n * p1 + n * (n - 1) / 2 * p2
119
+ var_w = (n * p1 * (1 - p1) + n * (n - 1) / 2 * p2 * (1 - p2)
120
+ + n * (n - 1) * (n - 2) * (p3 - p2**2) + 2 * n * (n - 1) * (p4 - p1 * p2))
121
+ return 2 * mean_w / total - 1, 4 * max(var_w, 0.0) / total**2
122
+
123
+
124
+ def _rank_biserial_bounds(r: float, n: int) -> Tuple[float, float]:
125
+ """Score-type 95% interval for the population rank-biserial ``rho = 2 P(D1 + D2 > 0) - 1``.
126
+
127
+ Keeps every ``delta`` whose model mean of ``r`` lies within ``z`` model SDs of the observed
128
+ ``r`` — the bounds solve ``(r - E[r | delta])^2 = z^2 Var(r | delta)`` — and maps them to
129
+ ``rho = 2 Phi(sqrt(2) delta) - 1``. Because the variance is evaluated at each candidate
130
+ rather than at the estimate, the interval stays inside ``[-1, 1]`` and does not collapse
131
+ when every difference has the same sign (``r = ±1``).
132
+ """
133
+ lim = 8.0 # |delta| = 8 puts rho within 1e-15 of ±1
134
+
135
+ def mean_minus_r(d: float) -> float:
136
+ return _rank_biserial_moments(d, n)[0] - r
137
+
138
+ def gap(d: float) -> float:
139
+ mean, var = _rank_biserial_moments(d, n)
140
+ return (r - mean) ** 2 - _Z**2 * var
141
+
142
+ if r >= 1:
143
+ d_hat = lim
144
+ elif r <= -1:
145
+ d_hat = -lim
146
+ else:
147
+ d_hat = _opt.brentq(mean_minus_r, -lim, lim, xtol=1e-12)
148
+ d_lo = -lim if r <= -1 else _opt.brentq(gap, -lim, d_hat, xtol=1e-12)
149
+ d_hi = lim if r >= 1 else _opt.brentq(gap, d_hat, lim, xtol=1e-12)
150
+ rho = lambda d: 2 * _sp.norm.cdf(np.sqrt(2) * d) - 1 # noqa: E731
151
+ return rho(d_lo), rho(d_hi)
152
+
153
+
154
+ def wilcoxon_rank_biserial(a: Any, b: Any, *, alternative: str = "two-sided", names=("a", "b")) -> Dict[str, Any]:
155
+ """Wilcoxon signed-rank test of ``a`` vs ``b`` with the matched-pairs rank-biserial correlation.
156
+
157
+ For matched observations compared by the ranks of their differences ``a - b`` (chosen before
158
+ looking at the results; assumes the differences are meaningful in size and symmetrically
159
+ distributed). Zero differences are dropped (Wilcoxon's convention, scipy's
160
+ ``zero_method="wilcox"``) and tied ``|a - b|`` share their average rank. The p-value is scipy's
161
+ two-sided ``wilcoxon`` (``method="auto"``: exact for small samples, otherwise the normal
162
+ approximation).
163
+
164
+ The effect size is Kerby's (2014) matched-pairs rank-biserial correlation
165
+ ``r = (W+ - W-) / (W+ + W-)`` over the ``n`` non-zero differences: the share of the rank sum
166
+ favouring ``a > b`` minus the share favouring ``a < b``, in ``[-1, 1]``.
167
+
168
+ Its 95% CI is a score-type interval (the construction of Wilson's and Newcombe's intervals)
169
+ for the population value ``rho = 2 P(D_i + D_j > 0) - 1``: it inverts the normal approximation
170
+ to ``W+`` using ``W+``'s exact mean and variance, evaluated under a normal working model for the
171
+ differences. It stays inside ``[-1, 1]`` and does not collapse when every difference has the
172
+ same sign. In simulation (n = 6–20; normal, t3, Laplace and uniform differences) its coverage
173
+ is 93–97%. The Fisher-z interval of the R ``effectsize`` package, by contrast, covers as little
174
+ as 43–84% at n = 6.
175
+
176
+ Parameters
177
+ ----------
178
+ a, b
179
+ The paired samples, equal length, ``a[i]`` matched with ``b[i]``; at least 2 pairs of finite
180
+ values, and at least one non-zero difference.
181
+ alternative
182
+ ``"two-sided"`` (default), ``"less"`` or ``"greater"`` for the differences ``a - b``. The
183
+ CI is always two-sided.
184
+ names
185
+ The names of the two conditions, recorded in the row's ``groups``.
186
+
187
+ Returns
188
+ -------
189
+ dict
190
+ One reporting row keyed by :data:`REPORT_COLUMNS`. The statistic is ``W+``, the sum of the
191
+ ranks of the positive differences (R's ``V``; scipy's two-sided statistic is instead
192
+ ``min(W+, W-)``); ``dof`` is ``None`` (a rank test has no degrees of freedom).
193
+ """
194
+ a, b = as_pairs(a, b)
195
+ diff = a - b
196
+ diff = diff[diff != 0]
197
+ n = diff.size
198
+ if n == 0:
199
+ raise ValueError("every paired difference is zero; the signed-rank test is undefined")
200
+
201
+ w = _sp.wilcoxon(a, b, zero_method="wilcox", alternative=check_alternative(alternative), method="auto")
202
+ ranks = _sp.rankdata(np.abs(diff)) # average ranks for ties
203
+ w_plus = ranks[diff > 0].sum()
204
+ w_minus = ranks[diff < 0].sum()
205
+ r = (w_plus - w_minus) / (w_plus + w_minus)
206
+ lo, hi = _rank_biserial_bounds(r, n)
207
+ return report_row("Wilcoxon signed-rank test", w_plus, w.pvalue, None,
208
+ "Matched-pairs rank-biserial correlation", r, lo, hi,
209
+ n=(a.size, b.size), n_total=a.size, groups=names, alternative=alternative)