fluxplot 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fluxplot/__init__.py +115 -0
- fluxplot/_fieldmap.py +97 -0
- fluxplot/_mesh_reduce.py +54 -0
- fluxplot/_scene3d_size.py +95 -0
- fluxplot/_viewer/THIRD-PARTY.txt +23 -0
- fluxplot/_viewer/flux-model3d-viewer.min.js +4221 -0
- fluxplot/_viewer/stamp.json +4 -0
- fluxplot/api.py +1196 -0
- fluxplot/autotag.py +164 -0
- fluxplot/base.mplstyle +0 -0
- fluxplot/brackets.py +242 -0
- fluxplot/canonical_json.py +23 -0
- fluxplot/capture.py +150 -0
- fluxplot/colorcheck.py +285 -0
- fluxplot/colors.py +727 -0
- fluxplot/colorscale.py +477 -0
- fluxplot/data.py +178 -0
- fluxplot/definitions/colormaps.json +1639 -0
- fluxplot/definitions/flexoki.tokens.json +2571 -0
- fluxplot/definitions/palettes.json +2547 -0
- fluxplot/descriptors.py +87 -0
- fluxplot/fields.py +611 -0
- fluxplot/fits.py +240 -0
- fluxplot/glb.py +84 -0
- fluxplot/ids.py +173 -0
- fluxplot/images.py +362 -0
- fluxplot/integrity.py +27 -0
- fluxplot/manifest.py +788 -0
- fluxplot/mesh3d.py +376 -0
- fluxplot/panels.py +284 -0
- fluxplot/postprocess.py +638 -0
- fluxplot/presets.py +66 -0
- fluxplot/provenance.py +177 -0
- fluxplot/raster.py +295 -0
- fluxplot/recipe.py +178 -0
- fluxplot/render.py +66 -0
- fluxplot/roles.py +147 -0
- fluxplot/scene3d.py +386 -0
- fluxplot/scene3d_manifest.py +112 -0
- fluxplot/scene3d_viewer.py +633 -0
- fluxplot/schemas/.gitkeep +0 -0
- fluxplot/schemas/manifest.schema.json +2479 -0
- fluxplot/schemas/recipe.schema.json +179 -0
- fluxplot/schemas/scene3d.schema.json +461 -0
- fluxplot/seaborn_adapters.py +323 -0
- fluxplot/signature_fluxplots/__init__.py +18 -0
- fluxplot/signature_fluxplots/_colour.py +412 -0
- fluxplot/signature_fluxplots/fluxbox.py +433 -0
- fluxplot/signature_fluxplots/glowbar.py +769 -0
- fluxplot/signature_fluxplots/hexmatrix.py +927 -0
- fluxplot/stats/__init__.py +63 -0
- fluxplot/stats/_common.py +196 -0
- fluxplot/stats/multi_group.py +443 -0
- fluxplot/stats/paired.py +209 -0
- fluxplot/stats/two_group.py +149 -0
- fluxplot/style.py +469 -0
- fluxplot/surface.py +487 -0
- fluxplot/surface3d.py +197 -0
- fluxplot/tagger.py +561 -0
- fluxplot/version.py +19 -0
- fluxplot-0.1.0.dist-info/METADATA +1199 -0
- fluxplot-0.1.0.dist-info/RECORD +65 -0
- fluxplot-0.1.0.dist-info/WHEEL +4 -0
- fluxplot-0.1.0.dist-info/licenses/LICENSE +21 -0
- fluxplot-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +472 -0
|
@@ -0,0 +1,443 @@
|
|
|
1
|
+
"""Tests for three or more groups — the omnibus question ("do the groups differ at all?") and the
|
|
2
|
+
post-hoc pairwise comparisons that follow it — plus :func:`pairwise`, which runs any two-group
|
|
3
|
+
test of this package over every pair of a family and corrects the p-values.
|
|
4
|
+
|
|
5
|
+
Each omnibus test returns one row keyed by :data:`REPORT_COLUMNS` (``groups`` lists every group,
|
|
6
|
+
``n_total`` their combined size, ``dof`` / ``dof_error`` the numerator and denominator dof of an F
|
|
7
|
+
test); each post-hoc test returns one row per pair (``groups = [a, b]``, signs follow ``a - b``),
|
|
8
|
+
ready for :func:`fluxplot.brackets`, which draws them over a plot and records which test each star
|
|
9
|
+
came from.
|
|
10
|
+
|
|
11
|
+
Repeated-measures designs take a long table (``table``, a pandas / polars DataFrame or a dict of
|
|
12
|
+
columns) with a ``subject`` column, a ``within`` (condition) column and a ``dv`` column; every
|
|
13
|
+
subject must have every condition exactly once.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import itertools
|
|
18
|
+
from typing import Any, Callable, Dict, Iterable, List, Optional, Sequence, Tuple, Union
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
from scipy import optimize as _opt
|
|
22
|
+
from scipy import stats as _sp
|
|
23
|
+
|
|
24
|
+
from ._common import REPORT_COLUMNS, as_sample, bh, holm, report_row
|
|
25
|
+
from .two_group import cliffs_delta, hedges_unpooled
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"anova_oneway", "welch_anova", "kruskal_epsilon", "rm_anova", "friedman_kendall",
|
|
29
|
+
"tukey_hsd", "games_howell", "dunn", "pairwise", "REPORT_COLUMNS",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
_Z = _sp.norm.ppf(0.975)
|
|
33
|
+
_BOOT_SEED = 0
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ---------------------------------------------------------------------------------------------
|
|
37
|
+
# inputs
|
|
38
|
+
# ---------------------------------------------------------------------------------------------
|
|
39
|
+
def _groups(groups: Sequence[Any], names: Optional[Sequence[str]], who: str) -> Tuple[List[np.ndarray], List[str]]:
|
|
40
|
+
if len(groups) == 1 and isinstance(groups[0], dict): # a {name: sample} mapping
|
|
41
|
+
names = list(groups[0]) if names is None else list(names)
|
|
42
|
+
groups = list(groups[0].values())
|
|
43
|
+
if len(groups) < 2:
|
|
44
|
+
raise ValueError(f"{who} needs at least 2 groups, got {len(groups)}")
|
|
45
|
+
if names is None:
|
|
46
|
+
names = [f"group{i + 1}" for i in range(len(groups))]
|
|
47
|
+
names = [str(n) for n in names]
|
|
48
|
+
if len(names) != len(groups):
|
|
49
|
+
raise ValueError(f"{who}: names has {len(names)} entries for {len(groups)} groups")
|
|
50
|
+
return [as_sample(g, n) for g, n in zip(groups, names)], names
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _column(table: Any, name: str) -> list:
|
|
54
|
+
try:
|
|
55
|
+
col = table[name]
|
|
56
|
+
except Exception as exc: # KeyError (pandas/dict), ColumnNotFoundError (polars), …
|
|
57
|
+
raise KeyError(f"{name!r} is not a column of the table") from exc
|
|
58
|
+
for attr in ("to_list", "tolist"):
|
|
59
|
+
if hasattr(col, attr):
|
|
60
|
+
return list(getattr(col, attr)())
|
|
61
|
+
return list(col)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _matrix(table: Any, subject: str, within: str, dv: str, who: str) -> Tuple[np.ndarray, List[str], List[Any]]:
|
|
65
|
+
"""The complete ``subjects × conditions`` matrix of a long table: conditions in order of first
|
|
66
|
+
appearance, subjects likewise; every cell filled exactly once."""
|
|
67
|
+
subs, conds, vals = _column(table, subject), _column(table, within), _column(table, dv)
|
|
68
|
+
if not (len(subs) == len(conds) == len(vals)):
|
|
69
|
+
raise ValueError(f"{who}: the columns differ in length")
|
|
70
|
+
sub_order = list(dict.fromkeys(subs))
|
|
71
|
+
cond_order = list(dict.fromkeys(conds))
|
|
72
|
+
n, k = len(sub_order), len(cond_order)
|
|
73
|
+
if k < 2:
|
|
74
|
+
raise ValueError(f"{who} needs at least 2 conditions in {within!r}, got {k}")
|
|
75
|
+
if n < 2:
|
|
76
|
+
raise ValueError(f"{who} needs at least 2 subjects in {subject!r}, got {n}")
|
|
77
|
+
m = np.full((n, k), np.nan)
|
|
78
|
+
si, ci = {s: i for i, s in enumerate(sub_order)}, {c: j for j, c in enumerate(cond_order)}
|
|
79
|
+
seen = set()
|
|
80
|
+
for s, c, v in zip(subs, conds, vals):
|
|
81
|
+
cell = (si[s], ci[c])
|
|
82
|
+
if cell in seen:
|
|
83
|
+
raise ValueError(f"{who}: subject {s!r} has more than one row for {within}={c!r}")
|
|
84
|
+
seen.add(cell)
|
|
85
|
+
m[cell] = float(v) if v is not None else np.nan
|
|
86
|
+
if not np.all(np.isfinite(m)):
|
|
87
|
+
missing = [(sub_order[i], cond_order[j]) for i, j in zip(*np.where(~np.isfinite(m)))]
|
|
88
|
+
raise ValueError(f"{who}: incomplete or non-finite cells for {missing[:3]}{'…' if len(missing) > 3 else ''}; "
|
|
89
|
+
"every subject needs a finite value for every condition")
|
|
90
|
+
return m, [str(c) for c in cond_order], sub_order
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# ---------------------------------------------------------------------------------------------
|
|
94
|
+
# noncentral-F confidence intervals for variance-explained effect sizes (Steiger 2004)
|
|
95
|
+
# ---------------------------------------------------------------------------------------------
|
|
96
|
+
def _ncf_ncp_for(f_obs: float, df1: float, df2: float, target: float) -> float:
|
|
97
|
+
"""The noncentrality ``lam >= 0`` with ``ncf.cdf(f_obs, df1, df2, lam) == target``, or 0 when
|
|
98
|
+
even the central distribution puts less than ``target`` below ``f_obs``."""
|
|
99
|
+
f = lambda lam: _sp.ncf.cdf(f_obs, df1, df2, lam) - target # noqa: E731
|
|
100
|
+
if f(0.0) <= 0:
|
|
101
|
+
return 0.0
|
|
102
|
+
hi = max(1.0, f_obs * df1)
|
|
103
|
+
while f(hi) > 0:
|
|
104
|
+
hi *= 2
|
|
105
|
+
return _opt.brentq(f, 0.0, hi, xtol=1e-10)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _variance_explained_ci(f_obs: float, df1: float, df2: float) -> Tuple[float, float]:
|
|
109
|
+
"""95% CI for the population proportion of variance explained (``eta^2`` / ``omega^2`` /
|
|
110
|
+
partial ``eta^2`` estimate the same parameter) from the noncentral F distribution of the
|
|
111
|
+
observed F: the bounds are ``lam / (lam + df1 + df2 + 1)`` for the noncentralities bracketing
|
|
112
|
+
``f_obs`` at the 2.5% and 97.5% points (Steiger 2004; Kelley 2007). The lower bound is 0 when
|
|
113
|
+
``p >= 0.025``."""
|
|
114
|
+
lo = _ncf_ncp_for(f_obs, df1, df2, 0.975)
|
|
115
|
+
hi = _ncf_ncp_for(f_obs, df1, df2, 0.025)
|
|
116
|
+
n_eff = df1 + df2 + 1
|
|
117
|
+
return lo / (lo + n_eff), hi / (hi + n_eff)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _bootstrap_ci(statistic: Callable[[np.random.Generator], float], n_boot: int) -> Tuple[float, float]:
|
|
121
|
+
"""A seeded percentile-bootstrap 95% interval (deterministic across runs)."""
|
|
122
|
+
rng = np.random.default_rng(_BOOT_SEED)
|
|
123
|
+
draws = np.array([statistic(rng) for _ in range(int(n_boot))])
|
|
124
|
+
draws = draws[np.isfinite(draws)]
|
|
125
|
+
if draws.size == 0:
|
|
126
|
+
return float("nan"), float("nan")
|
|
127
|
+
lo, hi = np.percentile(draws, [2.5, 97.5])
|
|
128
|
+
return float(lo), float(hi)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# ---------------------------------------------------------------------------------------------
|
|
132
|
+
# omnibus tests, independent groups
|
|
133
|
+
# ---------------------------------------------------------------------------------------------
|
|
134
|
+
def _sums_of_squares(groups: List[np.ndarray]):
|
|
135
|
+
allv = np.concatenate(groups)
|
|
136
|
+
grand = allv.mean()
|
|
137
|
+
ss_b = sum(g.size * (g.mean() - grand) ** 2 for g in groups)
|
|
138
|
+
ss_w = sum(((g - g.mean()) ** 2).sum() for g in groups)
|
|
139
|
+
return ss_b, ss_w, allv.size
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def anova_oneway(*groups: Any, names: Optional[Sequence[str]] = None, effect: str = "eta2") -> Dict[str, Any]:
|
|
143
|
+
"""Classic one-way ANOVA (equal variances assumed) with η² or ω² and its 95% CI.
|
|
144
|
+
|
|
145
|
+
``F = MS_between / MS_within`` on ``(k - 1, N - k)`` dof (scipy's ``f_oneway``). The effect
|
|
146
|
+
size is η² = SS_between / SS_total (``effect="eta2"``, the default) or the less biased
|
|
147
|
+
ω² = (SS_between − (k − 1) MS_within) / (SS_total + MS_within) (``effect="omega2"``). Both
|
|
148
|
+
estimate the population proportion of variance explained, so both get the same 95% CI: the
|
|
149
|
+
noncentral-F interval of Steiger (2004), whose lower bound is 0 whenever ``p >= 0.025``.
|
|
150
|
+
|
|
151
|
+
Parameters
|
|
152
|
+
----------
|
|
153
|
+
*groups
|
|
154
|
+
Two or more samples, or a single ``{name: sample}`` dict.
|
|
155
|
+
names
|
|
156
|
+
Group names (recorded in ``groups``); default ``group1, group2, …``.
|
|
157
|
+
effect
|
|
158
|
+
``"eta2"`` or ``"omega2"``.
|
|
159
|
+
"""
|
|
160
|
+
gs, names = _groups(groups, names, "anova_oneway")
|
|
161
|
+
if effect not in ("eta2", "omega2"):
|
|
162
|
+
raise ValueError(f"effect must be 'eta2' or 'omega2', got {effect!r}")
|
|
163
|
+
k = len(gs)
|
|
164
|
+
ss_b, ss_w, n = _sums_of_squares(gs)
|
|
165
|
+
df1, df2 = k - 1, n - k
|
|
166
|
+
if ss_w == 0:
|
|
167
|
+
raise ValueError("every group has zero within-group variance; F is undefined")
|
|
168
|
+
res = _sp.f_oneway(*gs)
|
|
169
|
+
ms_w = ss_w / df2
|
|
170
|
+
eta2 = ss_b / (ss_b + ss_w)
|
|
171
|
+
omega2 = (ss_b - df1 * ms_w) / (ss_b + ss_w + ms_w)
|
|
172
|
+
lo, hi = _variance_explained_ci(res.statistic, df1, df2)
|
|
173
|
+
value, method = (eta2, "Eta squared") if effect == "eta2" else (omega2, "Omega squared")
|
|
174
|
+
return report_row("One-way ANOVA", res.statistic, res.pvalue, df1, method, value, lo, hi,
|
|
175
|
+
n=[g.size for g in gs], groups=names, dof_error=df2)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def welch_anova(*groups: Any, names: Optional[Sequence[str]] = None) -> Dict[str, Any]:
|
|
179
|
+
"""Welch's ANOVA (unequal variances) with ω² and its 95% CI.
|
|
180
|
+
|
|
181
|
+
Welch's (1951) F weights each group by ``n_i / var_i``; the denominator dof are
|
|
182
|
+
``(k² − 1) / (3 Σ (1 − w_i / W)² / (n_i − 1))``. The effect size is
|
|
183
|
+
ω² = (k − 1)(F − 1) / ((k − 1)(F − 1) + N) (Kirk's formula; identical to the classic ω² when
|
|
184
|
+
the variances are equal), with the noncentral-F 95% CI on Welch's dof.
|
|
185
|
+
"""
|
|
186
|
+
gs, names = _groups(groups, names, "welch_anova")
|
|
187
|
+
k = len(gs)
|
|
188
|
+
n = np.array([g.size for g in gs], dtype=float)
|
|
189
|
+
m = np.array([g.mean() for g in gs])
|
|
190
|
+
v = np.array([g.var(ddof=1) for g in gs])
|
|
191
|
+
if np.any(v == 0):
|
|
192
|
+
raise ValueError("a group has zero variance; Welch's weights are undefined")
|
|
193
|
+
w = n / v
|
|
194
|
+
big_w = w.sum()
|
|
195
|
+
grand = (w * m).sum() / big_w
|
|
196
|
+
lam = ((1 - w / big_w) ** 2 / (n - 1)).sum()
|
|
197
|
+
f_stat = ((w * (m - grand) ** 2).sum() / (k - 1)) / (1 + 2 * (k - 2) / (k**2 - 1) * lam)
|
|
198
|
+
df1 = k - 1
|
|
199
|
+
df2 = (k**2 - 1) / (3 * lam)
|
|
200
|
+
p = _sp.f.sf(f_stat, df1, df2)
|
|
201
|
+
total = int(n.sum())
|
|
202
|
+
omega2 = df1 * (f_stat - 1) / (df1 * (f_stat - 1) + total)
|
|
203
|
+
lo, hi = _variance_explained_ci(f_stat, df1, df2)
|
|
204
|
+
return report_row("Welch's ANOVA", f_stat, p, df1, "Omega squared", omega2, lo, hi,
|
|
205
|
+
n=[g.size for g in gs], groups=names, dof_error=df2)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def kruskal_epsilon(*groups: Any, names: Optional[Sequence[str]] = None, n_boot: int = 2000) -> Dict[str, Any]:
|
|
209
|
+
"""Kruskal–Wallis H test with the ε² effect size and a bootstrap 95% CI.
|
|
210
|
+
|
|
211
|
+
``H`` is scipy's tie-corrected statistic on ``k − 1`` dof. ε² = H / (N − 1) (Kelley 1935; the
|
|
212
|
+
rank analogue of η², in ``[0, 1]``). Its 95% CI is a seeded percentile bootstrap of ε² over
|
|
213
|
+
``n_boot`` within-group resamples — deterministic across runs.
|
|
214
|
+
"""
|
|
215
|
+
gs, names = _groups(groups, names, "kruskal_epsilon")
|
|
216
|
+
res = _sp.kruskal(*gs)
|
|
217
|
+
total = sum(g.size for g in gs)
|
|
218
|
+
eps2 = res.statistic / (total - 1)
|
|
219
|
+
|
|
220
|
+
def draw(rng):
|
|
221
|
+
boot = [rng.choice(g, g.size, replace=True) for g in gs]
|
|
222
|
+
try:
|
|
223
|
+
return _sp.kruskal(*boot).statistic / (total - 1)
|
|
224
|
+
except ValueError: # every resampled value identical
|
|
225
|
+
return np.nan
|
|
226
|
+
|
|
227
|
+
lo, hi = _bootstrap_ci(draw, n_boot)
|
|
228
|
+
return report_row("Kruskal–Wallis H test", res.statistic, res.pvalue, len(gs) - 1,
|
|
229
|
+
"Epsilon squared", eps2, lo, hi, n=[g.size for g in gs], groups=names)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
# ---------------------------------------------------------------------------------------------
|
|
233
|
+
# omnibus tests, repeated measures
|
|
234
|
+
# ---------------------------------------------------------------------------------------------
|
|
235
|
+
def rm_anova(table: Any, subject: str, within: str, dv: str) -> Dict[str, Any]:
|
|
236
|
+
"""One-way repeated-measures ANOVA with the Greenhouse–Geisser correction and partial η².
|
|
237
|
+
|
|
238
|
+
The within-subject decomposition ``SS_total = SS_subjects + SS_conditions + SS_error`` gives
|
|
239
|
+
``F = MS_conditions / MS_error`` on ``(k − 1, (k − 1)(n − 1))`` dof. Sphericity is not assumed:
|
|
240
|
+
both dof are multiplied by the Greenhouse–Geisser ε (from the double-centred covariance matrix
|
|
241
|
+
of the conditions, clamped to ``[1 / (k − 1), 1]``) before the p-value is taken, and the row
|
|
242
|
+
reports those corrected dof. Partial η² = SS_conditions / (SS_conditions + SS_error), with the
|
|
243
|
+
noncentral-F 95% CI on the corrected dof.
|
|
244
|
+
|
|
245
|
+
``groups`` lists the conditions; ``n_total`` is the number of subjects.
|
|
246
|
+
"""
|
|
247
|
+
m, conds, subs = _matrix(table, subject, within, dv, "rm_anova")
|
|
248
|
+
n, k = m.shape
|
|
249
|
+
grand = m.mean()
|
|
250
|
+
ss_total = ((m - grand) ** 2).sum()
|
|
251
|
+
ss_subj = k * ((m.mean(axis=1) - grand) ** 2).sum()
|
|
252
|
+
ss_cond = n * ((m.mean(axis=0) - grand) ** 2).sum()
|
|
253
|
+
ss_err = ss_total - ss_subj - ss_cond
|
|
254
|
+
df1, df2 = k - 1, (k - 1) * (n - 1)
|
|
255
|
+
if ss_err <= 0:
|
|
256
|
+
raise ValueError("rm_anova: the error sum of squares is zero; F is undefined")
|
|
257
|
+
f_stat = (ss_cond / df1) / (ss_err / df2)
|
|
258
|
+
# Greenhouse–Geisser epsilon from the double-centred covariance matrix
|
|
259
|
+
cov = np.cov(m, rowvar=False, ddof=1)
|
|
260
|
+
centre = np.eye(k) - np.ones((k, k)) / k
|
|
261
|
+
dc = centre @ cov @ centre
|
|
262
|
+
eps = np.trace(dc) ** 2 / ((k - 1) * (dc**2).sum())
|
|
263
|
+
eps = float(min(1.0, max(eps, 1.0 / (k - 1))))
|
|
264
|
+
gdf1, gdf2 = eps * df1, eps * df2
|
|
265
|
+
p = _sp.f.sf(f_stat, gdf1, gdf2)
|
|
266
|
+
eta_p = ss_cond / (ss_cond + ss_err)
|
|
267
|
+
lo, hi = _variance_explained_ci(f_stat, gdf1, gdf2)
|
|
268
|
+
return report_row("Repeated-measures ANOVA (Greenhouse–Geisser)", f_stat, p, gdf1,
|
|
269
|
+
"Partial eta squared", eta_p, lo, hi, n=[n] * k, n_total=n, groups=conds, dof_error=gdf2)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def friedman_kendall(table: Any, subject: str, within: str, dv: str, n_boot: int = 2000) -> Dict[str, Any]:
|
|
273
|
+
"""Friedman's test with Kendall's W and a bootstrap 95% CI.
|
|
274
|
+
|
|
275
|
+
scipy's ``friedmanchisquare`` on the ``subjects × conditions`` matrix gives the χ² statistic on
|
|
276
|
+
``k − 1`` dof. Kendall's W = χ² / (n (k − 1)) is the agreement between subjects about the
|
|
277
|
+
ordering of the conditions, in ``[0, 1]``. Its 95% CI is a seeded percentile bootstrap over
|
|
278
|
+
subjects (rows resampled with replacement, ``n_boot`` draws).
|
|
279
|
+
|
|
280
|
+
``groups`` lists the conditions; ``n_total`` is the number of subjects.
|
|
281
|
+
"""
|
|
282
|
+
m, conds, subs = _matrix(table, subject, within, dv, "friedman_kendall")
|
|
283
|
+
n, k = m.shape
|
|
284
|
+
if k < 3:
|
|
285
|
+
raise ValueError("friedman_kendall needs at least 3 conditions (use a paired test for 2)")
|
|
286
|
+
res = _sp.friedmanchisquare(*m.T)
|
|
287
|
+
w = res.statistic / (n * (k - 1))
|
|
288
|
+
|
|
289
|
+
def draw(rng):
|
|
290
|
+
rows = m[rng.integers(0, n, n)]
|
|
291
|
+
try:
|
|
292
|
+
return _sp.friedmanchisquare(*rows.T).statistic / (n * (k - 1))
|
|
293
|
+
except ValueError:
|
|
294
|
+
return np.nan
|
|
295
|
+
|
|
296
|
+
lo, hi = _bootstrap_ci(draw, n_boot)
|
|
297
|
+
return report_row("Friedman test", res.statistic, res.pvalue, k - 1, "Kendall's W", w, lo, hi,
|
|
298
|
+
n=[n] * k, n_total=n, groups=conds)
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
# ---------------------------------------------------------------------------------------------
|
|
302
|
+
# post-hoc pairwise comparisons
|
|
303
|
+
# ---------------------------------------------------------------------------------------------
|
|
304
|
+
def _hedges_pooled(a: np.ndarray, b: np.ndarray) -> Tuple[float, float, float]:
|
|
305
|
+
"""Hedges' g with the pooled SD (equal variances) and its large-sample 95% CI (Hedges & Olkin
|
|
306
|
+
1985): ``(g, lo, hi)``."""
|
|
307
|
+
n1, n2 = a.size, b.size
|
|
308
|
+
sp = np.sqrt(((n1 - 1) * a.var(ddof=1) + (n2 - 1) * b.var(ddof=1)) / (n1 + n2 - 2))
|
|
309
|
+
if sp == 0:
|
|
310
|
+
raise ValueError("both samples have zero variance; the effect size is undefined")
|
|
311
|
+
d = (a.mean() - b.mean()) / sp
|
|
312
|
+
j = 1 - 3 / (4 * (n1 + n2 - 2) - 1)
|
|
313
|
+
g = j * d
|
|
314
|
+
se = np.sqrt((n1 + n2) / (n1 * n2) + g**2 / (2 * (n1 + n2)))
|
|
315
|
+
return float(g), float(g - _Z * se), float(g + _Z * se)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _pairs(k: int) -> List[Tuple[int, int]]:
|
|
319
|
+
return list(itertools.combinations(range(k), 2))
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def tukey_hsd(*groups: Any, names: Optional[Sequence[str]] = None) -> List[Dict[str, Any]]:
|
|
323
|
+
"""Tukey's honestly-significant-difference test over every pair, with Hedges' g (pooled SD).
|
|
324
|
+
|
|
325
|
+
scipy's ``tukey_hsd`` gives each pair's mean difference ``a − b`` and its p-value from the
|
|
326
|
+
studentized range on ``N − k`` dof. That p-value already controls the family-wise error rate,
|
|
327
|
+
so ``p_corrected_holm`` and ``p_corrected_bh`` equal it: do **not** correct these rows again.
|
|
328
|
+
The effect size is Hedges' g with the pooled SD (Tukey assumes equal variances), with the
|
|
329
|
+
Hedges–Olkin large-sample 95% CI.
|
|
330
|
+
"""
|
|
331
|
+
gs, names = _groups(groups, names, "tukey_hsd")
|
|
332
|
+
res = _sp.tukey_hsd(*gs)
|
|
333
|
+
total = sum(g.size for g in gs)
|
|
334
|
+
rows = []
|
|
335
|
+
for i, j in _pairs(len(gs)):
|
|
336
|
+
g, lo, hi = _hedges_pooled(gs[i], gs[j])
|
|
337
|
+
rows.append(report_row("Tukey HSD", res.statistic[i, j], res.pvalue[i, j], total - len(gs),
|
|
338
|
+
"Hedges' g (pooled SD)", g, lo, hi, n=(gs[i].size, gs[j].size),
|
|
339
|
+
groups=(names[i], names[j])))
|
|
340
|
+
return rows
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def games_howell(*groups: Any, names: Optional[Sequence[str]] = None) -> List[Dict[str, Any]]:
|
|
344
|
+
"""The Games–Howell test over every pair (unequal variances and sizes), with Hedges' g
|
|
345
|
+
(non-pooled SD).
|
|
346
|
+
|
|
347
|
+
Each pair's ``t = (mean_a − mean_b) / sqrt(var_a / n_a + var_b / n_b)`` on Welch–Satterthwaite
|
|
348
|
+
dof is referred to the studentized range distribution of ``k`` groups
|
|
349
|
+
(``p = P(Q_{k, df} > |t| sqrt 2)``), which controls the family-wise error rate: as for
|
|
350
|
+
:func:`tukey_hsd`, the corrected columns equal ``p-value``. The effect size and its Bonett CI
|
|
351
|
+
are those of :func:`fluxplot.stats.welch_hedges`.
|
|
352
|
+
"""
|
|
353
|
+
gs, names = _groups(groups, names, "games_howell")
|
|
354
|
+
k = len(gs)
|
|
355
|
+
rows = []
|
|
356
|
+
for i, j in _pairs(k):
|
|
357
|
+
a, b = gs[i], gs[j]
|
|
358
|
+
va, vb = a.var(ddof=1) / a.size, b.var(ddof=1) / b.size
|
|
359
|
+
if va + vb == 0:
|
|
360
|
+
raise ValueError(f"groups {names[i]!r} and {names[j]!r} have zero variance")
|
|
361
|
+
t = (a.mean() - b.mean()) / np.sqrt(va + vb)
|
|
362
|
+
df = (va + vb) ** 2 / (va**2 / (a.size - 1) + vb**2 / (b.size - 1))
|
|
363
|
+
p = _sp.studentized_range.sf(abs(t) * np.sqrt(2), k, df)
|
|
364
|
+
g, lo, hi = hedges_unpooled(a, b)
|
|
365
|
+
rows.append(report_row("Games–Howell test", t, p, df, "Hedges' g (non-pooled SD)", g, lo, hi,
|
|
366
|
+
n=(a.size, b.size), groups=(names[i], names[j])))
|
|
367
|
+
return rows
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def dunn(*groups: Any, names: Optional[Sequence[str]] = None, adjust: Union[str, Sequence[str]] = "holm") -> List[Dict[str, Any]]:
|
|
371
|
+
"""Dunn's (1964) rank-sum test over every pair after Kruskal–Wallis, with Cliff's delta.
|
|
372
|
+
|
|
373
|
+
The pooled ranks give each pair ``z = (R̄_a − R̄_b) / sqrt((N (N + 1) / 12 − T) (1 / n_a + 1 / n_b))``
|
|
374
|
+
with the tie term ``T = Σ (t³ − t) / (12 (N − 1))``; ``p-value`` is the raw two-sided normal
|
|
375
|
+
p, and ``adjust`` (``"holm"``, ``"bh"``, or both) fills the matching corrected column across
|
|
376
|
+
the pairs. The effect size is Cliff's delta with Newcombe's 95% CI, as in
|
|
377
|
+
:func:`fluxplot.stats.mann_whitney_cliff`.
|
|
378
|
+
"""
|
|
379
|
+
gs, names = _groups(groups, names, "dunn")
|
|
380
|
+
pooled = np.concatenate(gs)
|
|
381
|
+
ranks = _sp.rankdata(pooled)
|
|
382
|
+
total = pooled.size
|
|
383
|
+
_, counts = np.unique(pooled, return_counts=True)
|
|
384
|
+
tie_term = float(((counts**3 - counts).sum()) / (12 * (total - 1)))
|
|
385
|
+
bounds = np.cumsum([0] + [g.size for g in gs])
|
|
386
|
+
mean_ranks = [ranks[bounds[i]:bounds[i + 1]].mean() for i in range(len(gs))]
|
|
387
|
+
rows = []
|
|
388
|
+
for i, j in _pairs(len(gs)):
|
|
389
|
+
a, b = gs[i], gs[j]
|
|
390
|
+
se = np.sqrt((total * (total + 1) / 12 - tie_term) * (1 / a.size + 1 / b.size))
|
|
391
|
+
z = (mean_ranks[i] - mean_ranks[j]) / se
|
|
392
|
+
p = 2 * _sp.norm.sf(abs(z))
|
|
393
|
+
delta, lo, hi = cliffs_delta(a, b)
|
|
394
|
+
rows.append(report_row("Dunn's test", z, p, None, "Cliff's delta", delta, lo, hi,
|
|
395
|
+
n=(a.size, b.size), groups=(names[i], names[j])))
|
|
396
|
+
return _adjust(rows, adjust)
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def _adjust(rows: List[dict], adjust: Union[str, Sequence[str], None]) -> List[dict]:
|
|
400
|
+
if adjust is None:
|
|
401
|
+
return rows
|
|
402
|
+
kinds = (adjust,) if isinstance(adjust, str) else tuple(adjust)
|
|
403
|
+
for kind in kinds:
|
|
404
|
+
if kind == "holm":
|
|
405
|
+
rows = holm(rows)
|
|
406
|
+
elif kind == "bh":
|
|
407
|
+
rows = bh(rows)
|
|
408
|
+
else:
|
|
409
|
+
raise ValueError(f"adjust must be 'holm', 'bh' or a sequence of them, got {kind!r}")
|
|
410
|
+
return rows
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def pairwise(test: Callable[..., dict], groups: Dict[str, Any], pairs: Optional[Iterable[Tuple[str, str]]] = None,
|
|
414
|
+
adjust: Union[str, Sequence[str], None] = "holm") -> List[Dict[str, Any]]:
|
|
415
|
+
"""Run a two-group test of this package over pairs of named groups and correct the p-values.
|
|
416
|
+
|
|
417
|
+
>>> rows = fp.stats.pairwise(fp.stats.welch_hedges, {"ctl": ctl, "drug": drug, "sham": sham})
|
|
418
|
+
>>> fp.brackets(ax, rows, positions=gb.positions)
|
|
419
|
+
|
|
420
|
+
Parameters
|
|
421
|
+
----------
|
|
422
|
+
test
|
|
423
|
+
``welch_hedges``, ``mann_whitney_cliff``, ``paired_t_hedges`` or ``wilcoxon_rank_biserial``
|
|
424
|
+
(any callable ``test(a, b, names=(a_name, b_name)) -> row``).
|
|
425
|
+
groups
|
|
426
|
+
``{name: sample}``; the order fixes the default pairs and the sign convention (``a − b``
|
|
427
|
+
with ``a`` the earlier group).
|
|
428
|
+
pairs
|
|
429
|
+
The ``(a, b)`` name pairs to compare; default every pair in ``groups`` order.
|
|
430
|
+
adjust
|
|
431
|
+
``"holm"`` (default), ``"bh"``, a sequence of both, or ``None``: the correction(s) filled
|
|
432
|
+
in across the family (``p_corrected_holm`` / ``p_corrected_bh``).
|
|
433
|
+
"""
|
|
434
|
+
names = list(groups)
|
|
435
|
+
if pairs is None:
|
|
436
|
+
pairs = list(itertools.combinations(names, 2))
|
|
437
|
+
rows = []
|
|
438
|
+
for a, b in pairs:
|
|
439
|
+
for name in (a, b):
|
|
440
|
+
if name not in groups:
|
|
441
|
+
raise KeyError(f"pairwise: {name!r} is not one of the groups {names}")
|
|
442
|
+
rows.append(test(groups[a], groups[b], names=(a, b)))
|
|
443
|
+
return _adjust(rows, adjust)
|
fluxplot/stats/paired.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Tests for paired samples: matched observations where ``a[i]`` and ``b[i]`` belong together
|
|
2
|
+
(the same animal before/after, the same cell under two conditions).
|
|
3
|
+
|
|
4
|
+
Each function returns one row as a dict keyed by :data:`REPORT_COLUMNS`, like the independent-sample
|
|
5
|
+
tests in :mod:`fluxplot.stats.two_group`. Differences are always taken as ``a - b``.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import Any, Dict, Tuple
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
from scipy import optimize as _opt
|
|
13
|
+
from scipy import special as _sc
|
|
14
|
+
from scipy import stats as _sp
|
|
15
|
+
|
|
16
|
+
from ._common import REPORT_COLUMNS, as_pairs, check_alternative, report_row
|
|
17
|
+
|
|
18
|
+
__all__ = ["paired_t_hedges", "wilcoxon_rank_biserial", "REPORT_COLUMNS"]
|
|
19
|
+
|
|
20
|
+
_Z = _sp.norm.ppf(0.975)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _nct_ncp_for(t_obs: float, df: float, target: float) -> float:
|
|
24
|
+
"""The noncentrality ``ncp`` with ``nct.cdf(t_obs, df, ncp) == target``.
|
|
25
|
+
|
|
26
|
+
The cdf falls monotonically as ``ncp`` grows, so the root is bracketed by stepping outward from
|
|
27
|
+
``t_obs`` (where the cdf is near 0.5) in doubling steps, then solved with Brent's method.
|
|
28
|
+
"""
|
|
29
|
+
f = lambda ncp: _sp.nct.cdf(t_obs, df, ncp) - target # noqa: E731
|
|
30
|
+
step = 1.0
|
|
31
|
+
if target > 0.5: # the root lies below t_obs
|
|
32
|
+
lo, hi = t_obs - step, t_obs
|
|
33
|
+
while f(lo) < 0:
|
|
34
|
+
step *= 2
|
|
35
|
+
lo, hi = t_obs - step, lo
|
|
36
|
+
else: # the root lies above t_obs
|
|
37
|
+
lo, hi = t_obs, t_obs + step
|
|
38
|
+
while f(hi) > 0:
|
|
39
|
+
step *= 2
|
|
40
|
+
lo, hi = hi, t_obs + step
|
|
41
|
+
return _opt.brentq(f, lo, hi, xtol=1e-12)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def paired_t_hedges(a: Any, b: Any, *, alternative: str = "two-sided", names=("a", "b")) -> Dict[str, Any]:
|
|
45
|
+
"""Paired t-test of ``a`` vs ``b`` with Hedges' g_z and its 95% CI.
|
|
46
|
+
|
|
47
|
+
For matched observations, asking whether the mean of the paired differences ``a - b`` is
|
|
48
|
+
different from zero. The effect size is ``d_z = mean(a - b) / sd(a - b)`` (standardized by the
|
|
49
|
+
SD of the paired differences) times the small-sample correction ``J = 1 - 3 / (4 (n - 1) - 1)``.
|
|
50
|
+
The 95% CI is exact under normality: the noncentral-t interval for the population ``d_z``,
|
|
51
|
+
inverting the noncentral t distribution of the observed t with ``n - 1`` degrees of freedom.
|
|
52
|
+
It is not multiplied by ``J``: that would break its exactness (at n = 6 and ``d_z = 2``
|
|
53
|
+
coverage would fall to about 90%).
|
|
54
|
+
|
|
55
|
+
Signs follow ``a - b``: a positive statistic and effect size mean ``a`` is larger on average.
|
|
56
|
+
|
|
57
|
+
Parameters
|
|
58
|
+
----------
|
|
59
|
+
a, b
|
|
60
|
+
The paired samples, equal length, ``a[i]`` matched with ``b[i]``; at least 2 pairs of finite
|
|
61
|
+
values.
|
|
62
|
+
alternative
|
|
63
|
+
``"two-sided"`` (default), ``"less"`` or ``"greater"`` for the mean difference ``a - b``.
|
|
64
|
+
The CI is always two-sided.
|
|
65
|
+
names
|
|
66
|
+
The names of the two conditions, recorded in the row's ``groups``.
|
|
67
|
+
|
|
68
|
+
Returns
|
|
69
|
+
-------
|
|
70
|
+
dict
|
|
71
|
+
One reporting row keyed by :data:`REPORT_COLUMNS`: the t statistic, the p-value,
|
|
72
|
+
``dof = n - 1``, Hedges' g_z and its 95% CI. ``n_a = n_b = n_total = n``, the number of pairs.
|
|
73
|
+
"""
|
|
74
|
+
a, b = as_pairs(a, b)
|
|
75
|
+
diff = a - b
|
|
76
|
+
n = diff.size
|
|
77
|
+
sd = diff.std(ddof=1)
|
|
78
|
+
if sd == 0:
|
|
79
|
+
raise ValueError("the paired differences have zero variance; the effect size is undefined")
|
|
80
|
+
|
|
81
|
+
t = _sp.ttest_rel(a, b, alternative=check_alternative(alternative))
|
|
82
|
+
df = n - 1
|
|
83
|
+
d_z = diff.mean() / sd
|
|
84
|
+
j = 1 - 3 / (4 * df - 1)
|
|
85
|
+
t_obs = d_z * np.sqrt(n) # identical to t.statistic
|
|
86
|
+
lo = _nct_ncp_for(t_obs, df, 0.975) / np.sqrt(n)
|
|
87
|
+
hi = _nct_ncp_for(t_obs, df, 0.025) / np.sqrt(n)
|
|
88
|
+
return report_row("Paired t-test", t.statistic, t.pvalue, df, "Hedges' g_z", j * d_z, lo, hi,
|
|
89
|
+
n=(n, n), n_total=n, groups=names, alternative=alternative)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _bvn_cdf_same_sign(h: float, k: float, rho: float) -> float:
|
|
93
|
+
"""Standard bivariate-normal ``P(X < h, Y < k)`` with correlation ``rho``, for ``h * k > 0``
|
|
94
|
+
or ``h = k = 0`` (all this module needs), via Owen's (1956) T-function identity."""
|
|
95
|
+
if h == 0 and k == 0:
|
|
96
|
+
return 0.25 + np.arcsin(rho) / (2 * np.pi)
|
|
97
|
+
s = np.sqrt(1 - rho**2)
|
|
98
|
+
return (0.5 * (_sp.norm.cdf(h) + _sp.norm.cdf(k))
|
|
99
|
+
- _sc.owens_t(h, (k - rho * h) / (h * s)) - _sc.owens_t(k, (h - rho * k) / (k * s)))
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _rank_biserial_moments(delta: float, n: int) -> Tuple[float, float]:
|
|
103
|
+
"""Mean and variance of the matched-pairs rank-biserial ``r`` from ``n`` differences
|
|
104
|
+
``D ~ Normal(delta, 1)``.
|
|
105
|
+
|
|
106
|
+
``W+ = sum_{i<=j} 1[D_i + D_j > 0]`` (Walsh averages), so with ``p1 = P(D > 0)``,
|
|
107
|
+
``p2 = P(D1 + D2 > 0)``, ``p3 = P(D1 + D2 > 0, D1 + D3 > 0)`` and ``p4 = P(D1 > 0, D1 + D2 > 0)``:
|
|
108
|
+
``E W+ = n p1 + C(n, 2) p2`` and
|
|
109
|
+
``Var W+ = n p1 (1 - p1) + C(n, 2) p2 (1 - p2) + n (n-1) (n-2) (p3 - p2^2)
|
|
110
|
+
+ 2 n (n-1) (p4 - p1 p2)`` — at ``delta = 0`` this is the familiar ``n (n+1) (2n+1) / 24``.
|
|
111
|
+
"""
|
|
112
|
+
r2 = np.sqrt(2)
|
|
113
|
+
p1 = _sp.norm.cdf(delta)
|
|
114
|
+
p2 = _sp.norm.cdf(r2 * delta)
|
|
115
|
+
p3 = _bvn_cdf_same_sign(r2 * delta, r2 * delta, 0.5)
|
|
116
|
+
p4 = _bvn_cdf_same_sign(delta, r2 * delta, 1 / r2)
|
|
117
|
+
total = n * (n + 1) / 2
|
|
118
|
+
mean_w = n * p1 + n * (n - 1) / 2 * p2
|
|
119
|
+
var_w = (n * p1 * (1 - p1) + n * (n - 1) / 2 * p2 * (1 - p2)
|
|
120
|
+
+ n * (n - 1) * (n - 2) * (p3 - p2**2) + 2 * n * (n - 1) * (p4 - p1 * p2))
|
|
121
|
+
return 2 * mean_w / total - 1, 4 * max(var_w, 0.0) / total**2
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _rank_biserial_bounds(r: float, n: int) -> Tuple[float, float]:
|
|
125
|
+
"""Score-type 95% interval for the population rank-biserial ``rho = 2 P(D1 + D2 > 0) - 1``.
|
|
126
|
+
|
|
127
|
+
Keeps every ``delta`` whose model mean of ``r`` lies within ``z`` model SDs of the observed
|
|
128
|
+
``r`` — the bounds solve ``(r - E[r | delta])^2 = z^2 Var(r | delta)`` — and maps them to
|
|
129
|
+
``rho = 2 Phi(sqrt(2) delta) - 1``. Because the variance is evaluated at each candidate
|
|
130
|
+
rather than at the estimate, the interval stays inside ``[-1, 1]`` and does not collapse
|
|
131
|
+
when every difference has the same sign (``r = ±1``).
|
|
132
|
+
"""
|
|
133
|
+
lim = 8.0 # |delta| = 8 puts rho within 1e-15 of ±1
|
|
134
|
+
|
|
135
|
+
def mean_minus_r(d: float) -> float:
|
|
136
|
+
return _rank_biserial_moments(d, n)[0] - r
|
|
137
|
+
|
|
138
|
+
def gap(d: float) -> float:
|
|
139
|
+
mean, var = _rank_biserial_moments(d, n)
|
|
140
|
+
return (r - mean) ** 2 - _Z**2 * var
|
|
141
|
+
|
|
142
|
+
if r >= 1:
|
|
143
|
+
d_hat = lim
|
|
144
|
+
elif r <= -1:
|
|
145
|
+
d_hat = -lim
|
|
146
|
+
else:
|
|
147
|
+
d_hat = _opt.brentq(mean_minus_r, -lim, lim, xtol=1e-12)
|
|
148
|
+
d_lo = -lim if r <= -1 else _opt.brentq(gap, -lim, d_hat, xtol=1e-12)
|
|
149
|
+
d_hi = lim if r >= 1 else _opt.brentq(gap, d_hat, lim, xtol=1e-12)
|
|
150
|
+
rho = lambda d: 2 * _sp.norm.cdf(np.sqrt(2) * d) - 1 # noqa: E731
|
|
151
|
+
return rho(d_lo), rho(d_hi)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def wilcoxon_rank_biserial(a: Any, b: Any, *, alternative: str = "two-sided", names=("a", "b")) -> Dict[str, Any]:
|
|
155
|
+
"""Wilcoxon signed-rank test of ``a`` vs ``b`` with the matched-pairs rank-biserial correlation.
|
|
156
|
+
|
|
157
|
+
For matched observations compared by the ranks of their differences ``a - b`` (chosen before
|
|
158
|
+
looking at the results; assumes the differences are meaningful in size and symmetrically
|
|
159
|
+
distributed). Zero differences are dropped (Wilcoxon's convention, scipy's
|
|
160
|
+
``zero_method="wilcox"``) and tied ``|a - b|`` share their average rank. The p-value is scipy's
|
|
161
|
+
two-sided ``wilcoxon`` (``method="auto"``: exact for small samples, otherwise the normal
|
|
162
|
+
approximation).
|
|
163
|
+
|
|
164
|
+
The effect size is Kerby's (2014) matched-pairs rank-biserial correlation
|
|
165
|
+
``r = (W+ - W-) / (W+ + W-)`` over the ``n`` non-zero differences: the share of the rank sum
|
|
166
|
+
favouring ``a > b`` minus the share favouring ``a < b``, in ``[-1, 1]``.
|
|
167
|
+
|
|
168
|
+
Its 95% CI is a score-type interval (the construction of Wilson's and Newcombe's intervals)
|
|
169
|
+
for the population value ``rho = 2 P(D_i + D_j > 0) - 1``: it inverts the normal approximation
|
|
170
|
+
to ``W+`` using ``W+``'s exact mean and variance, evaluated under a normal working model for the
|
|
171
|
+
differences. It stays inside ``[-1, 1]`` and does not collapse when every difference has the
|
|
172
|
+
same sign. In simulation (n = 6–20; normal, t3, Laplace and uniform differences) its coverage
|
|
173
|
+
is 93–97%. The Fisher-z interval of the R ``effectsize`` package, by contrast, covers as little
|
|
174
|
+
as 43–84% at n = 6.
|
|
175
|
+
|
|
176
|
+
Parameters
|
|
177
|
+
----------
|
|
178
|
+
a, b
|
|
179
|
+
The paired samples, equal length, ``a[i]`` matched with ``b[i]``; at least 2 pairs of finite
|
|
180
|
+
values, and at least one non-zero difference.
|
|
181
|
+
alternative
|
|
182
|
+
``"two-sided"`` (default), ``"less"`` or ``"greater"`` for the differences ``a - b``. The
|
|
183
|
+
CI is always two-sided.
|
|
184
|
+
names
|
|
185
|
+
The names of the two conditions, recorded in the row's ``groups``.
|
|
186
|
+
|
|
187
|
+
Returns
|
|
188
|
+
-------
|
|
189
|
+
dict
|
|
190
|
+
One reporting row keyed by :data:`REPORT_COLUMNS`. The statistic is ``W+``, the sum of the
|
|
191
|
+
ranks of the positive differences (R's ``V``; scipy's two-sided statistic is instead
|
|
192
|
+
``min(W+, W-)``); ``dof`` is ``None`` (a rank test has no degrees of freedom).
|
|
193
|
+
"""
|
|
194
|
+
a, b = as_pairs(a, b)
|
|
195
|
+
diff = a - b
|
|
196
|
+
diff = diff[diff != 0]
|
|
197
|
+
n = diff.size
|
|
198
|
+
if n == 0:
|
|
199
|
+
raise ValueError("every paired difference is zero; the signed-rank test is undefined")
|
|
200
|
+
|
|
201
|
+
w = _sp.wilcoxon(a, b, zero_method="wilcox", alternative=check_alternative(alternative), method="auto")
|
|
202
|
+
ranks = _sp.rankdata(np.abs(diff)) # average ranks for ties
|
|
203
|
+
w_plus = ranks[diff > 0].sum()
|
|
204
|
+
w_minus = ranks[diff < 0].sum()
|
|
205
|
+
r = (w_plus - w_minus) / (w_plus + w_minus)
|
|
206
|
+
lo, hi = _rank_biserial_bounds(r, n)
|
|
207
|
+
return report_row("Wilcoxon signed-rank test", w_plus, w.pvalue, None,
|
|
208
|
+
"Matched-pairs rank-biserial correlation", r, lo, hi,
|
|
209
|
+
n=(a.size, b.size), n_total=a.size, groups=names, alternative=alternative)
|