PyMARE 0.0.12__tar.gz → 0.0.13__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {pymare-0.0.12 → pymare-0.0.13}/PKG-INFO +3 -3
  2. {pymare-0.0.12 → pymare-0.0.13}/PyMARE.egg-info/PKG-INFO +3 -3
  3. {pymare-0.0.12 → pymare-0.0.13}/PyMARE.egg-info/SOURCES.txt +8 -0
  4. {pymare-0.0.12 → pymare-0.0.13}/PyMARE.egg-info/requires.txt +2 -2
  5. {pymare-0.0.12 → pymare-0.0.13}/pymare/_version.py +3 -3
  6. {pymare-0.0.12 → pymare-0.0.13}/pymare/effectsize/base.py +19 -1
  7. {pymare-0.0.12 → pymare-0.0.13}/pymare/effectsize/expressions.json +5 -5
  8. {pymare-0.0.12 → pymare-0.0.13}/pymare/estimators/estimators.py +56 -6
  9. {pymare-0.0.12 → pymare-0.0.13}/pymare/results.py +86 -7
  10. {pymare-0.0.12 → pymare-0.0.13}/pymare/stats.py +124 -11
  11. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/conftest.py +12 -0
  12. pymare-0.0.13/pymare/tests/data/clubsandwich_reference.json +107 -0
  13. pymare-0.0.13/pymare/tests/data/metafor_escalc_inputs.csv +9 -0
  14. pymare-0.0.13/pymare/tests/data/metafor_escalc_reference.json +83 -0
  15. pymare-0.0.13/pymare/tests/data/metafor_permutest_reference.json +50 -0
  16. pymare-0.0.13/pymare/tests/data/metafor_reference.json +2531 -0
  17. pymare-0.0.13/pymare/tests/test_clubsandwich_alignment.py +317 -0
  18. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_effectsize_base.py +29 -1
  19. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_estimators.py +45 -9
  20. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_metafor_alignment.py +7 -5
  21. pymare-0.0.13/pymare/tests/test_metafor_escalc.py +432 -0
  22. pymare-0.0.13/pymare/tests/test_metafor_permutest.py +215 -0
  23. pymare-0.0.13/pymare/tests/test_metafor_random_effects.py +380 -0
  24. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_results.py +5 -2
  25. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_stats.py +60 -3
  26. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/utils.py +7 -2
  27. {pymare-0.0.12 → pymare-0.0.13}/pyproject.toml +1 -0
  28. {pymare-0.0.12 → pymare-0.0.13}/setup.cfg +1 -1
  29. pymare-0.0.12/pymare/tests/data/metafor_reference.json +0 -1450
  30. {pymare-0.0.12 → pymare-0.0.13}/LICENSE +0 -0
  31. {pymare-0.0.12 → pymare-0.0.13}/MANIFEST.in +0 -0
  32. {pymare-0.0.12 → pymare-0.0.13}/PyMARE.egg-info/dependency_links.txt +0 -0
  33. {pymare-0.0.12 → pymare-0.0.13}/PyMARE.egg-info/not-zip-safe +0 -0
  34. {pymare-0.0.12 → pymare-0.0.13}/PyMARE.egg-info/top_level.txt +0 -0
  35. {pymare-0.0.12 → pymare-0.0.13}/README.md +0 -0
  36. {pymare-0.0.12 → pymare-0.0.13}/pymare/__init__.py +0 -0
  37. {pymare-0.0.12 → pymare-0.0.13}/pymare/core.py +0 -0
  38. {pymare-0.0.12 → pymare-0.0.13}/pymare/datasets/__init__.py +0 -0
  39. {pymare-0.0.12 → pymare-0.0.13}/pymare/datasets/metadat.py +0 -0
  40. {pymare-0.0.12 → pymare-0.0.13}/pymare/effectsize/__init__.py +0 -0
  41. {pymare-0.0.12 → pymare-0.0.13}/pymare/effectsize/expressions.py +0 -0
  42. {pymare-0.0.12 → pymare-0.0.13}/pymare/estimators/__init__.py +0 -0
  43. {pymare-0.0.12 → pymare-0.0.13}/pymare/estimators/combination.py +0 -0
  44. {pymare-0.0.12 → pymare-0.0.13}/pymare/estimators/stan/meta_regression.stan +0 -0
  45. {pymare-0.0.12 → pymare-0.0.13}/pymare/resources/datasets/michael2013.json +0 -0
  46. {pymare-0.0.12 → pymare-0.0.13}/pymare/resources/datasets/michael2013.tsv +0 -0
  47. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/__init__.py +0 -0
  48. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/data/metafor_small_sample.csv +0 -0
  49. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/data/robumeta_correlated_effects.csv +0 -0
  50. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/data/robumeta_reference.json +0 -0
  51. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/data/stan_validation.json +0 -0
  52. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_combination_tests.py +0 -0
  53. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_core.py +0 -0
  54. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_datasets.py +0 -0
  55. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_effectsize_expressions.py +0 -0
  56. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_robumeta_alignment.py +0 -0
  57. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_stan_estimators.py +0 -0
  58. {pymare-0.0.12 → pymare-0.0.13}/pymare/tests/test_utils.py +0 -0
  59. {pymare-0.0.12 → pymare-0.0.13}/pymare/utils.py +0 -0
  60. {pymare-0.0.12 → pymare-0.0.13}/pypi_description.md +0 -0
  61. {pymare-0.0.12 → pymare-0.0.13}/setup.py +0 -0
  62. {pymare-0.0.12 → pymare-0.0.13}/versioneer.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyMARE
3
- Version: 0.0.12
3
+ Version: 0.0.13
4
4
  Summary: PyMARE: Python Meta-Analysis & Regression Engine
5
5
  Home-page: https://github.com/neurostuff/PyMARE
6
6
  Author: PyMARE developers
@@ -27,7 +27,7 @@ Requires-Dist: pandas
27
27
  Requires-Dist: scipy
28
28
  Requires-Dist: sympy
29
29
  Provides-Extra: doc
30
- Requires-Dist: m2r2; extra == "doc"
30
+ Requires-Dist: m2r2>=1.1; extra == "doc"
31
31
  Requires-Dist: matplotlib; extra == "doc"
32
32
  Requires-Dist: mistune; extra == "doc"
33
33
  Requires-Dist: numpydoc; extra == "doc"
@@ -55,7 +55,7 @@ Requires-Dist: cmdstanpy<2,>=1.2; extra == "stan"
55
55
  Requires-Dist: arviz>=0.17; extra == "stan"
56
56
  Requires-Dist: scipy<1.13; python_version < "3.10" and extra == "stan"
57
57
  Provides-Extra: all
58
- Requires-Dist: m2r2; extra == "all"
58
+ Requires-Dist: m2r2>=1.1; extra == "all"
59
59
  Requires-Dist: matplotlib; extra == "all"
60
60
  Requires-Dist: mistune; extra == "all"
61
61
  Requires-Dist: numpydoc; extra == "all"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyMARE
3
- Version: 0.0.12
3
+ Version: 0.0.13
4
4
  Summary: PyMARE: Python Meta-Analysis & Regression Engine
5
5
  Home-page: https://github.com/neurostuff/PyMARE
6
6
  Author: PyMARE developers
@@ -27,7 +27,7 @@ Requires-Dist: pandas
27
27
  Requires-Dist: scipy
28
28
  Requires-Dist: sympy
29
29
  Provides-Extra: doc
30
- Requires-Dist: m2r2; extra == "doc"
30
+ Requires-Dist: m2r2>=1.1; extra == "doc"
31
31
  Requires-Dist: matplotlib; extra == "doc"
32
32
  Requires-Dist: mistune; extra == "doc"
33
33
  Requires-Dist: numpydoc; extra == "doc"
@@ -55,7 +55,7 @@ Requires-Dist: cmdstanpy<2,>=1.2; extra == "stan"
55
55
  Requires-Dist: arviz>=0.17; extra == "stan"
56
56
  Requires-Dist: scipy<1.13; python_version < "3.10" and extra == "stan"
57
57
  Provides-Extra: all
58
- Requires-Dist: m2r2; extra == "all"
58
+ Requires-Dist: m2r2>=1.1; extra == "all"
59
59
  Requires-Dist: matplotlib; extra == "all"
60
60
  Requires-Dist: mistune; extra == "all"
61
61
  Requires-Dist: numpydoc; extra == "all"
@@ -32,6 +32,7 @@ pymare/resources/datasets/michael2013.json
32
32
  pymare/resources/datasets/michael2013.tsv
33
33
  pymare/tests/__init__.py
34
34
  pymare/tests/conftest.py
35
+ pymare/tests/test_clubsandwich_alignment.py
35
36
  pymare/tests/test_combination_tests.py
36
37
  pymare/tests/test_core.py
37
38
  pymare/tests/test_datasets.py
@@ -39,12 +40,19 @@ pymare/tests/test_effectsize_base.py
39
40
  pymare/tests/test_effectsize_expressions.py
40
41
  pymare/tests/test_estimators.py
41
42
  pymare/tests/test_metafor_alignment.py
43
+ pymare/tests/test_metafor_escalc.py
44
+ pymare/tests/test_metafor_permutest.py
45
+ pymare/tests/test_metafor_random_effects.py
42
46
  pymare/tests/test_results.py
43
47
  pymare/tests/test_robumeta_alignment.py
44
48
  pymare/tests/test_stan_estimators.py
45
49
  pymare/tests/test_stats.py
46
50
  pymare/tests/test_utils.py
47
51
  pymare/tests/utils.py
52
+ pymare/tests/data/clubsandwich_reference.json
53
+ pymare/tests/data/metafor_escalc_inputs.csv
54
+ pymare/tests/data/metafor_escalc_reference.json
55
+ pymare/tests/data/metafor_permutest_reference.json
48
56
  pymare/tests/data/metafor_reference.json
49
57
  pymare/tests/data/metafor_small_sample.csv
50
58
  pymare/tests/data/robumeta_correlated_effects.csv
@@ -4,7 +4,7 @@ scipy
4
4
  sympy
5
5
 
6
6
  [all]
7
- m2r2
7
+ m2r2>=1.1
8
8
  matplotlib
9
9
  mistune
10
10
  numpydoc
@@ -33,7 +33,7 @@ arviz>=0.17
33
33
  scipy<1.13
34
34
 
35
35
  [doc]
36
- m2r2
36
+ m2r2>=1.1
37
37
  matplotlib
38
38
  mistune
39
39
  numpydoc
@@ -8,11 +8,11 @@ import json
8
8
 
9
9
  version_json = '''
10
10
  {
11
- "date": "2026-08-22T11:31:11-0500",
11
+ "date": "2026-09-28T19:56:53-0500",
12
12
  "dirty": false,
13
13
  "error": null,
14
- "full-revisionid": "613c06ba432451ad536fb06877fe9fd653ceef85",
15
- "version": "0.0.12"
14
+ "full-revisionid": "9f8800eabf56b798820426c6439c111187409d55",
15
+ "version": "0.0.13"
16
16
  }
17
17
  ''' # END VERSION_JSON
18
18
 
@@ -250,6 +250,19 @@ class OneSampleEffectSizeConverter(EffectSizeConverter):
250
250
  summaries, and are _not_ individual data points. E.g., do not pass in
251
251
  a vector of point estimates as `m` and a scalar for the SDs `sd`.
252
252
  The lengths of all inputs must match.
253
+
254
+ .. versionchanged:: 0.0.13
255
+
256
+ The sampling variance of a raw correlation (``'R'``) is now
257
+ ``(1 - r**2)**2 / (n - 1)``, matching ``metafor::escalc(measure="COR")``.
258
+ It was ``(1 - r**2) / (n - 2)``, which is the squared standard error of
259
+ ``r`` under the null hypothesis of *no* correlation rather than its
260
+ sampling variance at the observed value. The two agree near ``r = 0``
261
+ and diverge by a factor of ``(n - 1) / ((n - 2)(1 - r**2))`` -- 51x at
262
+ ``r = 0.99``. Since that factor depends on the data, the old expression
263
+ did not simply inflate variances: it reweighted studies against one
264
+ another, pulling a pooled estimate toward those with the weakest
265
+ correlations. ``'ZR'`` is unaffected and remains the measure to prefer.
253
266
  """
254
267
 
255
268
  _type = 1
@@ -272,7 +285,12 @@ class OneSampleEffectSizeConverter(EffectSizeConverter):
272
285
  a bias correction applied.
273
286
  - 'D': Cohen's d. Note that no bias correction is applied
274
287
  (use 'SM' instead).
275
- - 'R': Raw correlation coefficient.
288
+ - 'R': Raw correlation coefficient. Prefer 'ZR' for
289
+ meta-analysis: the sampling variance of a raw correlation
290
+ depends strongly on the correlation itself, so studies are
291
+ weighted very unequally by how large their correlations
292
+ happen to be, which is the problem the Fisher transform
293
+ exists to remove.
276
294
  - 'ZR': Fisher z-transformed correlation coefficient.
277
295
  **kwargs
278
296
  Optional keyword arguments to pass onto the Dataset
@@ -15,7 +15,7 @@
15
15
  "description": "Cohen's d (one-sample)"
16
16
  },
17
17
  {
18
- "expression": "v_d - ((n - 1)/(n - 3)) * (1 / n + d**2) - d**2 / j**2 * n",
18
+ "expression": "v_d - (((n - 1)/(n - 3)) * (1 / n + d**2) - d**2 / j**2)",
19
19
  "type": 1,
20
20
  "description": "Variance of Cohen's d"
21
21
  },
@@ -30,7 +30,7 @@
30
30
  "description": "Standardized mean (Hedges's g)"
31
31
  },
32
32
  {
33
- "expression": "v_sm - ((n - 1)/(n - 3)) * j**2 * (1 / n + d**2) - d**2",
33
+ "expression": "v_sm - (((n - 1)/(n - 3)) * j**2 * (1 / n + d**2) - d**2)",
34
34
  "type": 1,
35
35
  "description": "Variance of standardized mean"
36
36
  },
@@ -40,7 +40,7 @@
40
40
  "description": "Raw correlation coefficient"
41
41
  },
42
42
  {
43
- "expression": "v_r - (1 - r**2) / (n - 2)",
43
+ "expression": "v_r - (1 - r**2)**2 / (n - 1)",
44
44
  "type": 1,
45
45
  "description": "Variance of raw correlation coefficient"
46
46
  },
@@ -60,7 +60,7 @@
60
60
  "description": "Raw mean difference"
61
61
  },
62
62
  {
63
- "expression": "v_rmd - (sd1**2 / n1) + (sd2**2 / n2)",
63
+ "expression": "v_rmd - ((sd1**2 / n1) + (sd2**2 / n2))",
64
64
  "type": 2,
65
65
  "description": "Variance of raw mean difference"
66
66
  },
@@ -75,7 +75,7 @@
75
75
  "description": "Cohen's d (two-sample)"
76
76
  },
77
77
  {
78
- "expression": "v_d - ((n1 + n2)/(n1 * n2) + d**2 / 2 * (n1 + n2 - 2))",
78
+ "expression": "v_d - ((n1 + n2)/(n1 * n2) + d**2 / (2 * (n1 + n2 - 2)))",
79
79
  "type": 2,
80
80
  "description": "Variance of Cohen's d"
81
81
  },
@@ -570,7 +570,9 @@ def _dersimonian_laird_tau2(y, v, X):
570
570
  Returns
571
571
  -------
572
572
  :obj:`numpy.ndarray` of shape (D,)
573
- The tau^2 estimate per parallel dataset, floored at zero.
573
+ The tau^2 estimate per parallel dataset, floored at zero. Zero when the
574
+ design is saturated (``K <= P``), where there is no residual dispersion
575
+ to measure.
574
576
 
575
577
  Notes
576
578
  -----
@@ -585,6 +587,15 @@ def _dersimonian_laird_tau2(y, v, X):
585
587
  # Estimate initial betas with WLS, assuming tau^2=0
586
588
  beta_wls, model_cov = weighted_least_squares(y, v, X, return_cov=True)
587
589
 
590
+ if k <= p:
591
+ # A saturated design fits every observation exactly, so Q and A are both
592
+ # zero in exact arithmetic and the quotient below is one rounding
593
+ # residue over another -- 2.5e-31 / -8.9e-16 on one machine, and a
594
+ # positive numerator over an A that underflowed to zero on the next,
595
+ # which is +inf. There is no dispersion left to measure either way, so
596
+ # report none rather than let the residues decide.
597
+ return np.zeros(np.atleast_2d(y).shape[1])
598
+
588
599
  # Cochran's Q
589
600
  w = 1.0 / v
590
601
  w_sum = w.sum(0)
@@ -1134,9 +1145,19 @@ class Hedges(BaseEstimator):
1134
1145
  The ``X`` matrix must be identical for all iterates.
1135
1146
 
1136
1147
  Unlike the coefficients, tau^2 is derived from an *unweighted* fit: it is the excess of
1137
- the ordinary mean squared error over the mean sampling variance. The coefficients are
1138
- then refitted with ``1 / (v + tau^2)`` weights, and the reported covariance comes from
1139
- that second fit.
1148
+ the ordinary mean squared error over what that error is expected to be when tau^2 is
1149
+ zero, namely ``tr(PV) / (K - P)`` with ``P`` the ordinary-least-squares residual maker
1150
+ and ``V`` the diagonal matrix of sampling variances. The coefficients are then refitted
1151
+ with ``1 / (v + tau^2)`` weights, and the reported covariance comes from that second
1152
+ fit.
1153
+
1154
+ .. versionchanged:: 0.0.13
1155
+
1156
+ The subtracted term was previously the mean sampling variance ``sum(v) / K``. That
1157
+ is the same quantity when the intercept is the only predictor, but not otherwise,
1158
+ so tau^2 was out by up to 0.14 relative in a meta-regression -- the divergence from
1159
+ ``metafor``'s ``HE`` that ``validation/metafor/README.md`` recorded. Intercept-only
1160
+ models are unaffected.
1140
1161
 
1141
1162
  .. versionchanged:: 0.0.11
1142
1163
 
@@ -1216,8 +1237,37 @@ class Hedges(BaseEstimator):
1216
1237
  # feeds the variance component only; the coefficients are refitted with
1217
1238
  # inverse-variance weights below.
1218
1239
  tau_beta = weighted_least_squares(tau_y, np.ones_like(tau_y), tau_X)
1219
- mse = ((tau_y - tau_X.dot(tau_beta)) ** 2).sum(0) / (tau_k - tau_p)
1220
- tau_ho = np.maximum(0, mse - tau_v.sum(0) / tau_k)
1240
+ residual_ss = ((tau_y - tau_X.dot(tau_beta)) ** 2).sum(0)
1241
+
1242
+ # What that residual sum of squares is expected to be when tau^2 is
1243
+ # zero, which is what has to be subtracted off. With P the OLS residual
1244
+ # maker I - X (X'X)^-1 X' and V = diag(v), it is tr(PV) -- and since
1245
+ # only P's diagonal is needed, that is sum_i (1 - h_i) v_i for the OLS
1246
+ # leverages h_i.
1247
+ #
1248
+ # Not the mean sampling variance sum(v) / K, which is tr(PV) / (K - P)
1249
+ # only when the intercept is the only predictor: P is then I - J/K,
1250
+ # every h_i is 1/K, and the sum collapses to sum(v) (K - 1) / K. With a
1251
+ # moderator the two part company, and using the intercept-only form put
1252
+ # tau^2 out by up to 0.14 relative against metafor's HE.
1253
+ leverage = np.einsum("ij,jk,ik->i", tau_X, np.linalg.pinv(tau_X.T @ tau_X), tau_X)
1254
+ expected_ss = ((1.0 - leverage)[:, None] * tau_v).sum(0)
1255
+
1256
+ residual_dof = tau_k - tau_p
1257
+ if residual_dof > 0:
1258
+ # One division rather than two, which is also how metafor writes it.
1259
+ tau_ho = np.maximum(0, (residual_ss - expected_ss) / residual_dof)
1260
+ else:
1261
+ # A saturated design fits every observation exactly, so there is no
1262
+ # residual left to measure dispersion with and both terms above are
1263
+ # zero to rounding. Dividing anyway lets the sign of that rounding
1264
+ # decide the answer: the leverages come back as 1 +- 1e-16, so
1265
+ # expected_ss lands either side of zero and tau^2 comes out +inf on
1266
+ # one machine and NaN on the next, which is how this reached CI.
1267
+ # Report no excess dispersion instead, which is what
1268
+ # DerSimonianLaird and the likelihood estimators already do here,
1269
+ # and what this estimator already did for K < P.
1270
+ tau_ho = np.zeros_like(residual_ss)
1221
1271
 
1222
1272
  # Estimate beta with tau^2 estimate. The covariance has to come from
1223
1273
  # this fit rather than the OLS one above: (X'WX)^-1 is only the
@@ -121,6 +121,46 @@ def _random_unit_order(classes, n_units):
121
121
  return order
122
122
 
123
123
 
124
+ #: Relative slack allowed when deciding whether a permuted statistic is at least
125
+ #: as extreme as the observed one. The permutation that reproduces the observed
126
+ #: data -- the identity, always a member of the set -- must tie with it and so
127
+ #: must count; but the observed statistic is computed by a different code path
128
+ #: from the permuted ones, which refit every dataset in one batched call, and
129
+ #: the two can disagree by a unit in the last place. Without slack that drops
130
+ #: the identity, and with sign flipping its mirror too, understating the p-value
131
+ #: by 2 / 2**m. The square root of machine epsilon is the same constant
132
+ #: ``metafor::permutest`` uses for this, applied relatively rather than
133
+ #: absolutely so that it does not depend on the scale of the statistic. It is
134
+ #: seven orders of magnitude below the closest genuine near-tie observed on the
135
+ #: designs in ``pymare/tests/data/metafor_small_sample.csv``.
136
+ _PERMUTATION_TIE_RTOL = np.sqrt(np.finfo(np.float64).eps)
137
+
138
+
139
+ def _at_least_as_extreme(permuted, observed):
140
+ """Return which permuted statistics are at least as extreme as the observed one.
141
+
142
+ Parameters
143
+ ----------
144
+ permuted : :obj:`numpy.ndarray`
145
+ Absolute permuted statistics, permutations along the last axis.
146
+ observed : :obj:`numpy.ndarray`
147
+ The absolute observed statistic, broadcastable against ``permuted``.
148
+
149
+ Returns
150
+ -------
151
+ :obj:`numpy.ndarray` of :obj:`bool`
152
+ Elementwise ``permuted >= observed``, to within
153
+ :data:`_PERMUTATION_TIE_RTOL`.
154
+
155
+ Notes
156
+ -----
157
+ ``>=`` and not ``>``: a permuted statistic that ties with the observed one is
158
+ at least as extreme, and excluding ties makes the test anti-conservative on
159
+ discrete or degenerate data.
160
+ """
161
+ return permuted >= observed - np.abs(observed) * _PERMUTATION_TIE_RTOL
162
+
163
+
124
164
  class MetaRegressionResults:
125
165
  """Container for results generated by PyMARE meta-regression estimators.
126
166
 
@@ -626,6 +666,24 @@ class MetaRegressionResults:
626
666
 
627
667
  Notes
628
668
  -----
669
+ The statistic permuted is the coefficient divided by its standard error,
670
+ not the coefficient itself, and the same for tau^2. Refitting a permuted
671
+ dataset re-estimates tau^2, which moves the weights and so the standard
672
+ error, so ``|beta|`` is not pivotal and its permutation distribution is
673
+ not the null this test needs. The two coincide only where the standard
674
+ error is invariant under the permutation -- a fixed-effects,
675
+ intercept-only model under sign flipping. This matches
676
+ ``metafor::permutest``.
677
+
678
+ .. versionchanged:: 0.0.13
679
+ Counted ``|beta|`` in earlier releases, and compared it against the
680
+ observed value exactly. Both are fixed: the statistic is now
681
+ ``|beta / se|``, and the comparison allows a relative slack of the
682
+ square root of machine epsilon so that the identity permutation --
683
+ which reproduces the observed data and must therefore count -- is not
684
+ dropped over a unit in the last place. Every permutation p-value this
685
+ method reports is affected.
686
+
629
687
  If the number of possible permutations is smaller than n_perm, an exact test will be
630
688
  conducted.
631
689
  Otherwise an approximate test will be conducted by randomly shuffling the outcomes n_perm
@@ -745,16 +803,37 @@ class MetaRegressionResults:
745
803
  # freedom (or raising on reshape).
746
804
  params = copy.copy(self.estimator).fit(**kwargs).params_
747
805
 
806
+ # The statistic is the coefficient over its standard error, not
807
+ # the coefficient: refitting a permuted dataset re-estimates tau^2,
808
+ # which moves the weights and hence the standard error, so |beta|
809
+ # is not pivotal and its permutation distribution is not the null
810
+ # the test needs. The two coincide only where the standard error
811
+ # happens to be invariant under the permutation -- a fixed-effects
812
+ # intercept-only model under sign flipping. This is what
813
+ # metafor::permutest counts.
748
814
  fe_obs = fe_stats["est"][:, i]
815
+ se_obs = fe_stats["se"][:, i]
749
816
  if fe_obs.ndim == 1:
750
- fe_obs = fe_obs[:, None]
751
- # <=, not <: a permuted statistic that ties with the observed one
752
- # is at least as extreme, and excluding ties makes the test
753
- # anti-conservative on discrete or degenerate data.
754
- fe_p[:, i] = (np.abs(fe_obs) <= np.abs(params["fe_params"])).mean(1)
817
+ fe_obs, se_obs = fe_obs[:, None], se_obs[:, None]
818
+
819
+ # The permuted standard errors, taken from the covariance the
820
+ # estimator just reported, exactly as `fe_se` takes the observed
821
+ # one from `fe_cov` -- including whatever small-sample correction
822
+ # the estimator applies, which the observed statistic carries too.
823
+ # (P, P, n_perm) on every path this method can drive -- the
824
+ # closed-form estimators, the two likelihood ones, the sample
825
+ # size-based one and the cluster-robust branch all report a
826
+ # covariance per parallel dataset.
827
+ perm_cov = np.asarray(params["inv_cov"])
828
+ # A zero standard error divides to +-inf; the comparison below
829
+ # still orders those correctly, and a NaN counts as not extreme.
830
+ with np.errstate(invalid="ignore", divide="ignore"):
831
+ perm_se = np.sqrt(np.diagonal(perm_cov)).T
832
+ fe_p[:, i] = _at_least_as_extreme(
833
+ np.abs(params["fe_params"] / perm_se), np.abs(fe_obs / se_obs)
834
+ ).mean(1)
755
835
  if rfx:
756
- abs_obs = np.abs(tau2[i])
757
- tau_p[i] = (abs_obs <= np.abs(params["tau2"])).mean()
836
+ tau_p[i] = _at_least_as_extreme(np.abs(params["tau2"]), np.abs(tau2[i])).mean()
758
837
 
759
838
  # p-values can't be smaller than 1/n_perm
760
839
  params = {"fe_p": np.maximum(1 / n_perm, fe_p)}
@@ -19,7 +19,7 @@ from typing import NamedTuple
19
19
 
20
20
  import numpy as np
21
21
  import scipy.stats as ss
22
- from scipy.optimize import Bounds, minimize
22
+ from scipy.optimize import brentq
23
23
  from scipy.special import gammaln
24
24
 
25
25
  # At or below this many clusters, robust variance estimation is known to be
@@ -1515,6 +1515,27 @@ def _cr2_scores(X, w, resid, group_members, bread):
1515
1515
  above is the solution for :math:`\Phi = I` in the whitened metric, i.e.
1516
1516
  under the assumption that the weights are correct and the observations
1517
1517
  independent -- the same assumption the sandwich exists to avoid relying on.
1518
+
1519
+ That condition has many solutions, because it constrains :math:`A_j` only
1520
+ through :math:`A_j B_j A_j'`. ``clubSandwich`` selects the symmetric one,
1521
+ :math:`A_j = \Psi_j^{1/2} (\Psi_j^{1/2} B_j \Psi_j^{1/2})^{-1/2}
1522
+ \Psi_j^{1/2}`; the form here is symmetric in the whitened metric instead,
1523
+ and the two agree exactly when :math:`W_j` is a multiple of the identity --
1524
+ that is, when the sampling variances are constant within the group. Both
1525
+ satisfy the condition and both are exactly unbiased under the working model,
1526
+ so the choice is not between a right and a wrong one.
1527
+
1528
+ It is made this way for the reason the next paragraph gives: in the whitened
1529
+ metric :math:`I_j - H_j` is the identity minus a rank-:math:`p` term, so its
1530
+ spectrum collapses to :math:`p` non-unit eigenvalues however large the group
1531
+ is, and :func:`_cr2_low_rank_factors` can take the inverse square root in
1532
+ :math:`p \times p` work. ``clubSandwich``'s matrix is
1533
+ :math:`\Psi_j^2` minus a rank-:math:`p` term, whose diagonal part is not a
1534
+ multiple of the identity, so it has :math:`n_j` distinct eigenvalues and
1535
+ needs the full :math:`n_j \times n_j` eigendecomposition -- a factor of 400
1536
+ more work at :math:`n_j = 200` and 5,900 at :math:`n_j = 800`. The square
1537
+ root not commuting with an asymmetric congruence is at once why the two
1538
+ forms differ and why only one of them factors.
1518
1539
  That is pragmatic rather than circular: simulation shows the correction
1519
1540
  helps substantially even when the working model is wrong
1520
1541
  :footcite:p:`tipton2015small,imbens2016robust`, and its influence fades as
@@ -1945,6 +1966,22 @@ def cluster_robust_cov(
1945
1966
  - ``"CR2"`` (default) inflates each group's residuals by
1946
1967
  :math:`(I_j - H_j)^{-1/2}` to undo the shrinkage caused by fitting
1947
1968
  :math:`\beta` with that group included :footcite:p:`bell2002bias`.
1969
+
1970
+ .. note::
1971
+
1972
+ This is a CR2 in the defining sense -- it satisfies
1973
+ :math:`A_j B_j A_j' = \Psi_j` exactly, and the resulting
1974
+ sandwich is exactly unbiased for the model-based covariance
1975
+ under the working model -- but it is not bit-for-bit
1976
+ ``clubSandwich``'s ``CR2``. That condition does not pin
1977
+ :math:`A_j` down uniquely: ``clubSandwich`` closes it by taking
1978
+ :math:`A_j` symmetric, and this takes it symmetric in the
1979
+ whitened metric instead. The two coincide exactly when the
1980
+ weights are constant within a group, and differ otherwise --
1981
+ by up to 1e-2 relative on the standard errors of the designs in
1982
+ ``validation/clubsandwich``, which measures it. See
1983
+ :func:`_cr2_scores` for why the whitened form is the one
1984
+ implemented.
1948
1985
  - ``"CR0"`` uses the raw residuals with the blunt ``m / (m - p)``
1949
1986
  scaling. This is the historical behaviour.
1950
1987
 
@@ -2291,6 +2328,76 @@ def ensure_2d(arr):
2291
2328
  return arr
2292
2329
 
2293
2330
 
2331
+ #: Doublings allowed when widening the bracket for a Q-profile bound. Q(tau^2)
2332
+ #: falls to zero as tau^2 grows, so a root exists whenever Q(0) exceeds the
2333
+ #: critical value and the search terminates long before this; it is a stop for
2334
+ #: the degenerate case of a zero critical value, which no root can reach.
2335
+ _Q_PROFILE_MAX_DOUBLINGS = 200
2336
+
2337
+ #: Relative tolerance asked of the Q-profile root finder. Four times machine
2338
+ #: epsilon is the smallest :func:`scipy.optimize.brentq` accepts, which is what
2339
+ #: makes the bounds agree with metafor's ``confint.rma.uni`` to the last few
2340
+ #: bits rather than to the third decimal.
2341
+ _Q_PROFILE_RTOL = 4 * np.finfo(np.float64).eps
2342
+
2343
+
2344
+ def _invert_q(excess, crit, scale):
2345
+ """Solve ``Q(tau^2) = crit`` for tau^2 >= 0.
2346
+
2347
+ Parameters
2348
+ ----------
2349
+ excess : callable
2350
+ ``(tau2, crit) -> Q(tau2) - crit``.
2351
+ crit : :obj:`float`
2352
+ The chi-squared quantile being inverted.
2353
+ scale : :obj:`float`
2354
+ A point estimate of tau^2, used to size the first bracket tried.
2355
+
2356
+ Returns
2357
+ -------
2358
+ :obj:`float`
2359
+ The root, ``0.0`` when ``Q(0) <= crit`` so that no positive root
2360
+ exists, or :obj:`numpy.nan` when the bracket could not be widened far
2361
+ enough to contain one.
2362
+
2363
+ Notes
2364
+ -----
2365
+ Solved as a root rather than as the minimum of ``(Q(tau^2) - crit)**2``,
2366
+ which is what this function replaced. Squaring is what makes the difference:
2367
+ it turns a transversal crossing into a tangential minimum, so the gradient
2368
+ the minimizer follows vanishes as ``crit`` is approached and it stops while
2369
+ still far from the root in the flat upper tail, where ``Q`` changes slowly.
2370
+ The upper bound was wrong by up to 4% relative against
2371
+ ``metafor::confint.rma.uni`` on the designs in
2372
+ ``pymare/tests/data/metafor_small_sample.csv`` for that reason; it now
2373
+ agrees to a few multiples of machine epsilon.
2374
+
2375
+ ``Q`` is monotonically decreasing in tau^2, but Brent's method needs only a
2376
+ sign change over the bracket, so nothing here relies on that.
2377
+ """
2378
+ if excess(0.0, crit) <= 0:
2379
+ # Q at tau^2 = 0 is already at or below the critical value, so the
2380
+ # profile never crosses it. metafor reports the boundary here too.
2381
+ return 0.0
2382
+
2383
+ # Q(tau^2) -> 0 as tau^2 -> infinity, since the weights approach a common
2384
+ # 1 / tau^2 that scales the residual sum of squares away. So a root exists
2385
+ # for any positive crit, and doubling finds it. What reaches the fallback
2386
+ # below is a crit that is not a number at all: a saturated design has
2387
+ # K - P = 0 degrees of freedom, `scipy.stats.chi2.ppf` returns NaN there,
2388
+ # and every comparison against NaN is False, so the loop runs out. NaN
2389
+ # bounds are the right answer for a design with no residual to profile.
2390
+ upper = max(abs(scale), 1.0)
2391
+ for _ in range(_Q_PROFILE_MAX_DOUBLINGS):
2392
+ if excess(upper, crit) <= 0:
2393
+ break
2394
+ upper *= 2.0
2395
+ else:
2396
+ return np.nan
2397
+
2398
+ return brentq(excess, 0.0, upper, args=(crit,), rtol=_Q_PROFILE_RTOL, maxiter=200)
2399
+
2400
+
2294
2401
  def q_profile(y, v, X, alpha=0.05, groups=None):
2295
2402
  """Get the CI for tau^2 via the Q-Profile method.
2296
2403
 
@@ -2343,22 +2450,28 @@ def q_profile(y, v, X, alpha=0.05, groups=None):
2343
2450
  l_crit = ss.chi2.ppf(1 - alpha / 2, df)
2344
2451
  u_crit = ss.chi2.ppf(alpha / 2, df)
2345
2452
  args = (ensure_2d(y), ensure_2d(v), X)
2346
- bds = Bounds([0], [np.inf], keep_feasible=True)
2347
2453
 
2348
- # Use a point estimate of tau^2 as a starting point; when using a fixed
2349
- # value, minimize() sometimes fails to stay in bounds. It has to be the
2350
- # estimator that matches the Q being inverted, or the search can start on
2351
- # the wrong side of the upper root.
2454
+ def excess(tau2, crit):
2455
+ """Q(tau^2) - crit, the function whose root is a bound."""
2456
+ return float(np.ravel(q_gen(*args, float(tau2), groups))[0]) - crit
2457
+
2458
+ # A scale for the bracket search, not a starting point: the root finder
2459
+ # below needs an interval that contains the root, and the point estimate
2460
+ # says what order of magnitude tau^2 lives at. It has to be the estimator
2461
+ # that matches the Q being inverted, so that the first bracket tried is
2462
+ # usually already wide enough.
2352
2463
  if groups is None:
2353
2464
  from .estimators import DerSimonianLaird
2354
2465
 
2355
- ub_start = 2 * DerSimonianLaird().fit(y, v, X).params_["tau2"]
2466
+ scale = DerSimonianLaird().fit(y, v, X).params_["tau2"]
2356
2467
  else:
2357
- ub_start = 2 * correlated_effects_tau2(*args, groups)
2468
+ scale = correlated_effects_tau2(*args, groups)
2358
2469
 
2359
- lb = minimize(lambda x: (q_gen(*args, x, groups) - l_crit) ** 2, [0], bounds=bds).x[0]
2360
- ub = minimize(lambda x: (q_gen(*args, x, groups) - u_crit) ** 2, ub_start, bounds=bds).x[0]
2361
- return {"ci_l": lb, "ci_u": ub}
2470
+ scale = float(np.ravel(scale)[0])
2471
+ return {
2472
+ "ci_l": _invert_q(excess, l_crit, scale),
2473
+ "ci_u": _invert_q(excess, u_crit, scale),
2474
+ }
2362
2475
 
2363
2476
 
2364
2477
  #: Iterations allowed in the continued fraction of :func:`log_chi2_sf`. A safety
@@ -383,6 +383,18 @@ def robumeta_dataset():
383
383
  return frame, designs
384
384
 
385
385
 
386
+ @pytest.fixture(scope="package")
387
+ def clubsandwich_dataset():
388
+ """Load the dataset the clubSandwich reference values were computed on.
389
+
390
+ The same CSV :func:`robumeta_dataset` reads, and returning only the frame:
391
+ the clubSandwich alignment builds its designs through
392
+ :class:`~pymare.core.Dataset` so that the group labels travel with them,
393
+ rather than assembling a bare design matrix as the robumeta alignment does.
394
+ """
395
+ return pd.read_csv(op.join(get_test_data_path(), "robumeta_correlated_effects.csv"))
396
+
397
+
386
398
  @pytest.fixture(scope="package")
387
399
  def metafor_dataset():
388
400
  """Load the designs the metafor reference values were computed on."""