microdf-python 1.4.1__tar.gz → 1.5.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {microdf_python-1.4.1/microdf_python.egg-info → microdf_python-1.5.1}/PKG-INFO +1 -1
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/__init__.py +4 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/microdataframe.py +5 -11
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/microseries.py +70 -65
- microdf_python-1.5.1/microdf/replication.py +192 -0
- microdf_python-1.5.1/microdf/tests/test_binary_weight_alignment.py +195 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_dataframe_weight_storage.py +47 -0
- microdf_python-1.5.1/microdf/tests/test_replication.py +544 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1/microdf_python.egg-info}/PKG-INFO +1 -1
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/SOURCES.txt +3 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/pyproject.toml +1 -1
- {microdf_python-1.4.1 → microdf_python-1.5.1}/LICENSE +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/README.md +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/_weights.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/conftest.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_aggregation_errors.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_microseries_dataframe.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_nullify_weights_index.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_pandas3_compatibility.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_quantile_missing_values.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_serialization.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_sum_axes.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_version_metadata.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_weight_propagation.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_weighted_cov_corr.py +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/dependency_links.txt +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/requires.txt +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/top_level.txt +0 -0
- {microdf_python-1.4.1 → microdf_python-1.5.1}/setup.cfg +0 -0
|
@@ -2,6 +2,7 @@ from importlib.metadata import PackageNotFoundError, version
|
|
|
2
2
|
|
|
3
3
|
from .microdataframe import MicroDataFrame, MicroDataFrameGroupBy
|
|
4
4
|
from .microseries import MicroSeries, MicroSeriesGroupBy
|
|
5
|
+
from .replication import replicate_standard_error, replicate_variance
|
|
5
6
|
|
|
6
7
|
name = "microdf"
|
|
7
8
|
|
|
@@ -19,4 +20,7 @@ __all__ = [
|
|
|
19
20
|
# microdataframe.py
|
|
20
21
|
"MicroDataFrame",
|
|
21
22
|
"MicroDataFrameGroupBy",
|
|
23
|
+
# replication.py
|
|
24
|
+
"replicate_variance",
|
|
25
|
+
"replicate_standard_error",
|
|
22
26
|
]
|
|
@@ -460,7 +460,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
|
|
|
460
460
|
if inplace:
|
|
461
461
|
# Snapshot weight *values* positionally — the index is about
|
|
462
462
|
# to change and reset_index preserves row order.
|
|
463
|
-
weight_values = np.
|
|
463
|
+
weight_values = np.array(self.weights, dtype=float, copy=True)
|
|
464
464
|
super().reset_index(
|
|
465
465
|
level=level,
|
|
466
466
|
drop=drop,
|
|
@@ -470,7 +470,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
|
|
|
470
470
|
allow_duplicates=allow_duplicates,
|
|
471
471
|
names=names,
|
|
472
472
|
)
|
|
473
|
-
self.weights =
|
|
473
|
+
self.weights = weight_series(weight_values, self.index)
|
|
474
474
|
self._link_all_weights()
|
|
475
475
|
return None
|
|
476
476
|
else:
|
|
@@ -483,15 +483,9 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
|
|
|
483
483
|
allow_duplicates=allow_duplicates,
|
|
484
484
|
names=names,
|
|
485
485
|
)
|
|
486
|
-
|
|
487
|
-
#
|
|
488
|
-
|
|
489
|
-
out.weights = pd.Series(
|
|
490
|
-
np.asarray(self.weights.values, dtype=float),
|
|
491
|
-
index=out.index,
|
|
492
|
-
dtype=float,
|
|
493
|
-
)
|
|
494
|
-
return out
|
|
486
|
+
# Own a positional copy: reset_index changes labels but
|
|
487
|
+
# preserves row order.
|
|
488
|
+
return MicroDataFrame(res, weights=weight_series(self.weights, res.index))
|
|
495
489
|
|
|
496
490
|
def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame":
|
|
497
491
|
return super().copy(deep)
|
|
@@ -120,6 +120,16 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
120
120
|
super().__finalize__(other, method=method, **kwargs)
|
|
121
121
|
return finalize_weights(self, other, method, previous)
|
|
122
122
|
|
|
123
|
+
def _construct_result(self, *args, **kwargs):
|
|
124
|
+
# pandas has already aligned this Series before constructing a binary
|
|
125
|
+
# result. Retain its row weights, even when pandas 3 also finalizes
|
|
126
|
+
# metadata from the other operand. Delegate values and names to pandas.
|
|
127
|
+
result = super()._construct_result(*args, **kwargs)
|
|
128
|
+
if not isinstance(result, tuple):
|
|
129
|
+
result.weights = weight_series(self.weights, result.index)
|
|
130
|
+
# divmod constructs both tuple members through this same hook.
|
|
131
|
+
return result
|
|
132
|
+
|
|
123
133
|
def __setattr__(self, name, value):
|
|
124
134
|
weights = self.__dict__.get("weights") if name == "index" else None
|
|
125
135
|
super().__setattr__(name, value)
|
|
@@ -579,6 +589,51 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
579
589
|
"""
|
|
580
590
|
return self.quantile(0.5, skipna=skipna)
|
|
581
591
|
|
|
592
|
+
def replicate_standard_error(
|
|
593
|
+
self,
|
|
594
|
+
statistic: Callable,
|
|
595
|
+
replicate_weights,
|
|
596
|
+
method: str = "jackknife",
|
|
597
|
+
fay_k: Optional[float] = None,
|
|
598
|
+
*,
|
|
599
|
+
centering: str = "full-sample",
|
|
600
|
+
) -> float:
|
|
601
|
+
"""Standard error of ``statistic`` from a set of replicate weights.
|
|
602
|
+
|
|
603
|
+
Recomputes the statistic once per replicate and scales the spread by
|
|
604
|
+
the factor appropriate to how the replicates were built. Statistical
|
|
605
|
+
validity depends on both the statistic and the survey design.
|
|
606
|
+
Nonsmooth statistics such as quantiles can require an appropriate
|
|
607
|
+
replication method or smoothing of replicate estimates.
|
|
608
|
+
|
|
609
|
+
The factor and centering convention must match the survey design.
|
|
610
|
+
Supported schemes use a common factor: jackknife covers unstratified
|
|
611
|
+
JK1 or common-factor delete-group replication, not arbitrary stratified
|
|
612
|
+
jackknife. Averaged bootstrap requiring additional factors is not
|
|
613
|
+
supported. See :func:`microdf.replication.replicate_variance` for factors.
|
|
614
|
+
|
|
615
|
+
Changing main weights without corresponding design-consistent replicate
|
|
616
|
+
adjustments invalidates the original replicates. Calibration can be
|
|
617
|
+
valid when repeated appropriately for every replicate.
|
|
618
|
+
|
|
619
|
+
:param statistic: Callable taking a MicroSeries and returning a float, e.g.
|
|
620
|
+
``lambda s: s.median()``.
|
|
621
|
+
:param replicate_weights: Array or frame of shape ``(len(self), R)``
|
|
622
|
+
in the same row order as this series. DataFrame labels are ignored.
|
|
623
|
+
:param method: ``jackknife``, ``brr``, ``bootstrap``,
|
|
624
|
+
``successive-difference`` or ``fay``.
|
|
625
|
+
:param fay_k: Fay's perturbation constant, for ``method="fay"``.
|
|
626
|
+
:param centering: ``full-sample`` (default) centers on ``statistic(self)``;
|
|
627
|
+
``replicate-mean`` centers on the mean of the replicate estimates.
|
|
628
|
+
The method's scale factor is unchanged.
|
|
629
|
+
:returns: The estimated standard error.
|
|
630
|
+
"""
|
|
631
|
+
from microdf.replication import replicate_standard_error
|
|
632
|
+
|
|
633
|
+
return replicate_standard_error(
|
|
634
|
+
self, statistic, replicate_weights, method, fay_k, centering=centering
|
|
635
|
+
)
|
|
636
|
+
|
|
582
637
|
@scalar_function
|
|
583
638
|
def gini(self, negatives: Optional[str] = None) -> float:
|
|
584
639
|
"""Calculates Gini index.
|
|
@@ -890,95 +945,45 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
890
945
|
def __getattr__(self, name: str) -> "MicroSeries":
|
|
891
946
|
return MicroSeries(super().__getattr__(name), weights=self.weights)
|
|
892
947
|
|
|
893
|
-
#
|
|
894
|
-
|
|
895
|
-
def __add__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
896
|
-
return MicroSeries(super().__add__(other), weights=self.weights)
|
|
897
|
-
|
|
898
|
-
def __sub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
899
|
-
return MicroSeries(super().__sub__(other), weights=self.weights)
|
|
900
|
-
|
|
901
|
-
def __mul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
902
|
-
return MicroSeries(super().__mul__(other), weights=self.weights)
|
|
903
|
-
|
|
904
|
-
def __floordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
905
|
-
return MicroSeries(super().__floordiv__(other), weights=self.weights)
|
|
906
|
-
|
|
907
|
-
def __truediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
908
|
-
return MicroSeries(super().__truediv__(other), weights=self.weights)
|
|
909
|
-
|
|
910
|
-
def __mod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
911
|
-
return MicroSeries(super().__mod__(other), weights=self.weights)
|
|
912
|
-
|
|
913
|
-
def __pow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
914
|
-
return MicroSeries(super().__pow__(other), weights=self.weights)
|
|
915
|
-
|
|
916
|
-
def __xor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
917
|
-
return MicroSeries(super().__xor__(other), weights=self.weights)
|
|
918
|
-
|
|
919
|
-
def __and__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
920
|
-
return MicroSeries(super().__and__(other), weights=self.weights)
|
|
921
|
-
|
|
922
|
-
def __or__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
923
|
-
return MicroSeries(super().__or__(other), weights=self.weights)
|
|
924
|
-
|
|
925
|
-
def __invert__(self) -> "MicroSeries":
|
|
926
|
-
return MicroSeries(super().__invert__(), weights=self.weights)
|
|
927
|
-
|
|
948
|
+
# Explicit reverse overrides give this subclass priority when a plain
|
|
949
|
+
# pandas Series is on the left. _construct_result retains aligned weights.
|
|
928
950
|
def __radd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
929
|
-
return
|
|
951
|
+
return super().__radd__(other)
|
|
930
952
|
|
|
931
953
|
def __rsub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
932
|
-
return
|
|
954
|
+
return super().__rsub__(other)
|
|
933
955
|
|
|
934
956
|
def __rmul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
935
|
-
return
|
|
957
|
+
return super().__rmul__(other)
|
|
936
958
|
|
|
937
959
|
def __rfloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
938
|
-
return
|
|
960
|
+
return super().__rfloordiv__(other)
|
|
939
961
|
|
|
940
962
|
def __rtruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
941
|
-
return
|
|
963
|
+
return super().__rtruediv__(other)
|
|
942
964
|
|
|
943
965
|
def __rmod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
944
|
-
return
|
|
966
|
+
return super().__rmod__(other)
|
|
945
967
|
|
|
946
968
|
def __rpow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
947
|
-
return
|
|
969
|
+
return super().__rpow__(other)
|
|
948
970
|
|
|
949
971
|
def __rand__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
950
|
-
return
|
|
972
|
+
return super().__rand__(other)
|
|
951
973
|
|
|
952
974
|
def __ror__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
953
|
-
return
|
|
975
|
+
return super().__ror__(other)
|
|
954
976
|
|
|
955
977
|
def __rxor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
956
|
-
return
|
|
978
|
+
return super().__rxor__(other)
|
|
979
|
+
|
|
980
|
+
def __invert__(self) -> "MicroSeries":
|
|
981
|
+
return MicroSeries(super().__invert__(), weights=self.weights)
|
|
957
982
|
|
|
958
983
|
def sqrt(self) -> "MicroSeries":
|
|
959
984
|
sqrt_values = np.sqrt(self._values)
|
|
960
985
|
return MicroSeries(sqrt_values, index=self.index, weights=self.weights)
|
|
961
986
|
|
|
962
|
-
# comparators
|
|
963
|
-
|
|
964
|
-
def __lt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
965
|
-
return MicroSeries(super().__lt__(other), weights=self.weights)
|
|
966
|
-
|
|
967
|
-
def __le__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
968
|
-
return MicroSeries(super().__le__(other), weights=self.weights)
|
|
969
|
-
|
|
970
|
-
def __eq__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
971
|
-
return MicroSeries(super().__eq__(other), weights=self.weights)
|
|
972
|
-
|
|
973
|
-
def __ne__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
974
|
-
return MicroSeries(super().__ne__(other), weights=self.weights)
|
|
975
|
-
|
|
976
|
-
def __ge__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
977
|
-
return MicroSeries(super().__ge__(other), weights=self.weights)
|
|
978
|
-
|
|
979
|
-
def __gt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
980
|
-
return MicroSeries(super().__gt__(other), weights=self.weights)
|
|
981
|
-
|
|
982
987
|
# assignment operators
|
|
983
988
|
|
|
984
989
|
def __iadd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Variance estimation from replicate weights.
|
|
2
|
+
|
|
3
|
+
Many survey products publish a set of replicate weight vectors alongside the
|
|
4
|
+
main weight. Recomputing a statistic once per replicate and measuring the
|
|
5
|
+
spread gives a variance estimate without an analytic variance formula. Its
|
|
6
|
+
statistical validity depends on both the statistic and the replication design.
|
|
7
|
+
Nonsmooth statistics such as quantiles can require an appropriate replication
|
|
8
|
+
method or smoothing; these functions apply the supplied statistic directly.
|
|
9
|
+
|
|
10
|
+
The scale factor and centering convention must match the survey's replication
|
|
11
|
+
design. The default centers on the full-sample estimate; ``replicate-mean``
|
|
12
|
+
centering is also available. These estimators support common-factor replication
|
|
13
|
+
schemes, not arbitrary stratified jackknife or averaged-bootstrap designs that
|
|
14
|
+
require additional or replicate-specific factors.
|
|
15
|
+
|
|
16
|
+
Changing the main weights without corresponding design-consistent adjustments
|
|
17
|
+
to the replicate weights invalidates the original replicates. Calibration can
|
|
18
|
+
be valid when the required calibration is repeated appropriately for every
|
|
19
|
+
replicate.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from typing import Callable
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
import pandas as pd
|
|
28
|
+
|
|
29
|
+
# Scale applied to the sum of squared deviations from the selected center.
|
|
30
|
+
# R is the number of replicates.
|
|
31
|
+
METHOD_FACTORS = {
|
|
32
|
+
# Unstratified JK1 / common-factor delete-group jackknife: (R - 1) / R.
|
|
33
|
+
"jackknife": lambda r: (r - 1) / r,
|
|
34
|
+
# Balanced repeated replication: 1 / R.
|
|
35
|
+
"brr": lambda r: 1 / r,
|
|
36
|
+
# Bootstrap replicates: 1 / R.
|
|
37
|
+
"bootstrap": lambda r: 1 / r,
|
|
38
|
+
# Successive difference replication, as used for the ACS and CPS: 4 / R.
|
|
39
|
+
"successive-difference": lambda r: 4 / r,
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _fay_factor(r: int, fay_k: float) -> float:
|
|
44
|
+
"""Scale for Fay's variant of BRR, which perturbs rather than deletes."""
|
|
45
|
+
if not 0 <= fay_k < 1:
|
|
46
|
+
raise ValueError(f"fay_k must be in [0, 1), got {fay_k}")
|
|
47
|
+
return 1 / (r * (1 - fay_k) ** 2)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def replicate_variance(
|
|
51
|
+
series,
|
|
52
|
+
statistic: Callable,
|
|
53
|
+
replicate_weights: np.ndarray | pd.DataFrame,
|
|
54
|
+
method: str = "jackknife",
|
|
55
|
+
fay_k: float | None = None,
|
|
56
|
+
*,
|
|
57
|
+
centering: str = "full-sample",
|
|
58
|
+
) -> float:
|
|
59
|
+
"""Variance of ``statistic`` estimated from replicate weights.
|
|
60
|
+
|
|
61
|
+
Variance is the method's scale factor times the sum of squared deviations
|
|
62
|
+
of replicate estimates from the selected center. The supported factors
|
|
63
|
+
are ``(R - 1) / R`` for unstratified JK1 or common-factor delete-group
|
|
64
|
+
jackknife, ``1 / R`` for BRR and bootstrap, ``4 / R`` for successive
|
|
65
|
+
difference, and ``1 / (R * (1 - fay_k)**2)`` for Fay's BRR. Select the
|
|
66
|
+
factor and center specified by the survey; arbitrary stratified jackknife
|
|
67
|
+
and averaged-bootstrap schemes requiring other factors are unsupported.
|
|
68
|
+
Validity also depends on the statistic: nonsmooth quantiles can require
|
|
69
|
+
an appropriate replication method or smoothing of replicate estimates.
|
|
70
|
+
|
|
71
|
+
:param series: A MicroSeries. Its own weights give the point estimate.
|
|
72
|
+
:param statistic: Callable taking a MicroSeries and returning a float,
|
|
73
|
+
for example ``lambda s: s.gini()``.
|
|
74
|
+
:param replicate_weights: Array or frame of shape ``(len(series), R)``
|
|
75
|
+
in the same row order as ``series``. Rows are matched by position;
|
|
76
|
+
DataFrame index labels are ignored.
|
|
77
|
+
:param method: One of ``jackknife``, ``brr``, ``bootstrap``,
|
|
78
|
+
``successive-difference``, or ``fay`` (which requires ``fay_k``).
|
|
79
|
+
:param fay_k: Fay's perturbation constant, required when
|
|
80
|
+
``method="fay"``.
|
|
81
|
+
:param centering: ``full-sample`` (default) centers on ``statistic(series)``;
|
|
82
|
+
``replicate-mean`` centers on the mean of the replicate estimates.
|
|
83
|
+
This choice does not change the method's scale factor.
|
|
84
|
+
:returns: The estimated variance of the statistic.
|
|
85
|
+
"""
|
|
86
|
+
from microdf.microseries import MicroSeries
|
|
87
|
+
|
|
88
|
+
if centering not in ("full-sample", "replicate-mean"):
|
|
89
|
+
raise ValueError(
|
|
90
|
+
f"centering must be 'full-sample' or 'replicate-mean', got {centering!r}"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
weights = np.asarray(replicate_weights, dtype=float)
|
|
94
|
+
if weights.ndim != 2:
|
|
95
|
+
raise ValueError(
|
|
96
|
+
f"replicate_weights must be 2-dimensional, got shape {weights.shape}"
|
|
97
|
+
)
|
|
98
|
+
if weights.shape[0] != len(series):
|
|
99
|
+
raise ValueError(
|
|
100
|
+
f"replicate_weights has {weights.shape[0]} rows but the series has "
|
|
101
|
+
f"{len(series)}"
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
n_replicates = weights.shape[1]
|
|
105
|
+
if n_replicates < 2:
|
|
106
|
+
raise ValueError("At least two replicate weights are required")
|
|
107
|
+
|
|
108
|
+
if method == "fay":
|
|
109
|
+
if fay_k is None:
|
|
110
|
+
raise ValueError("method='fay' requires fay_k")
|
|
111
|
+
factor = _fay_factor(n_replicates, fay_k)
|
|
112
|
+
elif method in METHOD_FACTORS:
|
|
113
|
+
if fay_k is not None:
|
|
114
|
+
raise ValueError("fay_k applies only to method='fay'")
|
|
115
|
+
factor = METHOD_FACTORS[method](n_replicates)
|
|
116
|
+
else:
|
|
117
|
+
known = ", ".join(sorted([*METHOD_FACTORS, "fay"]))
|
|
118
|
+
raise ValueError(f"Unknown method {method!r}; expected one of {known}")
|
|
119
|
+
|
|
120
|
+
# Keep in-place callback transformations out of the caller and replicates.
|
|
121
|
+
center = (
|
|
122
|
+
float(statistic(series.copy(deep=True))) if centering == "full-sample" else None
|
|
123
|
+
)
|
|
124
|
+
values = series.array
|
|
125
|
+
index = series.index
|
|
126
|
+
|
|
127
|
+
estimates = []
|
|
128
|
+
for column in range(n_replicates):
|
|
129
|
+
replicate = MicroSeries(
|
|
130
|
+
values.copy(),
|
|
131
|
+
weights=weights[:, column].copy(),
|
|
132
|
+
index=index.copy(),
|
|
133
|
+
name=series.name,
|
|
134
|
+
dtype=series.dtype,
|
|
135
|
+
)
|
|
136
|
+
estimates.append(float(statistic(replicate)))
|
|
137
|
+
|
|
138
|
+
estimates = np.asarray(estimates)
|
|
139
|
+
if not np.all(np.isfinite(estimates)) or (
|
|
140
|
+
center is not None and not np.isfinite(center)
|
|
141
|
+
):
|
|
142
|
+
# Retain the original infinity/NaN propagation for nonfinite callbacks.
|
|
143
|
+
if centering == "replicate-mean":
|
|
144
|
+
center = float(np.mean(estimates))
|
|
145
|
+
return factor * float(np.sum(np.square(estimates - center)))
|
|
146
|
+
|
|
147
|
+
with np.errstate(over="ignore"):
|
|
148
|
+
if centering == "replicate-mean":
|
|
149
|
+
# Keep the common offset out of the mean so small differences are
|
|
150
|
+
# preserved even when the absolute mean is not representable.
|
|
151
|
+
deviations = estimates - estimates[0]
|
|
152
|
+
if not np.all(np.isfinite(deviations)):
|
|
153
|
+
return float("inf")
|
|
154
|
+
deviations -= np.mean(deviations)
|
|
155
|
+
else:
|
|
156
|
+
deviations = estimates - center
|
|
157
|
+
|
|
158
|
+
scale = np.max(np.abs(deviations))
|
|
159
|
+
if scale == 0:
|
|
160
|
+
return 0.0
|
|
161
|
+
if not np.isfinite(scale):
|
|
162
|
+
return float("inf")
|
|
163
|
+
|
|
164
|
+
scaled_squares = np.sum(np.square(deviations / scale))
|
|
165
|
+
# Restore the scale by its binary exponent after applying the factor.
|
|
166
|
+
# Squaring scale first could overflow, or underflow before Fay's factor
|
|
167
|
+
# brings a tiny squared deviation back into the representable range.
|
|
168
|
+
significand, exponent = np.frexp(scale)
|
|
169
|
+
with np.errstate(over="ignore"):
|
|
170
|
+
return float(np.ldexp(significand**2 * factor * scaled_squares, 2 * exponent))
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def replicate_standard_error(
|
|
174
|
+
series,
|
|
175
|
+
statistic: Callable,
|
|
176
|
+
replicate_weights: np.ndarray | pd.DataFrame,
|
|
177
|
+
method: str = "jackknife",
|
|
178
|
+
fay_k: float | None = None,
|
|
179
|
+
*,
|
|
180
|
+
centering: str = "full-sample",
|
|
181
|
+
) -> float:
|
|
182
|
+
"""Standard error of ``statistic``, the square root of its variance.
|
|
183
|
+
|
|
184
|
+
Takes the same arguments as :func:`replicate_variance`.
|
|
185
|
+
"""
|
|
186
|
+
return float(
|
|
187
|
+
np.sqrt(
|
|
188
|
+
replicate_variance(
|
|
189
|
+
series, statistic, replicate_weights, method, fay_k, centering=centering
|
|
190
|
+
)
|
|
191
|
+
)
|
|
192
|
+
)
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""Binary operations retain the calling Series' observation weights."""
|
|
2
|
+
|
|
3
|
+
import inspect
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from microdf import MicroSeries
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
ARITHMETIC = ["add", "sub", "mul", "truediv", "floordiv", "mod", "pow"]
|
|
13
|
+
LOGICAL = ["and", "or", "xor"]
|
|
14
|
+
COMPARISONS = ["lt", "le", "eq", "ne", "ge", "gt"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def assert_weighted_result(result, expected, source):
|
|
18
|
+
assert isinstance(result, MicroSeries)
|
|
19
|
+
pd.testing.assert_series_equal(pd.Series(result), expected)
|
|
20
|
+
expected_weights = (
|
|
21
|
+
source.weights
|
|
22
|
+
if source.index.equals(expected.index)
|
|
23
|
+
else source.weights.reindex(expected.index)
|
|
24
|
+
)
|
|
25
|
+
pd.testing.assert_series_equal(result.weights, expected_weights)
|
|
26
|
+
assert result.weights is not source.weights
|
|
27
|
+
assert result.sum() == expected.multiply(expected_weights).sum()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@pytest.mark.parametrize(
|
|
31
|
+
"method",
|
|
32
|
+
[f"__{prefix}{op}__" for op in ARITHMETIC + LOGICAL for prefix in ["", "r"]],
|
|
33
|
+
)
|
|
34
|
+
@pytest.mark.parametrize("weighted_other", [False, True])
|
|
35
|
+
def test_binary_operators_align_weights_with_labels(method, weighted_other):
|
|
36
|
+
source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x")
|
|
37
|
+
other = pd.Series([2, 1], index=["a", "b"], name="x")
|
|
38
|
+
if weighted_other:
|
|
39
|
+
other = MicroSeries(other, weights=[9, 1])
|
|
40
|
+
expected = getattr(pd.Series(source), method)(pd.Series(other))
|
|
41
|
+
|
|
42
|
+
result = getattr(source, method)(other)
|
|
43
|
+
|
|
44
|
+
assert_weighted_result(result, expected, source)
|
|
45
|
+
# Addition is 22 * 9 + 11 * 1 = 209, rather than the positional 121.
|
|
46
|
+
if method in ["__add__", "__radd__"]:
|
|
47
|
+
assert result.sum() == 209
|
|
48
|
+
result.weights.iloc[0] = 100
|
|
49
|
+
np.testing.assert_array_equal(source.weights, [1, 9])
|
|
50
|
+
if weighted_other:
|
|
51
|
+
np.testing.assert_array_equal(other.weights, [9, 1])
|
|
52
|
+
source.weights.iloc[1] = 200
|
|
53
|
+
assert result.weights.iloc[0] == 100
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@pytest.mark.parametrize(
|
|
57
|
+
"method",
|
|
58
|
+
ARITHMETIC + [f"r{op}" for op in ARITHMETIC] + COMPARISONS + ["div", "rdiv"],
|
|
59
|
+
)
|
|
60
|
+
@pytest.mark.parametrize("permuted", [False, True])
|
|
61
|
+
def test_named_binary_methods_use_calling_series_weights(method, permuted):
|
|
62
|
+
source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x")
|
|
63
|
+
other = MicroSeries(
|
|
64
|
+
[2, 1], index=["a", "b"] if permuted else ["b", "a"], weights=[5, 7], name="x"
|
|
65
|
+
)
|
|
66
|
+
expected = getattr(pd.Series(source), method)(pd.Series(other))
|
|
67
|
+
|
|
68
|
+
result = getattr(source, method)(other)
|
|
69
|
+
|
|
70
|
+
assert_weighted_result(result, expected, source)
|
|
71
|
+
# Inherited public methods keep the installed pandas API signatures.
|
|
72
|
+
assert inspect.signature(getattr(MicroSeries, method)) == inspect.signature(
|
|
73
|
+
getattr(pd.Series, method)
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@pytest.mark.parametrize("method", [f"__{op}__" for op in COMPARISONS])
|
|
78
|
+
def test_comparison_operators_keep_calling_series_weights_and_pandas_errors(method):
|
|
79
|
+
source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
|
|
80
|
+
other = MicroSeries([20, 10], index=source.index, weights=[5, 7])
|
|
81
|
+
expected = getattr(pd.Series(source), method)(pd.Series(other))
|
|
82
|
+
assert_weighted_result(getattr(source, method)(other), expected, source)
|
|
83
|
+
|
|
84
|
+
other.index = ["a", "b"]
|
|
85
|
+
with pytest.raises(ValueError) as pandas_error:
|
|
86
|
+
getattr(pd.Series(source), method)(pd.Series(other))
|
|
87
|
+
with pytest.raises(ValueError) as microdf_error:
|
|
88
|
+
getattr(source, method)(other)
|
|
89
|
+
assert str(microdf_error.value) == str(pandas_error.value)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
|
|
93
|
+
@pytest.mark.parametrize("operand", [3, [2, 1], np.array([2, 1])])
|
|
94
|
+
def test_scalar_and_array_binary_operands_keep_weights(method, operand):
|
|
95
|
+
source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
|
|
96
|
+
expected = getattr(pd.Series(source), method)(operand)
|
|
97
|
+
assert_weighted_result(getattr(source, method)(operand), expected, source)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
|
|
101
|
+
def test_matching_duplicate_indexes_keep_positional_weights(method):
|
|
102
|
+
source = MicroSeries([10, 20, 30], index=["a", "a", "b"], weights=[1, 9, 3])
|
|
103
|
+
other = MicroSeries([2, 1, 4], index=source.index, weights=[5, 7, 11])
|
|
104
|
+
expected = getattr(pd.Series(source), method)(pd.Series(other))
|
|
105
|
+
assert_weighted_result(getattr(source, method)(other), expected, source)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@pytest.mark.parametrize("method", ["__divmod__", "__rdivmod__", "divmod", "rdivmod"])
|
|
109
|
+
def test_divmod_results_keep_calling_series_weights(method):
|
|
110
|
+
source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
|
|
111
|
+
other = MicroSeries([3, 4], index=["a", "b"], weights=[5, 7])
|
|
112
|
+
expected = getattr(pd.Series(source), method)(pd.Series(other))
|
|
113
|
+
result = getattr(source, method)(other)
|
|
114
|
+
assert isinstance(result, tuple)
|
|
115
|
+
for actual, plain in zip(result, expected):
|
|
116
|
+
assert_weighted_result(actual, plain, source)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
|
|
120
|
+
@pytest.mark.parametrize(
|
|
121
|
+
"left_index,right_index",
|
|
122
|
+
[
|
|
123
|
+
(["b", "a"], ["a", "c"]),
|
|
124
|
+
(["b", "a", "b"], ["a", "b", "b"]),
|
|
125
|
+
],
|
|
126
|
+
ids=["new-rows", "ambiguous-duplicates"],
|
|
127
|
+
)
|
|
128
|
+
def test_binary_operations_reject_unknown_row_weights(method, left_index, right_index):
|
|
129
|
+
source = MicroSeries(
|
|
130
|
+
range(len(left_index)), index=left_index, weights=range(1, len(left_index) + 1)
|
|
131
|
+
)
|
|
132
|
+
other = MicroSeries(
|
|
133
|
+
range(len(right_index)),
|
|
134
|
+
index=right_index,
|
|
135
|
+
weights=range(4, len(right_index) + 4),
|
|
136
|
+
)
|
|
137
|
+
with pytest.raises(ValueError, match="weights"):
|
|
138
|
+
getattr(source, method)(other)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@pytest.mark.parametrize("method", ["add", "rsub", "lt"])
|
|
142
|
+
def test_named_binary_arguments_preserve_pandas_values_and_errors(method):
|
|
143
|
+
index = pd.MultiIndex.from_tuples([("b", 2), ("a", 1)], names=["group", "row"])
|
|
144
|
+
source = MicroSeries([np.nan, 20], index=index, weights=[1, 9])
|
|
145
|
+
other = pd.Series([2, 1], index=pd.Index(["a", "b"], name="group"))
|
|
146
|
+
kwargs = {"level": "group", "fill_value": 0, "axis": "index"}
|
|
147
|
+
expected = getattr(pd.Series(source), method)(other, **kwargs)
|
|
148
|
+
assert_weighted_result(getattr(source, method)(other, **kwargs), expected, source)
|
|
149
|
+
for args, options in [
|
|
150
|
+
((other,), {"axis": 1}),
|
|
151
|
+
(([1],), {}),
|
|
152
|
+
((other,), {"unknown": True}),
|
|
153
|
+
]:
|
|
154
|
+
with pytest.raises((TypeError, ValueError)) as pandas_error:
|
|
155
|
+
getattr(pd.Series(source), method)(*args, **options)
|
|
156
|
+
with pytest.raises(type(pandas_error.value)) as microdf_error:
|
|
157
|
+
getattr(source, method)(*args, **options)
|
|
158
|
+
# pandas identifies the concrete subclass in invalid-axis messages.
|
|
159
|
+
expected_error = str(pandas_error.value).replace(
|
|
160
|
+
"object type Series", "object type MicroSeries"
|
|
161
|
+
)
|
|
162
|
+
assert str(microdf_error.value) == expected_error
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
@pytest.mark.parametrize(
|
|
166
|
+
"operation",
|
|
167
|
+
[
|
|
168
|
+
lambda plain, weighted: plain + weighted,
|
|
169
|
+
lambda plain, weighted: plain - weighted,
|
|
170
|
+
lambda plain, weighted: plain * weighted,
|
|
171
|
+
lambda plain, weighted: plain / weighted,
|
|
172
|
+
lambda plain, weighted: plain // weighted,
|
|
173
|
+
lambda plain, weighted: plain % weighted,
|
|
174
|
+
lambda plain, weighted: plain**weighted,
|
|
175
|
+
lambda plain, weighted: plain & weighted,
|
|
176
|
+
lambda plain, weighted: plain | weighted,
|
|
177
|
+
lambda plain, weighted: plain ^ weighted,
|
|
178
|
+
],
|
|
179
|
+
ids=ARITHMETIC + LOGICAL,
|
|
180
|
+
)
|
|
181
|
+
@pytest.mark.parametrize("indexes", ["matching", "permuted", "duplicates"])
|
|
182
|
+
def test_plain_series_left_expressions_preserve_weighted_dispatch(operation, indexes):
|
|
183
|
+
index = ["a", "a"] if indexes == "duplicates" else ["b", "a"]
|
|
184
|
+
source = MicroSeries([10, 20], index=index, weights=[1, 9], name="x")
|
|
185
|
+
other_index = ["a", "b"] if indexes == "permuted" else index
|
|
186
|
+
other = pd.Series([2, 1], index=other_index, name="x")
|
|
187
|
+
expected = operation(other, pd.Series(source))
|
|
188
|
+
|
|
189
|
+
result = operation(other, source)
|
|
190
|
+
|
|
191
|
+
assert_weighted_result(result, expected, source)
|
|
192
|
+
result.weights.iloc[0] = 100
|
|
193
|
+
np.testing.assert_array_equal(source.weights, [1, 9])
|
|
194
|
+
source.weights.iloc[1] = 200
|
|
195
|
+
assert result.weights.iloc[0] == 100
|
{microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_dataframe_weight_storage.py
RENAMED
|
@@ -59,3 +59,50 @@ def test_stored_weight_edits_do_not_change_weight_column(dtype):
|
|
|
59
59
|
|
|
60
60
|
np.testing.assert_array_equal(df["w"], [1, 2])
|
|
61
61
|
assert df.sum()["x"] == 10 * 100 + 20 * 2
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@pytest.mark.parametrize("drop", [False, True])
|
|
65
|
+
@pytest.mark.parametrize("inplace", [False, True])
|
|
66
|
+
@pytest.mark.parametrize(
|
|
67
|
+
"index,level",
|
|
68
|
+
[
|
|
69
|
+
(pd.Index(["b", "a", "a"], name="row"), None),
|
|
70
|
+
(
|
|
71
|
+
pd.MultiIndex.from_tuples(
|
|
72
|
+
[("b", 2), ("a", 1), ("a", 1)], names=["group", "row"]
|
|
73
|
+
),
|
|
74
|
+
None,
|
|
75
|
+
),
|
|
76
|
+
(
|
|
77
|
+
pd.MultiIndex.from_tuples(
|
|
78
|
+
[("b", 2), ("a", 1), ("a", 1)], names=["group", "row"]
|
|
79
|
+
),
|
|
80
|
+
"group",
|
|
81
|
+
),
|
|
82
|
+
],
|
|
83
|
+
)
|
|
84
|
+
def test_reset_index_owns_independently_mutable_weights(index, level, drop, inplace):
|
|
85
|
+
source = mdf.MicroDataFrame({"x": [10, 20, 30]}, index=index, weights=[1, 9, 3])
|
|
86
|
+
original_weights = source.weights
|
|
87
|
+
expected = pd.DataFrame(source).reset_index(level=level, drop=drop)
|
|
88
|
+
|
|
89
|
+
result = source.reset_index(level=level, drop=drop, inplace=inplace)
|
|
90
|
+
|
|
91
|
+
if inplace:
|
|
92
|
+
assert result is None
|
|
93
|
+
result = source
|
|
94
|
+
assert isinstance(result, mdf.MicroDataFrame)
|
|
95
|
+
pd.testing.assert_frame_equal(pd.DataFrame(result), expected)
|
|
96
|
+
pd.testing.assert_series_equal(
|
|
97
|
+
result.weights, pd.Series([1.0, 9.0, 3.0], index=expected.index)
|
|
98
|
+
)
|
|
99
|
+
assert result.x.sum() == 10 * 1 + 20 * 9 + 30 * 3
|
|
100
|
+
assert result.weights is not original_weights
|
|
101
|
+
result.weights.iloc[0] = 100
|
|
102
|
+
np.testing.assert_array_equal(original_weights, [1, 9, 3])
|
|
103
|
+
assert result.x.sum() == 10 * 100 + 20 * 9 + 30 * 3
|
|
104
|
+
if not inplace:
|
|
105
|
+
assert source.x.sum() == 10 * 1 + 20 * 9 + 30 * 3
|
|
106
|
+
original_weights.iloc[1] = 200
|
|
107
|
+
np.testing.assert_array_equal(result.weights, [100, 9, 3])
|
|
108
|
+
assert result.x.sum() == 10 * 100 + 20 * 9 + 30 * 3
|
|
@@ -0,0 +1,544 @@
|
|
|
1
|
+
"""Variance estimation from replicate weights.
|
|
2
|
+
|
|
3
|
+
Recomputing a statistic once per replicate weight vector and measuring the
|
|
4
|
+
spread gives a variance estimate for statistics whose analytic variance is
|
|
5
|
+
awkward, such as the Gini coefficient or a quantile.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pandas as pd
|
|
10
|
+
import pytest
|
|
11
|
+
|
|
12
|
+
from microdf import MicroSeries, replicate_standard_error, replicate_variance
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@pytest.fixture
|
|
16
|
+
def series_and_replicates():
|
|
17
|
+
rng = np.random.default_rng(0)
|
|
18
|
+
values = rng.lognormal(mean=10, sigma=1.0, size=500)
|
|
19
|
+
weights = np.full(500, 40.0)
|
|
20
|
+
replicates = weights[:, None] * rng.poisson(1.0, size=(500, 200))
|
|
21
|
+
return MicroSeries(values, weights=weights), replicates
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_matches_the_analytic_standard_error_of_a_weighted_mean(
|
|
25
|
+
series_and_replicates,
|
|
26
|
+
):
|
|
27
|
+
"""The mean has a closed form, so it is the case we can check exactly."""
|
|
28
|
+
series, replicates = series_and_replicates
|
|
29
|
+
|
|
30
|
+
replicate_se = replicate_standard_error(
|
|
31
|
+
series, lambda s: s.mean(), replicates, method="bootstrap"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
values = np.asarray(series)
|
|
35
|
+
analytic_se = np.sqrt(np.var(values, ddof=1) / len(values))
|
|
36
|
+
|
|
37
|
+
assert replicate_se == pytest.approx(analytic_se, rel=0.15)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_works_for_statistics_with_no_analytic_variance(
|
|
41
|
+
series_and_replicates,
|
|
42
|
+
):
|
|
43
|
+
"""The point of the method: Gini and quantiles come out like anything
|
|
44
|
+
else."""
|
|
45
|
+
series, replicates = series_and_replicates
|
|
46
|
+
|
|
47
|
+
for statistic in (lambda s: s.gini(), lambda s: s.median()):
|
|
48
|
+
se = replicate_standard_error(series, statistic, replicates, method="bootstrap")
|
|
49
|
+
assert se > 0
|
|
50
|
+
assert np.isfinite(se)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@pytest.mark.parametrize(
|
|
54
|
+
"method,expected_factor",
|
|
55
|
+
[
|
|
56
|
+
("jackknife", 199 / 200),
|
|
57
|
+
("brr", 1 / 200),
|
|
58
|
+
("bootstrap", 1 / 200),
|
|
59
|
+
("successive-difference", 4 / 200),
|
|
60
|
+
],
|
|
61
|
+
)
|
|
62
|
+
def test_each_method_applies_its_own_scale(
|
|
63
|
+
series_and_replicates, method, expected_factor
|
|
64
|
+
):
|
|
65
|
+
"""The scale factor is what distinguishes the replication schemes."""
|
|
66
|
+
series, replicates = series_and_replicates
|
|
67
|
+
|
|
68
|
+
variance = replicate_variance(series, lambda s: s.mean(), replicates, method=method)
|
|
69
|
+
reference = replicate_variance(series, lambda s: s.mean(), replicates, method="brr")
|
|
70
|
+
|
|
71
|
+
assert variance == pytest.approx(reference * expected_factor * 200, rel=1e-9)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_fay_requires_and_uses_its_constant(series_and_replicates):
|
|
75
|
+
series, replicates = series_and_replicates
|
|
76
|
+
|
|
77
|
+
with pytest.raises(ValueError, match="requires fay_k"):
|
|
78
|
+
replicate_variance(series, lambda s: s.mean(), replicates, method="fay")
|
|
79
|
+
|
|
80
|
+
fay = replicate_variance(
|
|
81
|
+
series, lambda s: s.mean(), replicates, method="fay", fay_k=0.5
|
|
82
|
+
)
|
|
83
|
+
brr = replicate_variance(series, lambda s: s.mean(), replicates, method="brr")
|
|
84
|
+
|
|
85
|
+
# 1 / (R (1 - k)^2) against 1 / R, so a factor of four at k = 0.5.
|
|
86
|
+
assert fay == pytest.approx(brr * 4, rel=1e-9)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_rejects_input_that_cannot_be_right(series_and_replicates):
|
|
90
|
+
series, replicates = series_and_replicates
|
|
91
|
+
|
|
92
|
+
with pytest.raises(ValueError, match="2-dimensional"):
|
|
93
|
+
replicate_variance(series, lambda s: s.mean(), np.ones(500))
|
|
94
|
+
|
|
95
|
+
with pytest.raises(ValueError, match="rows but the series has"):
|
|
96
|
+
replicate_variance(series, lambda s: s.mean(), np.ones((499, 10)))
|
|
97
|
+
|
|
98
|
+
with pytest.raises(ValueError, match="At least two"):
|
|
99
|
+
replicate_variance(series, lambda s: s.mean(), np.ones((500, 1)))
|
|
100
|
+
|
|
101
|
+
with pytest.raises(ValueError, match="Unknown method"):
|
|
102
|
+
replicate_variance(series, lambda s: s.mean(), replicates, method="nonsense")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@pytest.mark.parametrize("dtype", ["object", "category", "string"])
|
|
106
|
+
def test_replicates_preserve_categorical_values_and_metadata(dtype):
|
|
107
|
+
series = MicroSeries(
|
|
108
|
+
["employed", "unemployed", "employed", None],
|
|
109
|
+
weights=[1, 1, 1, 1],
|
|
110
|
+
index=["a", "b", "c", "d"],
|
|
111
|
+
name="employment",
|
|
112
|
+
dtype=dtype,
|
|
113
|
+
)
|
|
114
|
+
replicates = np.array([[2, 2, 0, 0], [0, 0, 2, 2], [2, 0, 2, 0], [0, 2, 0, 2]])
|
|
115
|
+
original = pd.Series(series).copy(deep=True)
|
|
116
|
+
original_weights = series.weights.copy(deep=True)
|
|
117
|
+
original_replicates = replicates.copy()
|
|
118
|
+
|
|
119
|
+
def count(sample):
|
|
120
|
+
assert sample.dtype == series.dtype
|
|
121
|
+
assert sample.name == "employment"
|
|
122
|
+
pd.testing.assert_index_equal(sample.index, series.index)
|
|
123
|
+
return sample.count()
|
|
124
|
+
|
|
125
|
+
# Full count is 3; replicate counts are 4, 2, 4, 2.
|
|
126
|
+
assert replicate_variance(series, count, replicates, method="brr") == 1
|
|
127
|
+
pd.testing.assert_series_equal(pd.Series(series), original)
|
|
128
|
+
pd.testing.assert_series_equal(series.weights, original_weights)
|
|
129
|
+
np.testing.assert_array_equal(replicates, original_replicates)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@pytest.mark.parametrize("dtype", ["bool", "boolean"])
|
|
133
|
+
def test_replicates_preserve_boolean_domain_counts(dtype):
|
|
134
|
+
series = MicroSeries([True, False, True, False], weights=[1, 1, 1, 1], dtype=dtype)
|
|
135
|
+
replicates = np.array([[2, 2, 0, 0], [0, 0, 2, 2], [2, 0, 2, 0], [0, 2, 0, 2]])
|
|
136
|
+
# Complement counts are 0, 2, 2, 4, around the full-sample count of 2.
|
|
137
|
+
assert (
|
|
138
|
+
replicate_variance(
|
|
139
|
+
series, lambda sample: (~sample).sum(), replicates, method="brr"
|
|
140
|
+
)
|
|
141
|
+
== 2
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
@pytest.mark.parametrize("dtype", ["int64", "Int64"])
|
|
146
|
+
def test_replicates_preserve_exact_large_integer_categories(dtype):
|
|
147
|
+
category = 2**53
|
|
148
|
+
series = MicroSeries([category, category + 1], weights=[1, 2], dtype=dtype)
|
|
149
|
+
replicates = np.array([[1, 1], [2, 2]])
|
|
150
|
+
# Identical weights must keep the weighted category count exactly 1.
|
|
151
|
+
assert (
|
|
152
|
+
replicate_variance(
|
|
153
|
+
series, lambda sample: (sample == category).sum(), replicates, method="brr"
|
|
154
|
+
)
|
|
155
|
+
== 0
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
@pytest.mark.parametrize(
|
|
160
|
+
"method,fay_k,full_sample_variance,replicate_mean_variance",
|
|
161
|
+
[
|
|
162
|
+
("jackknife", None, 1088 / 3, 3136 / 9),
|
|
163
|
+
("brr", None, 544 / 3, 1568 / 9),
|
|
164
|
+
("bootstrap", None, 544 / 3, 1568 / 9),
|
|
165
|
+
("successive-difference", None, 2176 / 3, 6272 / 9),
|
|
166
|
+
("fay", 0.5, 2176 / 3, 6272 / 9),
|
|
167
|
+
],
|
|
168
|
+
)
|
|
169
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
170
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
171
|
+
def test_centering_uses_exact_nonlinear_replicate_estimates(
|
|
172
|
+
method, fay_k, full_sample_variance, replicate_mean_variance, centering, api
|
|
173
|
+
):
|
|
174
|
+
series = MicroSeries([1, 3], weights=[1, 1])
|
|
175
|
+
replicates = np.array([[2, 1, 0], [0, 1, 2]])
|
|
176
|
+
|
|
177
|
+
# Squared totals: full sample 16; replicates 4, 16, 36; replicate mean 56/3.
|
|
178
|
+
# Squared deviations sum to 544 from 16 and 1568/3 from 56/3.
|
|
179
|
+
def squared_total(sample):
|
|
180
|
+
return sample.sum() ** 2
|
|
181
|
+
|
|
182
|
+
expected = (
|
|
183
|
+
full_sample_variance if centering == "full-sample" else replicate_mean_variance
|
|
184
|
+
)
|
|
185
|
+
kwargs = {"method": method, "fay_k": fay_k, "centering": centering}
|
|
186
|
+
if api == "variance":
|
|
187
|
+
observed = replicate_variance(series, squared_total, replicates, **kwargs)
|
|
188
|
+
elif api == "standard_error":
|
|
189
|
+
observed = replicate_standard_error(series, squared_total, replicates, **kwargs)
|
|
190
|
+
expected = np.sqrt(expected)
|
|
191
|
+
else:
|
|
192
|
+
observed = series.replicate_standard_error(squared_total, replicates, **kwargs)
|
|
193
|
+
expected = np.sqrt(expected)
|
|
194
|
+
assert observed == pytest.approx(expected)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def test_default_centering_retains_full_sample_estimate():
|
|
198
|
+
series = MicroSeries([1, 3], weights=[1, 1])
|
|
199
|
+
replicates = np.array([[2, 1, 0], [0, 1, 2]])
|
|
200
|
+
expected_variance = 544 / 3
|
|
201
|
+
|
|
202
|
+
def squared_total(sample):
|
|
203
|
+
return sample.sum() ** 2
|
|
204
|
+
|
|
205
|
+
assert replicate_variance(
|
|
206
|
+
series, squared_total, replicates, method="bootstrap"
|
|
207
|
+
) == pytest.approx(expected_variance)
|
|
208
|
+
assert replicate_standard_error(
|
|
209
|
+
series, squared_total, replicates, method="bootstrap"
|
|
210
|
+
) == pytest.approx(np.sqrt(expected_variance))
|
|
211
|
+
assert series.replicate_standard_error(
|
|
212
|
+
squared_total, replicates, method="bootstrap"
|
|
213
|
+
) == pytest.approx(np.sqrt(expected_variance))
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
217
|
+
def test_rejects_unknown_centering_before_calling_statistic(api):
|
|
218
|
+
series = MicroSeries([1, 3], weights=[1, 1])
|
|
219
|
+
replicates = np.array([[2, 0], [0, 2]])
|
|
220
|
+
|
|
221
|
+
def unexpected_statistic(sample):
|
|
222
|
+
raise AssertionError("Invalid centering must be rejected first")
|
|
223
|
+
|
|
224
|
+
with pytest.raises(ValueError, match="centering"):
|
|
225
|
+
if api == "variance":
|
|
226
|
+
replicate_variance(
|
|
227
|
+
series, unexpected_statistic, replicates, centering="unknown"
|
|
228
|
+
)
|
|
229
|
+
elif api == "standard_error":
|
|
230
|
+
replicate_standard_error(
|
|
231
|
+
series, unexpected_statistic, replicates, centering="unknown"
|
|
232
|
+
)
|
|
233
|
+
else:
|
|
234
|
+
series.replicate_standard_error(
|
|
235
|
+
unexpected_statistic, replicates, centering="unknown"
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def test_replicate_weight_frames_use_positional_rows():
|
|
240
|
+
series = MicroSeries([1, 3], weights=[1, 1], index=["a", "b"])
|
|
241
|
+
replicates = pd.DataFrame([[2, 0], [0, 2]], index=["b", "a"])
|
|
242
|
+
# Position-defined means are 1 and 3 around the full-sample mean of 2.
|
|
243
|
+
assert (
|
|
244
|
+
replicate_variance(
|
|
245
|
+
series, lambda sample: sample.mean(), replicates, method="brr"
|
|
246
|
+
)
|
|
247
|
+
== 1
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _replication_result(series, statistic, replicates, api, **kwargs):
|
|
252
|
+
if api == "variance":
|
|
253
|
+
return replicate_variance(series, statistic, replicates, **kwargs)
|
|
254
|
+
if api == "standard_error":
|
|
255
|
+
return replicate_standard_error(series, statistic, replicates, **kwargs)
|
|
256
|
+
return series.replicate_standard_error(statistic, replicates, **kwargs)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
@pytest.mark.parametrize(
|
|
260
|
+
"method,fay_k,amplitude,expected_variance",
|
|
261
|
+
[
|
|
262
|
+
("jackknife", None, 5e153, 1.75e308),
|
|
263
|
+
("brr", None, 1e154, 1e308),
|
|
264
|
+
("bootstrap", None, 1e154, 1e308),
|
|
265
|
+
("successive-difference", None, 5e153, 1e308),
|
|
266
|
+
("fay", 0.5, 5e153, 1e308),
|
|
267
|
+
],
|
|
268
|
+
)
|
|
269
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
270
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
271
|
+
def test_large_finite_replicate_variance(
|
|
272
|
+
method, fay_k, amplitude, expected_variance, centering, api
|
|
273
|
+
):
|
|
274
|
+
series = MicroSeries([0.0, 2 * amplitude], weights=[1, 1])
|
|
275
|
+
replicates = np.tile([[2, 0], [0, 2]], (1, 4))
|
|
276
|
+
# Eight deviations are +/- amplitude. The factors are 7/8, 1/8 or 1/2.
|
|
277
|
+
# Every method has a finite variance despite an overflowing raw sum.
|
|
278
|
+
expected = expected_variance if api == "variance" else np.sqrt(expected_variance)
|
|
279
|
+
observed = _replication_result(
|
|
280
|
+
series,
|
|
281
|
+
lambda sample: sample.mean(),
|
|
282
|
+
replicates,
|
|
283
|
+
api,
|
|
284
|
+
method=method,
|
|
285
|
+
fay_k=fay_k,
|
|
286
|
+
centering=centering,
|
|
287
|
+
)
|
|
288
|
+
assert observed == pytest.approx(expected, rel=1e-14, abs=0)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
292
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
293
|
+
def test_identical_large_replicates_have_zero_variance(centering, api):
|
|
294
|
+
series = MicroSeries([1e308, 1e308], weights=[1, 1])
|
|
295
|
+
replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]])
|
|
296
|
+
# The median remains finite even though summing the four estimates overflows.
|
|
297
|
+
assert (
|
|
298
|
+
_replication_result(
|
|
299
|
+
series,
|
|
300
|
+
lambda sample: sample.median(),
|
|
301
|
+
replicates,
|
|
302
|
+
api,
|
|
303
|
+
method="brr",
|
|
304
|
+
centering=centering,
|
|
305
|
+
)
|
|
306
|
+
== 0
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
@pytest.mark.parametrize(
|
|
311
|
+
"method,fay_k,variance_multiplier",
|
|
312
|
+
[
|
|
313
|
+
("jackknife", None, 3),
|
|
314
|
+
("brr", None, 1),
|
|
315
|
+
("bootstrap", None, 1),
|
|
316
|
+
("successive-difference", None, 4),
|
|
317
|
+
("fay", 0.5, 4),
|
|
318
|
+
],
|
|
319
|
+
)
|
|
320
|
+
@pytest.mark.parametrize("offset", [0.0, float(2**53)])
|
|
321
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
322
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
323
|
+
def test_replicate_centering_preserves_small_differences(
|
|
324
|
+
method, fay_k, variance_multiplier, offset, centering, api
|
|
325
|
+
):
|
|
326
|
+
series = MicroSeries([offset, offset + 2], weights=[1, 1])
|
|
327
|
+
replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]])
|
|
328
|
+
# The exact replicate mean has deviations +/-1, including at 2**53.
|
|
329
|
+
# Full-sample centering must retain the callback's rounded mean at 2**53,
|
|
330
|
+
# so its deviations are 0 and 2, giving twice the centered variance.
|
|
331
|
+
expected = variance_multiplier
|
|
332
|
+
if offset and centering == "full-sample":
|
|
333
|
+
expected *= 2
|
|
334
|
+
if api != "variance":
|
|
335
|
+
expected = np.sqrt(expected)
|
|
336
|
+
observed = _replication_result(
|
|
337
|
+
series,
|
|
338
|
+
lambda sample: sample.mean(),
|
|
339
|
+
replicates,
|
|
340
|
+
api,
|
|
341
|
+
method=method,
|
|
342
|
+
fay_k=fay_k,
|
|
343
|
+
centering=centering,
|
|
344
|
+
)
|
|
345
|
+
assert observed == pytest.approx(expected, rel=1e-14, abs=0)
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
@pytest.mark.parametrize(
|
|
349
|
+
"amplitude,method,fay_k,expected_variance",
|
|
350
|
+
[
|
|
351
|
+
(2.0**-537, "brr", None, 2.0**-1074),
|
|
352
|
+
(2.0**-550, "fay", 1 - 2.0**-53, 2.0**-994),
|
|
353
|
+
],
|
|
354
|
+
)
|
|
355
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
356
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
357
|
+
def test_small_deviations_keep_representable_variance(
|
|
358
|
+
amplitude, method, fay_k, expected_variance, centering, api
|
|
359
|
+
):
|
|
360
|
+
series = MicroSeries([-amplitude, amplitude], weights=[1, 1])
|
|
361
|
+
replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]])
|
|
362
|
+
# BRR variance is amplitude**2. Fay's factor multiplies that by 2**106;
|
|
363
|
+
# it must be applied before rounding the initially unrepresentable square.
|
|
364
|
+
expected = expected_variance if api == "variance" else np.sqrt(expected_variance)
|
|
365
|
+
observed = _replication_result(
|
|
366
|
+
series,
|
|
367
|
+
lambda sample: sample.mean(),
|
|
368
|
+
replicates,
|
|
369
|
+
api,
|
|
370
|
+
method=method,
|
|
371
|
+
fay_k=fay_k,
|
|
372
|
+
centering=centering,
|
|
373
|
+
)
|
|
374
|
+
assert observed == expected
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
378
|
+
def test_finite_replicates_with_unrepresentable_variance(centering):
|
|
379
|
+
series = MicroSeries([-1e308, 1e308], weights=[1, 1])
|
|
380
|
+
replicates = np.array([[1, 0], [0, 1]])
|
|
381
|
+
# A weighted total gives finite estimates +/-1e308 and a zero point estimate.
|
|
382
|
+
with np.errstate(over="ignore", invalid="ignore"):
|
|
383
|
+
observed = replicate_variance(
|
|
384
|
+
series,
|
|
385
|
+
lambda sample: sample.sum(),
|
|
386
|
+
replicates,
|
|
387
|
+
method="brr",
|
|
388
|
+
centering=centering,
|
|
389
|
+
)
|
|
390
|
+
assert observed == np.inf
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
@pytest.mark.parametrize(
|
|
394
|
+
"point,estimates,centering,expected",
|
|
395
|
+
[
|
|
396
|
+
(0.0, [0.0, np.inf], "full-sample", np.inf),
|
|
397
|
+
(np.inf, [0.0, 1.0], "full-sample", np.inf),
|
|
398
|
+
(np.inf, [0.0, np.inf], "full-sample", np.nan),
|
|
399
|
+
(np.nan, [0.0, 1.0], "full-sample", np.nan),
|
|
400
|
+
(0.0, [0.0, np.nan], "full-sample", np.nan),
|
|
401
|
+
(0.0, [0.0, np.inf], "replicate-mean", np.nan),
|
|
402
|
+
(0.0, [np.inf, -np.inf], "replicate-mean", np.nan),
|
|
403
|
+
(0.0, [0.0, np.nan], "replicate-mean", np.nan),
|
|
404
|
+
],
|
|
405
|
+
)
|
|
406
|
+
def test_nonfinite_callback_results_keep_existing_behavior(
|
|
407
|
+
point, estimates, centering, expected
|
|
408
|
+
):
|
|
409
|
+
series = MicroSeries([1, 2], weights=[1, 1])
|
|
410
|
+
replicates = np.array([[2, 0], [0, 2]])
|
|
411
|
+
results = iter([point, *estimates] if centering == "full-sample" else estimates)
|
|
412
|
+
with np.errstate(over="ignore", invalid="ignore"):
|
|
413
|
+
observed = replicate_variance(
|
|
414
|
+
series,
|
|
415
|
+
lambda sample: next(results),
|
|
416
|
+
replicates,
|
|
417
|
+
method="brr",
|
|
418
|
+
centering=centering,
|
|
419
|
+
)
|
|
420
|
+
if np.isnan(expected):
|
|
421
|
+
assert np.isnan(observed)
|
|
422
|
+
else:
|
|
423
|
+
assert observed == expected
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
@pytest.mark.parametrize("inplace", [False, True])
|
|
427
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
428
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
429
|
+
def test_callback_transforms_each_sample_once(inplace, centering, api):
|
|
430
|
+
series = MicroSeries([120.0, 180.0], weights=[1, 1])
|
|
431
|
+
replicates = np.array([[2.0, 0.0], [0.0, 2.0]])
|
|
432
|
+
original_values = pd.Series(series).copy(deep=True)
|
|
433
|
+
original_weights = series.weights.copy(deep=True)
|
|
434
|
+
original_replicates = replicates.copy()
|
|
435
|
+
calls = []
|
|
436
|
+
|
|
437
|
+
def total_after_allowance(sample):
|
|
438
|
+
calls.append(sample.weights.tolist())
|
|
439
|
+
if inplace:
|
|
440
|
+
sample -= 100
|
|
441
|
+
return sample.sum()
|
|
442
|
+
return (sample - 100).sum()
|
|
443
|
+
|
|
444
|
+
# Transformed values are 20 and 80; totals are 100, 40 and 160.
|
|
445
|
+
# BRR variance is ((40 - 100)**2 + (160 - 100)**2) / 2 = 3600.
|
|
446
|
+
expected = 3600 if api == "variance" else 60
|
|
447
|
+
assert (
|
|
448
|
+
_replication_result(
|
|
449
|
+
series,
|
|
450
|
+
total_after_allowance,
|
|
451
|
+
replicates,
|
|
452
|
+
api,
|
|
453
|
+
method="brr",
|
|
454
|
+
centering=centering,
|
|
455
|
+
)
|
|
456
|
+
== expected
|
|
457
|
+
)
|
|
458
|
+
expected_calls = [[2.0, 0.0], [0.0, 2.0]]
|
|
459
|
+
if centering == "full-sample":
|
|
460
|
+
expected_calls.insert(0, [1.0, 1.0])
|
|
461
|
+
assert calls == expected_calls
|
|
462
|
+
pd.testing.assert_series_equal(pd.Series(series), original_values)
|
|
463
|
+
pd.testing.assert_series_equal(series.weights, original_weights)
|
|
464
|
+
np.testing.assert_array_equal(replicates, original_replicates)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
468
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
469
|
+
def test_callbacks_isolate_values_weights_and_metadata(centering, api):
|
|
470
|
+
series = MicroSeries(
|
|
471
|
+
[120, 180],
|
|
472
|
+
weights=[1, 1],
|
|
473
|
+
index=pd.Index(["a", "b"], name="person"),
|
|
474
|
+
name="income",
|
|
475
|
+
dtype="Int64",
|
|
476
|
+
)
|
|
477
|
+
replicates = np.array([[2.0, 0.0], [0.0, 2.0]])
|
|
478
|
+
original = series.copy(deep=True)
|
|
479
|
+
original_replicates = replicates.copy()
|
|
480
|
+
calls = []
|
|
481
|
+
|
|
482
|
+
def mutating_total(sample):
|
|
483
|
+
assert sample.tolist() == [120, 180]
|
|
484
|
+
assert sample.dtype == original.dtype
|
|
485
|
+
assert sample.name == "income"
|
|
486
|
+
pd.testing.assert_index_equal(sample.index, original.index)
|
|
487
|
+
calls.append(sample.weights.tolist())
|
|
488
|
+
sample -= 100
|
|
489
|
+
estimate = sample.sum()
|
|
490
|
+
sample.weights.iloc[:] = 7
|
|
491
|
+
sample.index = pd.Index(["x", "y"], name="changed")
|
|
492
|
+
sample.name = "changed"
|
|
493
|
+
return estimate
|
|
494
|
+
|
|
495
|
+
expected = 3600 if api == "variance" else 60
|
|
496
|
+
assert (
|
|
497
|
+
_replication_result(
|
|
498
|
+
series, mutating_total, replicates, api, method="brr", centering=centering
|
|
499
|
+
)
|
|
500
|
+
== expected
|
|
501
|
+
)
|
|
502
|
+
expected_calls = [[2.0, 0.0], [0.0, 2.0]]
|
|
503
|
+
if centering == "full-sample":
|
|
504
|
+
expected_calls.insert(0, [1.0, 1.0])
|
|
505
|
+
assert calls == expected_calls
|
|
506
|
+
pd.testing.assert_series_equal(pd.Series(series), pd.Series(original))
|
|
507
|
+
pd.testing.assert_series_equal(series.weights, original.weights)
|
|
508
|
+
np.testing.assert_array_equal(replicates, original_replicates)
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
|
|
512
|
+
@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
|
|
513
|
+
def test_callback_exception_preserves_inputs(centering, api):
|
|
514
|
+
series = MicroSeries(
|
|
515
|
+
[120.0, 180.0], weights=[1, 1], index=["a", "b"], name="income"
|
|
516
|
+
)
|
|
517
|
+
replicates = np.array([[2.0, 0.0], [0.0, 2.0]])
|
|
518
|
+
original = series.copy(deep=True)
|
|
519
|
+
original_replicates = replicates.copy()
|
|
520
|
+
callback_error = ValueError("statistic failed after mutation")
|
|
521
|
+
calls = []
|
|
522
|
+
|
|
523
|
+
def failing_statistic(sample):
|
|
524
|
+
calls.append(sample.weights.tolist())
|
|
525
|
+
sample -= 100
|
|
526
|
+
sample.weights.iloc[:] = 7
|
|
527
|
+
sample.index = ["x", "y"]
|
|
528
|
+
sample.name = "changed"
|
|
529
|
+
raise callback_error
|
|
530
|
+
|
|
531
|
+
with pytest.raises(ValueError, match="statistic failed") as raised:
|
|
532
|
+
_replication_result(
|
|
533
|
+
series,
|
|
534
|
+
failing_statistic,
|
|
535
|
+
replicates,
|
|
536
|
+
api,
|
|
537
|
+
method="brr",
|
|
538
|
+
centering=centering,
|
|
539
|
+
)
|
|
540
|
+
assert raised.value is callback_error
|
|
541
|
+
assert calls == ([[1.0, 1.0]] if centering == "full-sample" else [[2.0, 0.0]])
|
|
542
|
+
pd.testing.assert_series_equal(pd.Series(series), pd.Series(original))
|
|
543
|
+
pd.testing.assert_series_equal(series.weights, original.weights)
|
|
544
|
+
np.testing.assert_array_equal(replicates, original_replicates)
|
|
@@ -5,13 +5,16 @@ microdf/__init__.py
|
|
|
5
5
|
microdf/_weights.py
|
|
6
6
|
microdf/microdataframe.py
|
|
7
7
|
microdf/microseries.py
|
|
8
|
+
microdf/replication.py
|
|
8
9
|
microdf/tests/conftest.py
|
|
9
10
|
microdf/tests/test_aggregation_errors.py
|
|
11
|
+
microdf/tests/test_binary_weight_alignment.py
|
|
10
12
|
microdf/tests/test_dataframe_weight_storage.py
|
|
11
13
|
microdf/tests/test_microseries_dataframe.py
|
|
12
14
|
microdf/tests/test_nullify_weights_index.py
|
|
13
15
|
microdf/tests/test_pandas3_compatibility.py
|
|
14
16
|
microdf/tests/test_quantile_missing_values.py
|
|
17
|
+
microdf/tests/test_replication.py
|
|
15
18
|
microdf/tests/test_serialization.py
|
|
16
19
|
microdf/tests/test_sum_axes.py
|
|
17
20
|
microdf/tests/test_version_metadata.py
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|