microdf-python 1.4.1__tar.gz → 1.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. {microdf_python-1.4.1/microdf_python.egg-info → microdf_python-1.5.1}/PKG-INFO +1 -1
  2. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/__init__.py +4 -0
  3. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/microdataframe.py +5 -11
  4. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/microseries.py +70 -65
  5. microdf_python-1.5.1/microdf/replication.py +192 -0
  6. microdf_python-1.5.1/microdf/tests/test_binary_weight_alignment.py +195 -0
  7. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_dataframe_weight_storage.py +47 -0
  8. microdf_python-1.5.1/microdf/tests/test_replication.py +544 -0
  9. {microdf_python-1.4.1 → microdf_python-1.5.1/microdf_python.egg-info}/PKG-INFO +1 -1
  10. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/SOURCES.txt +3 -0
  11. {microdf_python-1.4.1 → microdf_python-1.5.1}/pyproject.toml +1 -1
  12. {microdf_python-1.4.1 → microdf_python-1.5.1}/LICENSE +0 -0
  13. {microdf_python-1.4.1 → microdf_python-1.5.1}/README.md +0 -0
  14. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/_weights.py +0 -0
  15. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/conftest.py +0 -0
  16. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_aggregation_errors.py +0 -0
  17. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_microseries_dataframe.py +0 -0
  18. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_nullify_weights_index.py +0 -0
  19. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_pandas3_compatibility.py +0 -0
  20. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_quantile_missing_values.py +0 -0
  21. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_serialization.py +0 -0
  22. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_sum_axes.py +0 -0
  23. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_version_metadata.py +0 -0
  24. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_weight_propagation.py +0 -0
  25. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf/tests/test_weighted_cov_corr.py +0 -0
  26. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/dependency_links.txt +0 -0
  27. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/requires.txt +0 -0
  28. {microdf_python-1.4.1 → microdf_python-1.5.1}/microdf_python.egg-info/top_level.txt +0 -0
  29. {microdf_python-1.4.1 → microdf_python-1.5.1}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.4.1
3
+ Version: 1.5.1
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -2,6 +2,7 @@ from importlib.metadata import PackageNotFoundError, version
2
2
 
3
3
  from .microdataframe import MicroDataFrame, MicroDataFrameGroupBy
4
4
  from .microseries import MicroSeries, MicroSeriesGroupBy
5
+ from .replication import replicate_standard_error, replicate_variance
5
6
 
6
7
  name = "microdf"
7
8
 
@@ -19,4 +20,7 @@ __all__ = [
19
20
  # microdataframe.py
20
21
  "MicroDataFrame",
21
22
  "MicroDataFrameGroupBy",
23
+ # replication.py
24
+ "replicate_variance",
25
+ "replicate_standard_error",
22
26
  ]
@@ -460,7 +460,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
460
460
  if inplace:
461
461
  # Snapshot weight *values* positionally — the index is about
462
462
  # to change and reset_index preserves row order.
463
- weight_values = np.asarray(self.weights.values, dtype=float)
463
+ weight_values = np.array(self.weights, dtype=float, copy=True)
464
464
  super().reset_index(
465
465
  level=level,
466
466
  drop=drop,
@@ -470,7 +470,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
470
470
  allow_duplicates=allow_duplicates,
471
471
  names=names,
472
472
  )
473
- self.weights = pd.Series(weight_values, index=self.index, dtype=float)
473
+ self.weights = weight_series(weight_values, self.index)
474
474
  self._link_all_weights()
475
475
  return None
476
476
  else:
@@ -483,15 +483,9 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
483
483
  allow_duplicates=allow_duplicates,
484
484
  names=names,
485
485
  )
486
- out = MicroDataFrame(res, weights=self.weights.values)
487
- # Ensure weights align to res.index (reset_index changes the
488
- # index but preserves row order, so pass values positionally).
489
- out.weights = pd.Series(
490
- np.asarray(self.weights.values, dtype=float),
491
- index=out.index,
492
- dtype=float,
493
- )
494
- return out
486
+ # Own a positional copy: reset_index changes labels but
487
+ # preserves row order.
488
+ return MicroDataFrame(res, weights=weight_series(self.weights, res.index))
495
489
 
496
490
  def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame":
497
491
  return super().copy(deep)
@@ -120,6 +120,16 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
120
120
  super().__finalize__(other, method=method, **kwargs)
121
121
  return finalize_weights(self, other, method, previous)
122
122
 
123
+ def _construct_result(self, *args, **kwargs):
124
+ # pandas has already aligned this Series before constructing a binary
125
+ # result. Retain its row weights, even when pandas 3 also finalizes
126
+ # metadata from the other operand. Delegate values and names to pandas.
127
+ result = super()._construct_result(*args, **kwargs)
128
+ if not isinstance(result, tuple):
129
+ result.weights = weight_series(self.weights, result.index)
130
+ # divmod constructs both tuple members through this same hook.
131
+ return result
132
+
123
133
  def __setattr__(self, name, value):
124
134
  weights = self.__dict__.get("weights") if name == "index" else None
125
135
  super().__setattr__(name, value)
@@ -579,6 +589,51 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
579
589
  """
580
590
  return self.quantile(0.5, skipna=skipna)
581
591
 
592
+ def replicate_standard_error(
593
+ self,
594
+ statistic: Callable,
595
+ replicate_weights,
596
+ method: str = "jackknife",
597
+ fay_k: Optional[float] = None,
598
+ *,
599
+ centering: str = "full-sample",
600
+ ) -> float:
601
+ """Standard error of ``statistic`` from a set of replicate weights.
602
+
603
+ Recomputes the statistic once per replicate and scales the spread by
604
+ the factor appropriate to how the replicates were built. Statistical
605
+ validity depends on both the statistic and the survey design.
606
+ Nonsmooth statistics such as quantiles can require an appropriate
607
+ replication method or smoothing of replicate estimates.
608
+
609
+ The factor and centering convention must match the survey design.
610
+ Supported schemes use a common factor: jackknife covers unstratified
611
+ JK1 or common-factor delete-group replication, not arbitrary stratified
612
+ jackknife. Averaged bootstrap requiring additional factors is not
613
+ supported. See :func:`microdf.replication.replicate_variance` for factors.
614
+
615
+ Changing main weights without corresponding design-consistent replicate
616
+ adjustments invalidates the original replicates. Calibration can be
617
+ valid when repeated appropriately for every replicate.
618
+
619
+ :param statistic: Callable taking a MicroSeries and returning a float, e.g.
620
+ ``lambda s: s.median()``.
621
+ :param replicate_weights: Array or frame of shape ``(len(self), R)``
622
+ in the same row order as this series. DataFrame labels are ignored.
623
+ :param method: ``jackknife``, ``brr``, ``bootstrap``,
624
+ ``successive-difference`` or ``fay``.
625
+ :param fay_k: Fay's perturbation constant, for ``method="fay"``.
626
+ :param centering: ``full-sample`` (default) centers on ``statistic(self)``;
627
+ ``replicate-mean`` centers on the mean of the replicate estimates.
628
+ The method's scale factor is unchanged.
629
+ :returns: The estimated standard error.
630
+ """
631
+ from microdf.replication import replicate_standard_error
632
+
633
+ return replicate_standard_error(
634
+ self, statistic, replicate_weights, method, fay_k, centering=centering
635
+ )
636
+
582
637
  @scalar_function
583
638
  def gini(self, negatives: Optional[str] = None) -> float:
584
639
  """Calculates Gini index.
@@ -890,95 +945,45 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
890
945
  def __getattr__(self, name: str) -> "MicroSeries":
891
946
  return MicroSeries(super().__getattr__(name), weights=self.weights)
892
947
 
893
- # operators
894
-
895
- def __add__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
896
- return MicroSeries(super().__add__(other), weights=self.weights)
897
-
898
- def __sub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
899
- return MicroSeries(super().__sub__(other), weights=self.weights)
900
-
901
- def __mul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
902
- return MicroSeries(super().__mul__(other), weights=self.weights)
903
-
904
- def __floordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
905
- return MicroSeries(super().__floordiv__(other), weights=self.weights)
906
-
907
- def __truediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
908
- return MicroSeries(super().__truediv__(other), weights=self.weights)
909
-
910
- def __mod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
911
- return MicroSeries(super().__mod__(other), weights=self.weights)
912
-
913
- def __pow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
914
- return MicroSeries(super().__pow__(other), weights=self.weights)
915
-
916
- def __xor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
917
- return MicroSeries(super().__xor__(other), weights=self.weights)
918
-
919
- def __and__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
920
- return MicroSeries(super().__and__(other), weights=self.weights)
921
-
922
- def __or__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
923
- return MicroSeries(super().__or__(other), weights=self.weights)
924
-
925
- def __invert__(self) -> "MicroSeries":
926
- return MicroSeries(super().__invert__(), weights=self.weights)
927
-
948
+ # Explicit reverse overrides give this subclass priority when a plain
949
+ # pandas Series is on the left. _construct_result retains aligned weights.
928
950
  def __radd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
929
- return MicroSeries(super().__radd__(other), weights=self.weights)
951
+ return super().__radd__(other)
930
952
 
931
953
  def __rsub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
932
- return MicroSeries(super().__rsub__(other), weights=self.weights)
954
+ return super().__rsub__(other)
933
955
 
934
956
  def __rmul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
935
- return MicroSeries(super().__rmul__(other), weights=self.weights)
957
+ return super().__rmul__(other)
936
958
 
937
959
  def __rfloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
938
- return MicroSeries(super().__rfloordiv__(other), weights=self.weights)
960
+ return super().__rfloordiv__(other)
939
961
 
940
962
  def __rtruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
941
- return MicroSeries(super().__rtruediv__(other), weights=self.weights)
963
+ return super().__rtruediv__(other)
942
964
 
943
965
  def __rmod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
944
- return MicroSeries(super().__rmod__(other), weights=self.weights)
966
+ return super().__rmod__(other)
945
967
 
946
968
  def __rpow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
947
- return MicroSeries(super().__rpow__(other), weights=self.weights)
969
+ return super().__rpow__(other)
948
970
 
949
971
  def __rand__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
950
- return MicroSeries(super().__rand__(other), weights=self.weights)
972
+ return super().__rand__(other)
951
973
 
952
974
  def __ror__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
953
- return MicroSeries(super().__ror__(other), weights=self.weights)
975
+ return super().__ror__(other)
954
976
 
955
977
  def __rxor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
956
- return MicroSeries(super().__rxor__(other), weights=self.weights)
978
+ return super().__rxor__(other)
979
+
980
+ def __invert__(self) -> "MicroSeries":
981
+ return MicroSeries(super().__invert__(), weights=self.weights)
957
982
 
958
983
  def sqrt(self) -> "MicroSeries":
959
984
  sqrt_values = np.sqrt(self._values)
960
985
  return MicroSeries(sqrt_values, index=self.index, weights=self.weights)
961
986
 
962
- # comparators
963
-
964
- def __lt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
965
- return MicroSeries(super().__lt__(other), weights=self.weights)
966
-
967
- def __le__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
968
- return MicroSeries(super().__le__(other), weights=self.weights)
969
-
970
- def __eq__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
971
- return MicroSeries(super().__eq__(other), weights=self.weights)
972
-
973
- def __ne__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
974
- return MicroSeries(super().__ne__(other), weights=self.weights)
975
-
976
- def __ge__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
977
- return MicroSeries(super().__ge__(other), weights=self.weights)
978
-
979
- def __gt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
980
- return MicroSeries(super().__gt__(other), weights=self.weights)
981
-
982
987
  # assignment operators
983
988
 
984
989
  def __iadd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
@@ -0,0 +1,192 @@
1
+ """Variance estimation from replicate weights.
2
+
3
+ Many survey products publish a set of replicate weight vectors alongside the
4
+ main weight. Recomputing a statistic once per replicate and measuring the
5
+ spread gives a variance estimate without an analytic variance formula. Its
6
+ statistical validity depends on both the statistic and the replication design.
7
+ Nonsmooth statistics such as quantiles can require an appropriate replication
8
+ method or smoothing; these functions apply the supplied statistic directly.
9
+
10
+ The scale factor and centering convention must match the survey's replication
11
+ design. The default centers on the full-sample estimate; ``replicate-mean``
12
+ centering is also available. These estimators support common-factor replication
13
+ schemes, not arbitrary stratified jackknife or averaged-bootstrap designs that
14
+ require additional or replicate-specific factors.
15
+
16
+ Changing the main weights without corresponding design-consistent adjustments
17
+ to the replicate weights invalidates the original replicates. Calibration can
18
+ be valid when the required calibration is repeated appropriately for every
19
+ replicate.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from typing import Callable
25
+
26
+ import numpy as np
27
+ import pandas as pd
28
+
29
+ # Scale applied to the sum of squared deviations from the selected center.
30
+ # R is the number of replicates.
31
+ METHOD_FACTORS = {
32
+ # Unstratified JK1 / common-factor delete-group jackknife: (R - 1) / R.
33
+ "jackknife": lambda r: (r - 1) / r,
34
+ # Balanced repeated replication: 1 / R.
35
+ "brr": lambda r: 1 / r,
36
+ # Bootstrap replicates: 1 / R.
37
+ "bootstrap": lambda r: 1 / r,
38
+ # Successive difference replication, as used for the ACS and CPS: 4 / R.
39
+ "successive-difference": lambda r: 4 / r,
40
+ }
41
+
42
+
43
+ def _fay_factor(r: int, fay_k: float) -> float:
44
+ """Scale for Fay's variant of BRR, which perturbs rather than deletes."""
45
+ if not 0 <= fay_k < 1:
46
+ raise ValueError(f"fay_k must be in [0, 1), got {fay_k}")
47
+ return 1 / (r * (1 - fay_k) ** 2)
48
+
49
+
50
+ def replicate_variance(
51
+ series,
52
+ statistic: Callable,
53
+ replicate_weights: np.ndarray | pd.DataFrame,
54
+ method: str = "jackknife",
55
+ fay_k: float | None = None,
56
+ *,
57
+ centering: str = "full-sample",
58
+ ) -> float:
59
+ """Variance of ``statistic`` estimated from replicate weights.
60
+
61
+ Variance is the method's scale factor times the sum of squared deviations
62
+ of replicate estimates from the selected center. The supported factors
63
+ are ``(R - 1) / R`` for unstratified JK1 or common-factor delete-group
64
+ jackknife, ``1 / R`` for BRR and bootstrap, ``4 / R`` for successive
65
+ difference, and ``1 / (R * (1 - fay_k)**2)`` for Fay's BRR. Select the
66
+ factor and center specified by the survey; arbitrary stratified jackknife
67
+ and averaged-bootstrap schemes requiring other factors are unsupported.
68
+ Validity also depends on the statistic: nonsmooth quantiles can require
69
+ an appropriate replication method or smoothing of replicate estimates.
70
+
71
+ :param series: A MicroSeries. Its own weights give the point estimate.
72
+ :param statistic: Callable taking a MicroSeries and returning a float,
73
+ for example ``lambda s: s.gini()``.
74
+ :param replicate_weights: Array or frame of shape ``(len(series), R)``
75
+ in the same row order as ``series``. Rows are matched by position;
76
+ DataFrame index labels are ignored.
77
+ :param method: One of ``jackknife``, ``brr``, ``bootstrap``,
78
+ ``successive-difference``, or ``fay`` (which requires ``fay_k``).
79
+ :param fay_k: Fay's perturbation constant, required when
80
+ ``method="fay"``.
81
+ :param centering: ``full-sample`` (default) centers on ``statistic(series)``;
82
+ ``replicate-mean`` centers on the mean of the replicate estimates.
83
+ This choice does not change the method's scale factor.
84
+ :returns: The estimated variance of the statistic.
85
+ """
86
+ from microdf.microseries import MicroSeries
87
+
88
+ if centering not in ("full-sample", "replicate-mean"):
89
+ raise ValueError(
90
+ f"centering must be 'full-sample' or 'replicate-mean', got {centering!r}"
91
+ )
92
+
93
+ weights = np.asarray(replicate_weights, dtype=float)
94
+ if weights.ndim != 2:
95
+ raise ValueError(
96
+ f"replicate_weights must be 2-dimensional, got shape {weights.shape}"
97
+ )
98
+ if weights.shape[0] != len(series):
99
+ raise ValueError(
100
+ f"replicate_weights has {weights.shape[0]} rows but the series has "
101
+ f"{len(series)}"
102
+ )
103
+
104
+ n_replicates = weights.shape[1]
105
+ if n_replicates < 2:
106
+ raise ValueError("At least two replicate weights are required")
107
+
108
+ if method == "fay":
109
+ if fay_k is None:
110
+ raise ValueError("method='fay' requires fay_k")
111
+ factor = _fay_factor(n_replicates, fay_k)
112
+ elif method in METHOD_FACTORS:
113
+ if fay_k is not None:
114
+ raise ValueError("fay_k applies only to method='fay'")
115
+ factor = METHOD_FACTORS[method](n_replicates)
116
+ else:
117
+ known = ", ".join(sorted([*METHOD_FACTORS, "fay"]))
118
+ raise ValueError(f"Unknown method {method!r}; expected one of {known}")
119
+
120
+ # Keep in-place callback transformations out of the caller and replicates.
121
+ center = (
122
+ float(statistic(series.copy(deep=True))) if centering == "full-sample" else None
123
+ )
124
+ values = series.array
125
+ index = series.index
126
+
127
+ estimates = []
128
+ for column in range(n_replicates):
129
+ replicate = MicroSeries(
130
+ values.copy(),
131
+ weights=weights[:, column].copy(),
132
+ index=index.copy(),
133
+ name=series.name,
134
+ dtype=series.dtype,
135
+ )
136
+ estimates.append(float(statistic(replicate)))
137
+
138
+ estimates = np.asarray(estimates)
139
+ if not np.all(np.isfinite(estimates)) or (
140
+ center is not None and not np.isfinite(center)
141
+ ):
142
+ # Retain the original infinity/NaN propagation for nonfinite callbacks.
143
+ if centering == "replicate-mean":
144
+ center = float(np.mean(estimates))
145
+ return factor * float(np.sum(np.square(estimates - center)))
146
+
147
+ with np.errstate(over="ignore"):
148
+ if centering == "replicate-mean":
149
+ # Keep the common offset out of the mean so small differences are
150
+ # preserved even when the absolute mean is not representable.
151
+ deviations = estimates - estimates[0]
152
+ if not np.all(np.isfinite(deviations)):
153
+ return float("inf")
154
+ deviations -= np.mean(deviations)
155
+ else:
156
+ deviations = estimates - center
157
+
158
+ scale = np.max(np.abs(deviations))
159
+ if scale == 0:
160
+ return 0.0
161
+ if not np.isfinite(scale):
162
+ return float("inf")
163
+
164
+ scaled_squares = np.sum(np.square(deviations / scale))
165
+ # Restore the scale by its binary exponent after applying the factor.
166
+ # Squaring scale first could overflow, or underflow before Fay's factor
167
+ # brings a tiny squared deviation back into the representable range.
168
+ significand, exponent = np.frexp(scale)
169
+ with np.errstate(over="ignore"):
170
+ return float(np.ldexp(significand**2 * factor * scaled_squares, 2 * exponent))
171
+
172
+
173
+ def replicate_standard_error(
174
+ series,
175
+ statistic: Callable,
176
+ replicate_weights: np.ndarray | pd.DataFrame,
177
+ method: str = "jackknife",
178
+ fay_k: float | None = None,
179
+ *,
180
+ centering: str = "full-sample",
181
+ ) -> float:
182
+ """Standard error of ``statistic``, the square root of its variance.
183
+
184
+ Takes the same arguments as :func:`replicate_variance`.
185
+ """
186
+ return float(
187
+ np.sqrt(
188
+ replicate_variance(
189
+ series, statistic, replicate_weights, method, fay_k, centering=centering
190
+ )
191
+ )
192
+ )
@@ -0,0 +1,195 @@
1
+ """Binary operations retain the calling Series' observation weights."""
2
+
3
+ import inspect
4
+
5
+ import numpy as np
6
+ import pandas as pd
7
+ import pytest
8
+
9
+ from microdf import MicroSeries
10
+
11
+
12
+ ARITHMETIC = ["add", "sub", "mul", "truediv", "floordiv", "mod", "pow"]
13
+ LOGICAL = ["and", "or", "xor"]
14
+ COMPARISONS = ["lt", "le", "eq", "ne", "ge", "gt"]
15
+
16
+
17
+ def assert_weighted_result(result, expected, source):
18
+ assert isinstance(result, MicroSeries)
19
+ pd.testing.assert_series_equal(pd.Series(result), expected)
20
+ expected_weights = (
21
+ source.weights
22
+ if source.index.equals(expected.index)
23
+ else source.weights.reindex(expected.index)
24
+ )
25
+ pd.testing.assert_series_equal(result.weights, expected_weights)
26
+ assert result.weights is not source.weights
27
+ assert result.sum() == expected.multiply(expected_weights).sum()
28
+
29
+
30
+ @pytest.mark.parametrize(
31
+ "method",
32
+ [f"__{prefix}{op}__" for op in ARITHMETIC + LOGICAL for prefix in ["", "r"]],
33
+ )
34
+ @pytest.mark.parametrize("weighted_other", [False, True])
35
+ def test_binary_operators_align_weights_with_labels(method, weighted_other):
36
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x")
37
+ other = pd.Series([2, 1], index=["a", "b"], name="x")
38
+ if weighted_other:
39
+ other = MicroSeries(other, weights=[9, 1])
40
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
41
+
42
+ result = getattr(source, method)(other)
43
+
44
+ assert_weighted_result(result, expected, source)
45
+ # Addition is 22 * 9 + 11 * 1 = 209, rather than the positional 121.
46
+ if method in ["__add__", "__radd__"]:
47
+ assert result.sum() == 209
48
+ result.weights.iloc[0] = 100
49
+ np.testing.assert_array_equal(source.weights, [1, 9])
50
+ if weighted_other:
51
+ np.testing.assert_array_equal(other.weights, [9, 1])
52
+ source.weights.iloc[1] = 200
53
+ assert result.weights.iloc[0] == 100
54
+
55
+
56
+ @pytest.mark.parametrize(
57
+ "method",
58
+ ARITHMETIC + [f"r{op}" for op in ARITHMETIC] + COMPARISONS + ["div", "rdiv"],
59
+ )
60
+ @pytest.mark.parametrize("permuted", [False, True])
61
+ def test_named_binary_methods_use_calling_series_weights(method, permuted):
62
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x")
63
+ other = MicroSeries(
64
+ [2, 1], index=["a", "b"] if permuted else ["b", "a"], weights=[5, 7], name="x"
65
+ )
66
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
67
+
68
+ result = getattr(source, method)(other)
69
+
70
+ assert_weighted_result(result, expected, source)
71
+ # Inherited public methods keep the installed pandas API signatures.
72
+ assert inspect.signature(getattr(MicroSeries, method)) == inspect.signature(
73
+ getattr(pd.Series, method)
74
+ )
75
+
76
+
77
+ @pytest.mark.parametrize("method", [f"__{op}__" for op in COMPARISONS])
78
+ def test_comparison_operators_keep_calling_series_weights_and_pandas_errors(method):
79
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
80
+ other = MicroSeries([20, 10], index=source.index, weights=[5, 7])
81
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
82
+ assert_weighted_result(getattr(source, method)(other), expected, source)
83
+
84
+ other.index = ["a", "b"]
85
+ with pytest.raises(ValueError) as pandas_error:
86
+ getattr(pd.Series(source), method)(pd.Series(other))
87
+ with pytest.raises(ValueError) as microdf_error:
88
+ getattr(source, method)(other)
89
+ assert str(microdf_error.value) == str(pandas_error.value)
90
+
91
+
92
+ @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
93
+ @pytest.mark.parametrize("operand", [3, [2, 1], np.array([2, 1])])
94
+ def test_scalar_and_array_binary_operands_keep_weights(method, operand):
95
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
96
+ expected = getattr(pd.Series(source), method)(operand)
97
+ assert_weighted_result(getattr(source, method)(operand), expected, source)
98
+
99
+
100
+ @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
101
+ def test_matching_duplicate_indexes_keep_positional_weights(method):
102
+ source = MicroSeries([10, 20, 30], index=["a", "a", "b"], weights=[1, 9, 3])
103
+ other = MicroSeries([2, 1, 4], index=source.index, weights=[5, 7, 11])
104
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
105
+ assert_weighted_result(getattr(source, method)(other), expected, source)
106
+
107
+
108
+ @pytest.mark.parametrize("method", ["__divmod__", "__rdivmod__", "divmod", "rdivmod"])
109
+ def test_divmod_results_keep_calling_series_weights(method):
110
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
111
+ other = MicroSeries([3, 4], index=["a", "b"], weights=[5, 7])
112
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
113
+ result = getattr(source, method)(other)
114
+ assert isinstance(result, tuple)
115
+ for actual, plain in zip(result, expected):
116
+ assert_weighted_result(actual, plain, source)
117
+
118
+
119
+ @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
120
+ @pytest.mark.parametrize(
121
+ "left_index,right_index",
122
+ [
123
+ (["b", "a"], ["a", "c"]),
124
+ (["b", "a", "b"], ["a", "b", "b"]),
125
+ ],
126
+ ids=["new-rows", "ambiguous-duplicates"],
127
+ )
128
+ def test_binary_operations_reject_unknown_row_weights(method, left_index, right_index):
129
+ source = MicroSeries(
130
+ range(len(left_index)), index=left_index, weights=range(1, len(left_index) + 1)
131
+ )
132
+ other = MicroSeries(
133
+ range(len(right_index)),
134
+ index=right_index,
135
+ weights=range(4, len(right_index) + 4),
136
+ )
137
+ with pytest.raises(ValueError, match="weights"):
138
+ getattr(source, method)(other)
139
+
140
+
141
+ @pytest.mark.parametrize("method", ["add", "rsub", "lt"])
142
+ def test_named_binary_arguments_preserve_pandas_values_and_errors(method):
143
+ index = pd.MultiIndex.from_tuples([("b", 2), ("a", 1)], names=["group", "row"])
144
+ source = MicroSeries([np.nan, 20], index=index, weights=[1, 9])
145
+ other = pd.Series([2, 1], index=pd.Index(["a", "b"], name="group"))
146
+ kwargs = {"level": "group", "fill_value": 0, "axis": "index"}
147
+ expected = getattr(pd.Series(source), method)(other, **kwargs)
148
+ assert_weighted_result(getattr(source, method)(other, **kwargs), expected, source)
149
+ for args, options in [
150
+ ((other,), {"axis": 1}),
151
+ (([1],), {}),
152
+ ((other,), {"unknown": True}),
153
+ ]:
154
+ with pytest.raises((TypeError, ValueError)) as pandas_error:
155
+ getattr(pd.Series(source), method)(*args, **options)
156
+ with pytest.raises(type(pandas_error.value)) as microdf_error:
157
+ getattr(source, method)(*args, **options)
158
+ # pandas identifies the concrete subclass in invalid-axis messages.
159
+ expected_error = str(pandas_error.value).replace(
160
+ "object type Series", "object type MicroSeries"
161
+ )
162
+ assert str(microdf_error.value) == expected_error
163
+
164
+
165
+ @pytest.mark.parametrize(
166
+ "operation",
167
+ [
168
+ lambda plain, weighted: plain + weighted,
169
+ lambda plain, weighted: plain - weighted,
170
+ lambda plain, weighted: plain * weighted,
171
+ lambda plain, weighted: plain / weighted,
172
+ lambda plain, weighted: plain // weighted,
173
+ lambda plain, weighted: plain % weighted,
174
+ lambda plain, weighted: plain**weighted,
175
+ lambda plain, weighted: plain & weighted,
176
+ lambda plain, weighted: plain | weighted,
177
+ lambda plain, weighted: plain ^ weighted,
178
+ ],
179
+ ids=ARITHMETIC + LOGICAL,
180
+ )
181
+ @pytest.mark.parametrize("indexes", ["matching", "permuted", "duplicates"])
182
+ def test_plain_series_left_expressions_preserve_weighted_dispatch(operation, indexes):
183
+ index = ["a", "a"] if indexes == "duplicates" else ["b", "a"]
184
+ source = MicroSeries([10, 20], index=index, weights=[1, 9], name="x")
185
+ other_index = ["a", "b"] if indexes == "permuted" else index
186
+ other = pd.Series([2, 1], index=other_index, name="x")
187
+ expected = operation(other, pd.Series(source))
188
+
189
+ result = operation(other, source)
190
+
191
+ assert_weighted_result(result, expected, source)
192
+ result.weights.iloc[0] = 100
193
+ np.testing.assert_array_equal(source.weights, [1, 9])
194
+ source.weights.iloc[1] = 200
195
+ assert result.weights.iloc[0] == 100
@@ -59,3 +59,50 @@ def test_stored_weight_edits_do_not_change_weight_column(dtype):
59
59
 
60
60
  np.testing.assert_array_equal(df["w"], [1, 2])
61
61
  assert df.sum()["x"] == 10 * 100 + 20 * 2
62
+
63
+
64
+ @pytest.mark.parametrize("drop", [False, True])
65
+ @pytest.mark.parametrize("inplace", [False, True])
66
+ @pytest.mark.parametrize(
67
+ "index,level",
68
+ [
69
+ (pd.Index(["b", "a", "a"], name="row"), None),
70
+ (
71
+ pd.MultiIndex.from_tuples(
72
+ [("b", 2), ("a", 1), ("a", 1)], names=["group", "row"]
73
+ ),
74
+ None,
75
+ ),
76
+ (
77
+ pd.MultiIndex.from_tuples(
78
+ [("b", 2), ("a", 1), ("a", 1)], names=["group", "row"]
79
+ ),
80
+ "group",
81
+ ),
82
+ ],
83
+ )
84
+ def test_reset_index_owns_independently_mutable_weights(index, level, drop, inplace):
85
+ source = mdf.MicroDataFrame({"x": [10, 20, 30]}, index=index, weights=[1, 9, 3])
86
+ original_weights = source.weights
87
+ expected = pd.DataFrame(source).reset_index(level=level, drop=drop)
88
+
89
+ result = source.reset_index(level=level, drop=drop, inplace=inplace)
90
+
91
+ if inplace:
92
+ assert result is None
93
+ result = source
94
+ assert isinstance(result, mdf.MicroDataFrame)
95
+ pd.testing.assert_frame_equal(pd.DataFrame(result), expected)
96
+ pd.testing.assert_series_equal(
97
+ result.weights, pd.Series([1.0, 9.0, 3.0], index=expected.index)
98
+ )
99
+ assert result.x.sum() == 10 * 1 + 20 * 9 + 30 * 3
100
+ assert result.weights is not original_weights
101
+ result.weights.iloc[0] = 100
102
+ np.testing.assert_array_equal(original_weights, [1, 9, 3])
103
+ assert result.x.sum() == 10 * 100 + 20 * 9 + 30 * 3
104
+ if not inplace:
105
+ assert source.x.sum() == 10 * 1 + 20 * 9 + 30 * 3
106
+ original_weights.iloc[1] = 200
107
+ np.testing.assert_array_equal(result.weights, [100, 9, 3])
108
+ assert result.x.sum() == 10 * 100 + 20 * 9 + 30 * 3
@@ -0,0 +1,544 @@
1
+ """Variance estimation from replicate weights.
2
+
3
+ Recomputing a statistic once per replicate weight vector and measuring the
4
+ spread gives a variance estimate for statistics whose analytic variance is
5
+ awkward, such as the Gini coefficient or a quantile.
6
+ """
7
+
8
+ import numpy as np
9
+ import pandas as pd
10
+ import pytest
11
+
12
+ from microdf import MicroSeries, replicate_standard_error, replicate_variance
13
+
14
+
15
+ @pytest.fixture
16
+ def series_and_replicates():
17
+ rng = np.random.default_rng(0)
18
+ values = rng.lognormal(mean=10, sigma=1.0, size=500)
19
+ weights = np.full(500, 40.0)
20
+ replicates = weights[:, None] * rng.poisson(1.0, size=(500, 200))
21
+ return MicroSeries(values, weights=weights), replicates
22
+
23
+
24
+ def test_matches_the_analytic_standard_error_of_a_weighted_mean(
25
+ series_and_replicates,
26
+ ):
27
+ """The mean has a closed form, so it is the case we can check exactly."""
28
+ series, replicates = series_and_replicates
29
+
30
+ replicate_se = replicate_standard_error(
31
+ series, lambda s: s.mean(), replicates, method="bootstrap"
32
+ )
33
+
34
+ values = np.asarray(series)
35
+ analytic_se = np.sqrt(np.var(values, ddof=1) / len(values))
36
+
37
+ assert replicate_se == pytest.approx(analytic_se, rel=0.15)
38
+
39
+
40
+ def test_works_for_statistics_with_no_analytic_variance(
41
+ series_and_replicates,
42
+ ):
43
+ """The point of the method: Gini and quantiles come out like anything
44
+ else."""
45
+ series, replicates = series_and_replicates
46
+
47
+ for statistic in (lambda s: s.gini(), lambda s: s.median()):
48
+ se = replicate_standard_error(series, statistic, replicates, method="bootstrap")
49
+ assert se > 0
50
+ assert np.isfinite(se)
51
+
52
+
53
+ @pytest.mark.parametrize(
54
+ "method,expected_factor",
55
+ [
56
+ ("jackknife", 199 / 200),
57
+ ("brr", 1 / 200),
58
+ ("bootstrap", 1 / 200),
59
+ ("successive-difference", 4 / 200),
60
+ ],
61
+ )
62
+ def test_each_method_applies_its_own_scale(
63
+ series_and_replicates, method, expected_factor
64
+ ):
65
+ """The scale factor is what distinguishes the replication schemes."""
66
+ series, replicates = series_and_replicates
67
+
68
+ variance = replicate_variance(series, lambda s: s.mean(), replicates, method=method)
69
+ reference = replicate_variance(series, lambda s: s.mean(), replicates, method="brr")
70
+
71
+ assert variance == pytest.approx(reference * expected_factor * 200, rel=1e-9)
72
+
73
+
74
+ def test_fay_requires_and_uses_its_constant(series_and_replicates):
75
+ series, replicates = series_and_replicates
76
+
77
+ with pytest.raises(ValueError, match="requires fay_k"):
78
+ replicate_variance(series, lambda s: s.mean(), replicates, method="fay")
79
+
80
+ fay = replicate_variance(
81
+ series, lambda s: s.mean(), replicates, method="fay", fay_k=0.5
82
+ )
83
+ brr = replicate_variance(series, lambda s: s.mean(), replicates, method="brr")
84
+
85
+ # 1 / (R (1 - k)^2) against 1 / R, so a factor of four at k = 0.5.
86
+ assert fay == pytest.approx(brr * 4, rel=1e-9)
87
+
88
+
89
+ def test_rejects_input_that_cannot_be_right(series_and_replicates):
90
+ series, replicates = series_and_replicates
91
+
92
+ with pytest.raises(ValueError, match="2-dimensional"):
93
+ replicate_variance(series, lambda s: s.mean(), np.ones(500))
94
+
95
+ with pytest.raises(ValueError, match="rows but the series has"):
96
+ replicate_variance(series, lambda s: s.mean(), np.ones((499, 10)))
97
+
98
+ with pytest.raises(ValueError, match="At least two"):
99
+ replicate_variance(series, lambda s: s.mean(), np.ones((500, 1)))
100
+
101
+ with pytest.raises(ValueError, match="Unknown method"):
102
+ replicate_variance(series, lambda s: s.mean(), replicates, method="nonsense")
103
+
104
+
105
+ @pytest.mark.parametrize("dtype", ["object", "category", "string"])
106
+ def test_replicates_preserve_categorical_values_and_metadata(dtype):
107
+ series = MicroSeries(
108
+ ["employed", "unemployed", "employed", None],
109
+ weights=[1, 1, 1, 1],
110
+ index=["a", "b", "c", "d"],
111
+ name="employment",
112
+ dtype=dtype,
113
+ )
114
+ replicates = np.array([[2, 2, 0, 0], [0, 0, 2, 2], [2, 0, 2, 0], [0, 2, 0, 2]])
115
+ original = pd.Series(series).copy(deep=True)
116
+ original_weights = series.weights.copy(deep=True)
117
+ original_replicates = replicates.copy()
118
+
119
+ def count(sample):
120
+ assert sample.dtype == series.dtype
121
+ assert sample.name == "employment"
122
+ pd.testing.assert_index_equal(sample.index, series.index)
123
+ return sample.count()
124
+
125
+ # Full count is 3; replicate counts are 4, 2, 4, 2.
126
+ assert replicate_variance(series, count, replicates, method="brr") == 1
127
+ pd.testing.assert_series_equal(pd.Series(series), original)
128
+ pd.testing.assert_series_equal(series.weights, original_weights)
129
+ np.testing.assert_array_equal(replicates, original_replicates)
130
+
131
+
132
+ @pytest.mark.parametrize("dtype", ["bool", "boolean"])
133
+ def test_replicates_preserve_boolean_domain_counts(dtype):
134
+ series = MicroSeries([True, False, True, False], weights=[1, 1, 1, 1], dtype=dtype)
135
+ replicates = np.array([[2, 2, 0, 0], [0, 0, 2, 2], [2, 0, 2, 0], [0, 2, 0, 2]])
136
+ # Complement counts are 0, 2, 2, 4, around the full-sample count of 2.
137
+ assert (
138
+ replicate_variance(
139
+ series, lambda sample: (~sample).sum(), replicates, method="brr"
140
+ )
141
+ == 2
142
+ )
143
+
144
+
145
+ @pytest.mark.parametrize("dtype", ["int64", "Int64"])
146
+ def test_replicates_preserve_exact_large_integer_categories(dtype):
147
+ category = 2**53
148
+ series = MicroSeries([category, category + 1], weights=[1, 2], dtype=dtype)
149
+ replicates = np.array([[1, 1], [2, 2]])
150
+ # Identical weights must keep the weighted category count exactly 1.
151
+ assert (
152
+ replicate_variance(
153
+ series, lambda sample: (sample == category).sum(), replicates, method="brr"
154
+ )
155
+ == 0
156
+ )
157
+
158
+
159
+ @pytest.mark.parametrize(
160
+ "method,fay_k,full_sample_variance,replicate_mean_variance",
161
+ [
162
+ ("jackknife", None, 1088 / 3, 3136 / 9),
163
+ ("brr", None, 544 / 3, 1568 / 9),
164
+ ("bootstrap", None, 544 / 3, 1568 / 9),
165
+ ("successive-difference", None, 2176 / 3, 6272 / 9),
166
+ ("fay", 0.5, 2176 / 3, 6272 / 9),
167
+ ],
168
+ )
169
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
170
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
171
+ def test_centering_uses_exact_nonlinear_replicate_estimates(
172
+ method, fay_k, full_sample_variance, replicate_mean_variance, centering, api
173
+ ):
174
+ series = MicroSeries([1, 3], weights=[1, 1])
175
+ replicates = np.array([[2, 1, 0], [0, 1, 2]])
176
+
177
+ # Squared totals: full sample 16; replicates 4, 16, 36; replicate mean 56/3.
178
+ # Squared deviations sum to 544 from 16 and 1568/3 from 56/3.
179
+ def squared_total(sample):
180
+ return sample.sum() ** 2
181
+
182
+ expected = (
183
+ full_sample_variance if centering == "full-sample" else replicate_mean_variance
184
+ )
185
+ kwargs = {"method": method, "fay_k": fay_k, "centering": centering}
186
+ if api == "variance":
187
+ observed = replicate_variance(series, squared_total, replicates, **kwargs)
188
+ elif api == "standard_error":
189
+ observed = replicate_standard_error(series, squared_total, replicates, **kwargs)
190
+ expected = np.sqrt(expected)
191
+ else:
192
+ observed = series.replicate_standard_error(squared_total, replicates, **kwargs)
193
+ expected = np.sqrt(expected)
194
+ assert observed == pytest.approx(expected)
195
+
196
+
197
+ def test_default_centering_retains_full_sample_estimate():
198
+ series = MicroSeries([1, 3], weights=[1, 1])
199
+ replicates = np.array([[2, 1, 0], [0, 1, 2]])
200
+ expected_variance = 544 / 3
201
+
202
+ def squared_total(sample):
203
+ return sample.sum() ** 2
204
+
205
+ assert replicate_variance(
206
+ series, squared_total, replicates, method="bootstrap"
207
+ ) == pytest.approx(expected_variance)
208
+ assert replicate_standard_error(
209
+ series, squared_total, replicates, method="bootstrap"
210
+ ) == pytest.approx(np.sqrt(expected_variance))
211
+ assert series.replicate_standard_error(
212
+ squared_total, replicates, method="bootstrap"
213
+ ) == pytest.approx(np.sqrt(expected_variance))
214
+
215
+
216
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
217
+ def test_rejects_unknown_centering_before_calling_statistic(api):
218
+ series = MicroSeries([1, 3], weights=[1, 1])
219
+ replicates = np.array([[2, 0], [0, 2]])
220
+
221
+ def unexpected_statistic(sample):
222
+ raise AssertionError("Invalid centering must be rejected first")
223
+
224
+ with pytest.raises(ValueError, match="centering"):
225
+ if api == "variance":
226
+ replicate_variance(
227
+ series, unexpected_statistic, replicates, centering="unknown"
228
+ )
229
+ elif api == "standard_error":
230
+ replicate_standard_error(
231
+ series, unexpected_statistic, replicates, centering="unknown"
232
+ )
233
+ else:
234
+ series.replicate_standard_error(
235
+ unexpected_statistic, replicates, centering="unknown"
236
+ )
237
+
238
+
239
+ def test_replicate_weight_frames_use_positional_rows():
240
+ series = MicroSeries([1, 3], weights=[1, 1], index=["a", "b"])
241
+ replicates = pd.DataFrame([[2, 0], [0, 2]], index=["b", "a"])
242
+ # Position-defined means are 1 and 3 around the full-sample mean of 2.
243
+ assert (
244
+ replicate_variance(
245
+ series, lambda sample: sample.mean(), replicates, method="brr"
246
+ )
247
+ == 1
248
+ )
249
+
250
+
251
+ def _replication_result(series, statistic, replicates, api, **kwargs):
252
+ if api == "variance":
253
+ return replicate_variance(series, statistic, replicates, **kwargs)
254
+ if api == "standard_error":
255
+ return replicate_standard_error(series, statistic, replicates, **kwargs)
256
+ return series.replicate_standard_error(statistic, replicates, **kwargs)
257
+
258
+
259
+ @pytest.mark.parametrize(
260
+ "method,fay_k,amplitude,expected_variance",
261
+ [
262
+ ("jackknife", None, 5e153, 1.75e308),
263
+ ("brr", None, 1e154, 1e308),
264
+ ("bootstrap", None, 1e154, 1e308),
265
+ ("successive-difference", None, 5e153, 1e308),
266
+ ("fay", 0.5, 5e153, 1e308),
267
+ ],
268
+ )
269
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
270
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
271
+ def test_large_finite_replicate_variance(
272
+ method, fay_k, amplitude, expected_variance, centering, api
273
+ ):
274
+ series = MicroSeries([0.0, 2 * amplitude], weights=[1, 1])
275
+ replicates = np.tile([[2, 0], [0, 2]], (1, 4))
276
+ # Eight deviations are +/- amplitude. The factors are 7/8, 1/8 or 1/2.
277
+ # Every method has a finite variance despite an overflowing raw sum.
278
+ expected = expected_variance if api == "variance" else np.sqrt(expected_variance)
279
+ observed = _replication_result(
280
+ series,
281
+ lambda sample: sample.mean(),
282
+ replicates,
283
+ api,
284
+ method=method,
285
+ fay_k=fay_k,
286
+ centering=centering,
287
+ )
288
+ assert observed == pytest.approx(expected, rel=1e-14, abs=0)
289
+
290
+
291
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
292
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
293
+ def test_identical_large_replicates_have_zero_variance(centering, api):
294
+ series = MicroSeries([1e308, 1e308], weights=[1, 1])
295
+ replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]])
296
+ # The median remains finite even though summing the four estimates overflows.
297
+ assert (
298
+ _replication_result(
299
+ series,
300
+ lambda sample: sample.median(),
301
+ replicates,
302
+ api,
303
+ method="brr",
304
+ centering=centering,
305
+ )
306
+ == 0
307
+ )
308
+
309
+
310
+ @pytest.mark.parametrize(
311
+ "method,fay_k,variance_multiplier",
312
+ [
313
+ ("jackknife", None, 3),
314
+ ("brr", None, 1),
315
+ ("bootstrap", None, 1),
316
+ ("successive-difference", None, 4),
317
+ ("fay", 0.5, 4),
318
+ ],
319
+ )
320
+ @pytest.mark.parametrize("offset", [0.0, float(2**53)])
321
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
322
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
323
+ def test_replicate_centering_preserves_small_differences(
324
+ method, fay_k, variance_multiplier, offset, centering, api
325
+ ):
326
+ series = MicroSeries([offset, offset + 2], weights=[1, 1])
327
+ replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]])
328
+ # The exact replicate mean has deviations +/-1, including at 2**53.
329
+ # Full-sample centering must retain the callback's rounded mean at 2**53,
330
+ # so its deviations are 0 and 2, giving twice the centered variance.
331
+ expected = variance_multiplier
332
+ if offset and centering == "full-sample":
333
+ expected *= 2
334
+ if api != "variance":
335
+ expected = np.sqrt(expected)
336
+ observed = _replication_result(
337
+ series,
338
+ lambda sample: sample.mean(),
339
+ replicates,
340
+ api,
341
+ method=method,
342
+ fay_k=fay_k,
343
+ centering=centering,
344
+ )
345
+ assert observed == pytest.approx(expected, rel=1e-14, abs=0)
346
+
347
+
348
+ @pytest.mark.parametrize(
349
+ "amplitude,method,fay_k,expected_variance",
350
+ [
351
+ (2.0**-537, "brr", None, 2.0**-1074),
352
+ (2.0**-550, "fay", 1 - 2.0**-53, 2.0**-994),
353
+ ],
354
+ )
355
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
356
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
357
+ def test_small_deviations_keep_representable_variance(
358
+ amplitude, method, fay_k, expected_variance, centering, api
359
+ ):
360
+ series = MicroSeries([-amplitude, amplitude], weights=[1, 1])
361
+ replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]])
362
+ # BRR variance is amplitude**2. Fay's factor multiplies that by 2**106;
363
+ # it must be applied before rounding the initially unrepresentable square.
364
+ expected = expected_variance if api == "variance" else np.sqrt(expected_variance)
365
+ observed = _replication_result(
366
+ series,
367
+ lambda sample: sample.mean(),
368
+ replicates,
369
+ api,
370
+ method=method,
371
+ fay_k=fay_k,
372
+ centering=centering,
373
+ )
374
+ assert observed == expected
375
+
376
+
377
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
378
+ def test_finite_replicates_with_unrepresentable_variance(centering):
379
+ series = MicroSeries([-1e308, 1e308], weights=[1, 1])
380
+ replicates = np.array([[1, 0], [0, 1]])
381
+ # A weighted total gives finite estimates +/-1e308 and a zero point estimate.
382
+ with np.errstate(over="ignore", invalid="ignore"):
383
+ observed = replicate_variance(
384
+ series,
385
+ lambda sample: sample.sum(),
386
+ replicates,
387
+ method="brr",
388
+ centering=centering,
389
+ )
390
+ assert observed == np.inf
391
+
392
+
393
+ @pytest.mark.parametrize(
394
+ "point,estimates,centering,expected",
395
+ [
396
+ (0.0, [0.0, np.inf], "full-sample", np.inf),
397
+ (np.inf, [0.0, 1.0], "full-sample", np.inf),
398
+ (np.inf, [0.0, np.inf], "full-sample", np.nan),
399
+ (np.nan, [0.0, 1.0], "full-sample", np.nan),
400
+ (0.0, [0.0, np.nan], "full-sample", np.nan),
401
+ (0.0, [0.0, np.inf], "replicate-mean", np.nan),
402
+ (0.0, [np.inf, -np.inf], "replicate-mean", np.nan),
403
+ (0.0, [0.0, np.nan], "replicate-mean", np.nan),
404
+ ],
405
+ )
406
+ def test_nonfinite_callback_results_keep_existing_behavior(
407
+ point, estimates, centering, expected
408
+ ):
409
+ series = MicroSeries([1, 2], weights=[1, 1])
410
+ replicates = np.array([[2, 0], [0, 2]])
411
+ results = iter([point, *estimates] if centering == "full-sample" else estimates)
412
+ with np.errstate(over="ignore", invalid="ignore"):
413
+ observed = replicate_variance(
414
+ series,
415
+ lambda sample: next(results),
416
+ replicates,
417
+ method="brr",
418
+ centering=centering,
419
+ )
420
+ if np.isnan(expected):
421
+ assert np.isnan(observed)
422
+ else:
423
+ assert observed == expected
424
+
425
+
426
+ @pytest.mark.parametrize("inplace", [False, True])
427
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
428
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
429
+ def test_callback_transforms_each_sample_once(inplace, centering, api):
430
+ series = MicroSeries([120.0, 180.0], weights=[1, 1])
431
+ replicates = np.array([[2.0, 0.0], [0.0, 2.0]])
432
+ original_values = pd.Series(series).copy(deep=True)
433
+ original_weights = series.weights.copy(deep=True)
434
+ original_replicates = replicates.copy()
435
+ calls = []
436
+
437
+ def total_after_allowance(sample):
438
+ calls.append(sample.weights.tolist())
439
+ if inplace:
440
+ sample -= 100
441
+ return sample.sum()
442
+ return (sample - 100).sum()
443
+
444
+ # Transformed values are 20 and 80; totals are 100, 40 and 160.
445
+ # BRR variance is ((40 - 100)**2 + (160 - 100)**2) / 2 = 3600.
446
+ expected = 3600 if api == "variance" else 60
447
+ assert (
448
+ _replication_result(
449
+ series,
450
+ total_after_allowance,
451
+ replicates,
452
+ api,
453
+ method="brr",
454
+ centering=centering,
455
+ )
456
+ == expected
457
+ )
458
+ expected_calls = [[2.0, 0.0], [0.0, 2.0]]
459
+ if centering == "full-sample":
460
+ expected_calls.insert(0, [1.0, 1.0])
461
+ assert calls == expected_calls
462
+ pd.testing.assert_series_equal(pd.Series(series), original_values)
463
+ pd.testing.assert_series_equal(series.weights, original_weights)
464
+ np.testing.assert_array_equal(replicates, original_replicates)
465
+
466
+
467
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
468
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
469
+ def test_callbacks_isolate_values_weights_and_metadata(centering, api):
470
+ series = MicroSeries(
471
+ [120, 180],
472
+ weights=[1, 1],
473
+ index=pd.Index(["a", "b"], name="person"),
474
+ name="income",
475
+ dtype="Int64",
476
+ )
477
+ replicates = np.array([[2.0, 0.0], [0.0, 2.0]])
478
+ original = series.copy(deep=True)
479
+ original_replicates = replicates.copy()
480
+ calls = []
481
+
482
+ def mutating_total(sample):
483
+ assert sample.tolist() == [120, 180]
484
+ assert sample.dtype == original.dtype
485
+ assert sample.name == "income"
486
+ pd.testing.assert_index_equal(sample.index, original.index)
487
+ calls.append(sample.weights.tolist())
488
+ sample -= 100
489
+ estimate = sample.sum()
490
+ sample.weights.iloc[:] = 7
491
+ sample.index = pd.Index(["x", "y"], name="changed")
492
+ sample.name = "changed"
493
+ return estimate
494
+
495
+ expected = 3600 if api == "variance" else 60
496
+ assert (
497
+ _replication_result(
498
+ series, mutating_total, replicates, api, method="brr", centering=centering
499
+ )
500
+ == expected
501
+ )
502
+ expected_calls = [[2.0, 0.0], [0.0, 2.0]]
503
+ if centering == "full-sample":
504
+ expected_calls.insert(0, [1.0, 1.0])
505
+ assert calls == expected_calls
506
+ pd.testing.assert_series_equal(pd.Series(series), pd.Series(original))
507
+ pd.testing.assert_series_equal(series.weights, original.weights)
508
+ np.testing.assert_array_equal(replicates, original_replicates)
509
+
510
+
511
+ @pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"])
512
+ @pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"])
513
+ def test_callback_exception_preserves_inputs(centering, api):
514
+ series = MicroSeries(
515
+ [120.0, 180.0], weights=[1, 1], index=["a", "b"], name="income"
516
+ )
517
+ replicates = np.array([[2.0, 0.0], [0.0, 2.0]])
518
+ original = series.copy(deep=True)
519
+ original_replicates = replicates.copy()
520
+ callback_error = ValueError("statistic failed after mutation")
521
+ calls = []
522
+
523
+ def failing_statistic(sample):
524
+ calls.append(sample.weights.tolist())
525
+ sample -= 100
526
+ sample.weights.iloc[:] = 7
527
+ sample.index = ["x", "y"]
528
+ sample.name = "changed"
529
+ raise callback_error
530
+
531
+ with pytest.raises(ValueError, match="statistic failed") as raised:
532
+ _replication_result(
533
+ series,
534
+ failing_statistic,
535
+ replicates,
536
+ api,
537
+ method="brr",
538
+ centering=centering,
539
+ )
540
+ assert raised.value is callback_error
541
+ assert calls == ([[1.0, 1.0]] if centering == "full-sample" else [[2.0, 0.0]])
542
+ pd.testing.assert_series_equal(pd.Series(series), pd.Series(original))
543
+ pd.testing.assert_series_equal(series.weights, original.weights)
544
+ np.testing.assert_array_equal(replicates, original_replicates)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.4.1
3
+ Version: 1.5.1
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -5,13 +5,16 @@ microdf/__init__.py
5
5
  microdf/_weights.py
6
6
  microdf/microdataframe.py
7
7
  microdf/microseries.py
8
+ microdf/replication.py
8
9
  microdf/tests/conftest.py
9
10
  microdf/tests/test_aggregation_errors.py
11
+ microdf/tests/test_binary_weight_alignment.py
10
12
  microdf/tests/test_dataframe_weight_storage.py
11
13
  microdf/tests/test_microseries_dataframe.py
12
14
  microdf/tests/test_nullify_weights_index.py
13
15
  microdf/tests/test_pandas3_compatibility.py
14
16
  microdf/tests/test_quantile_missing_values.py
17
+ microdf/tests/test_replication.py
15
18
  microdf/tests/test_serialization.py
16
19
  microdf/tests/test_sum_axes.py
17
20
  microdf/tests/test_version_metadata.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "microdf-python"
7
- version = "1.4.1"
7
+ version = "1.5.1"
8
8
  description = "Weighted pandas DataFrames and Series for survey microdata"
9
9
  readme = "README.md"
10
10
  authors = [
File without changes
File without changes
File without changes