microdf-python 1.5.0__tar.gz → 1.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {microdf_python-1.5.0/microdf_python.egg-info → microdf_python-1.5.2}/PKG-INFO +1 -1
  2. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/microdataframe.py +5 -11
  3. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/microseries.py +104 -66
  4. microdf_python-1.5.2/microdf/tests/test_binary_weight_alignment.py +195 -0
  5. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_dataframe_weight_storage.py +47 -0
  6. microdf_python-1.5.2/microdf/tests/test_ufunc_weight_dispatch.py +640 -0
  7. {microdf_python-1.5.0 → microdf_python-1.5.2/microdf_python.egg-info}/PKG-INFO +1 -1
  8. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf_python.egg-info/SOURCES.txt +2 -0
  9. {microdf_python-1.5.0 → microdf_python-1.5.2}/pyproject.toml +1 -1
  10. {microdf_python-1.5.0 → microdf_python-1.5.2}/LICENSE +0 -0
  11. {microdf_python-1.5.0 → microdf_python-1.5.2}/README.md +0 -0
  12. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/__init__.py +0 -0
  13. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/_weights.py +0 -0
  14. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/replication.py +0 -0
  15. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/conftest.py +0 -0
  16. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_aggregation_errors.py +0 -0
  17. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_microseries_dataframe.py +0 -0
  18. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_nullify_weights_index.py +0 -0
  19. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_pandas3_compatibility.py +0 -0
  20. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_quantile_missing_values.py +0 -0
  21. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_replication.py +0 -0
  22. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_serialization.py +0 -0
  23. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_sum_axes.py +0 -0
  24. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_version_metadata.py +0 -0
  25. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_weight_propagation.py +0 -0
  26. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf/tests/test_weighted_cov_corr.py +0 -0
  27. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf_python.egg-info/dependency_links.txt +0 -0
  28. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf_python.egg-info/requires.txt +0 -0
  29. {microdf_python-1.5.0 → microdf_python-1.5.2}/microdf_python.egg-info/top_level.txt +0 -0
  30. {microdf_python-1.5.0 → microdf_python-1.5.2}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.5.0
3
+ Version: 1.5.2
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -460,7 +460,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
460
460
  if inplace:
461
461
  # Snapshot weight *values* positionally — the index is about
462
462
  # to change and reset_index preserves row order.
463
- weight_values = np.asarray(self.weights.values, dtype=float)
463
+ weight_values = np.array(self.weights, dtype=float, copy=True)
464
464
  super().reset_index(
465
465
  level=level,
466
466
  drop=drop,
@@ -470,7 +470,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
470
470
  allow_duplicates=allow_duplicates,
471
471
  names=names,
472
472
  )
473
- self.weights = pd.Series(weight_values, index=self.index, dtype=float)
473
+ self.weights = weight_series(weight_values, self.index)
474
474
  self._link_all_weights()
475
475
  return None
476
476
  else:
@@ -483,15 +483,9 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
483
483
  allow_duplicates=allow_duplicates,
484
484
  names=names,
485
485
  )
486
- out = MicroDataFrame(res, weights=self.weights.values)
487
- # Ensure weights align to res.index (reset_index changes the
488
- # index but preserves row order, so pass values positionally).
489
- out.weights = pd.Series(
490
- np.asarray(self.weights.values, dtype=float),
491
- index=out.index,
492
- dtype=float,
493
- )
494
- return out
486
+ # Own a positional copy: reset_index changes labels but
487
+ # preserves row order.
488
+ return MicroDataFrame(res, weights=weight_series(self.weights, res.index))
495
489
 
496
490
  def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame":
497
491
  return super().copy(deep)
@@ -6,7 +6,12 @@ from typing import Callable, List, Optional, Union
6
6
  import numpy as np
7
7
  import pandas as pd
8
8
 
9
- from microdf._weights import WeightPropagationMixin, finalize_weights, weight_series
9
+ from microdf._weights import (
10
+ WeightPropagationMixin,
11
+ aligned_weights,
12
+ finalize_weights,
13
+ weight_series,
14
+ )
10
15
 
11
16
  logger = logging.getLogger(__name__)
12
17
 
@@ -94,6 +99,10 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
94
99
  # Keep pandas' own metadata, including the Series name.
95
100
  _metadata = pd.Series._metadata + ["weights"]
96
101
 
102
+ # These operands previously shared pandas' ufunc handler with MicroSeries.
103
+ # Keep inherited fallback paths working after overriding that handler.
104
+ _HANDLED_TYPES = pd.Series._HANDLED_TYPES + (pd.Series, pd.DataFrame)
105
+
97
106
  def __init__(self, *args, weights: np.array = None, **kwargs):
98
107
  """A Series-inheriting class for weighted microdata.
99
108
 
@@ -115,11 +124,90 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
115
124
 
116
125
  return MicroDataFrame
117
126
 
127
+ def __array_ufunc__(self, ufunc, method, *inputs, **kwargs):
128
+ # Preserve deferral to foreign handlers. A higher-priority Series
129
+ # inheriting pandas' handler cannot take over: pandas would defer back
130
+ # to our distinct handler, so handle that case through plain Series.
131
+ known_handlers = (
132
+ pd.Series.__array_ufunc__,
133
+ MicroSeries.__array_ufunc__,
134
+ type(self).__array_ufunc__,
135
+ )
136
+ inherited_series_priority = False
137
+ dispatch_index = 0
138
+ for position, value in enumerate(inputs):
139
+ if value is not self and isinstance(value, (pd.Series, pd.DataFrame)):
140
+ handler = type(value).__array_ufunc__
141
+ if handler not in known_handlers:
142
+ return NotImplemented
143
+ if value.__array_priority__ > self.__array_priority__:
144
+ if (
145
+ isinstance(value, pd.Series)
146
+ and handler is pd.Series.__array_ufunc__
147
+ ):
148
+ inherited_series_priority = True
149
+ dispatch_index = position
150
+ else:
151
+ return NotImplemented
152
+
153
+ out = kwargs.get("out")
154
+ has_output = out is not None and any(value is not None for value in out)
155
+ if (
156
+ method == "__call__"
157
+ and len(inputs) == 2
158
+ and all(isinstance(value, pd.Series) for value in inputs)
159
+ and (not has_output or inherited_series_priority)
160
+ ):
161
+ # pandas' generic ufunc reconstruction drops metadata for multiple
162
+ # Series. Preserve pandas' selected handler receiver because it
163
+ # determines alignment order, including positional output masks.
164
+ plain = tuple(
165
+ pd.Series(value, copy=False).__finalize__(value) for value in inputs
166
+ )
167
+ if has_output:
168
+ # Match pandas' receiver-based alignment before it writes out.
169
+ # Reconstruction must not discover invalid row weights later.
170
+ result_index = plain[dispatch_index].index.union(plain[1].index)
171
+ aligned_weights(self, result_index)
172
+ for output in out:
173
+ if isinstance(output, MicroSeries):
174
+ aligned_weights(output, result_index)
175
+ result = pd.Series.__array_ufunc__(
176
+ plain[dispatch_index], ufunc, method, *plain, **kwargs
177
+ )
178
+
179
+ def restore_weights(value):
180
+ if isinstance(value, pd.Series):
181
+ return self._weighted_result(
182
+ value, aligned_weights(self, value.index)
183
+ ).__finalize__(value)
184
+ return value
185
+
186
+ if isinstance(result, tuple):
187
+ return tuple(restore_weights(value) for value in result)
188
+ return restore_weights(result)
189
+ return super().__array_ufunc__(ufunc, method, *inputs, **kwargs)
190
+
191
+ def __rdivmod__(self, other) -> tuple["MicroSeries", "MicroSeries"]:
192
+ # An explicit override gives the weighted subclass priority over a
193
+ # plain Series on the left, as for the other reverse operators.
194
+ return super().__rdivmod__(other)
195
+
118
196
  def __finalize__(self, other, method=None, **kwargs):
119
197
  previous = self.__dict__.get("weights")
120
198
  super().__finalize__(other, method=method, **kwargs)
121
199
  return finalize_weights(self, other, method, previous)
122
200
 
201
+ def _construct_result(self, *args, **kwargs):
202
+ # pandas has already aligned this Series before constructing a binary
203
+ # result. Retain its row weights, even when pandas 3 also finalizes
204
+ # metadata from the other operand. Delegate values and names to pandas.
205
+ result = super()._construct_result(*args, **kwargs)
206
+ if not isinstance(result, tuple):
207
+ result.weights = weight_series(self.weights, result.index)
208
+ # divmod constructs both tuple members through this same hook.
209
+ return result
210
+
123
211
  def __setattr__(self, name, value):
124
212
  weights = self.__dict__.get("weights") if name == "index" else None
125
213
  super().__setattr__(name, value)
@@ -935,95 +1023,45 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
935
1023
  def __getattr__(self, name: str) -> "MicroSeries":
936
1024
  return MicroSeries(super().__getattr__(name), weights=self.weights)
937
1025
 
938
- # operators
939
-
940
- def __add__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
941
- return MicroSeries(super().__add__(other), weights=self.weights)
942
-
943
- def __sub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
944
- return MicroSeries(super().__sub__(other), weights=self.weights)
945
-
946
- def __mul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
947
- return MicroSeries(super().__mul__(other), weights=self.weights)
948
-
949
- def __floordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
950
- return MicroSeries(super().__floordiv__(other), weights=self.weights)
951
-
952
- def __truediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
953
- return MicroSeries(super().__truediv__(other), weights=self.weights)
954
-
955
- def __mod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
956
- return MicroSeries(super().__mod__(other), weights=self.weights)
957
-
958
- def __pow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
959
- return MicroSeries(super().__pow__(other), weights=self.weights)
960
-
961
- def __xor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
962
- return MicroSeries(super().__xor__(other), weights=self.weights)
963
-
964
- def __and__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
965
- return MicroSeries(super().__and__(other), weights=self.weights)
966
-
967
- def __or__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
968
- return MicroSeries(super().__or__(other), weights=self.weights)
969
-
970
- def __invert__(self) -> "MicroSeries":
971
- return MicroSeries(super().__invert__(), weights=self.weights)
972
-
1026
+ # Explicit reverse overrides give this subclass priority when a plain
1027
+ # pandas Series is on the left. _construct_result retains aligned weights.
973
1028
  def __radd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
974
- return MicroSeries(super().__radd__(other), weights=self.weights)
1029
+ return super().__radd__(other)
975
1030
 
976
1031
  def __rsub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
977
- return MicroSeries(super().__rsub__(other), weights=self.weights)
1032
+ return super().__rsub__(other)
978
1033
 
979
1034
  def __rmul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
980
- return MicroSeries(super().__rmul__(other), weights=self.weights)
1035
+ return super().__rmul__(other)
981
1036
 
982
1037
  def __rfloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
983
- return MicroSeries(super().__rfloordiv__(other), weights=self.weights)
1038
+ return super().__rfloordiv__(other)
984
1039
 
985
1040
  def __rtruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
986
- return MicroSeries(super().__rtruediv__(other), weights=self.weights)
1041
+ return super().__rtruediv__(other)
987
1042
 
988
1043
  def __rmod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
989
- return MicroSeries(super().__rmod__(other), weights=self.weights)
1044
+ return super().__rmod__(other)
990
1045
 
991
1046
  def __rpow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
992
- return MicroSeries(super().__rpow__(other), weights=self.weights)
1047
+ return super().__rpow__(other)
993
1048
 
994
1049
  def __rand__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
995
- return MicroSeries(super().__rand__(other), weights=self.weights)
1050
+ return super().__rand__(other)
996
1051
 
997
1052
  def __ror__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
998
- return MicroSeries(super().__ror__(other), weights=self.weights)
1053
+ return super().__ror__(other)
999
1054
 
1000
1055
  def __rxor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1001
- return MicroSeries(super().__rxor__(other), weights=self.weights)
1056
+ return super().__rxor__(other)
1057
+
1058
+ def __invert__(self) -> "MicroSeries":
1059
+ return MicroSeries(super().__invert__(), weights=self.weights)
1002
1060
 
1003
1061
  def sqrt(self) -> "MicroSeries":
1004
1062
  sqrt_values = np.sqrt(self._values)
1005
1063
  return MicroSeries(sqrt_values, index=self.index, weights=self.weights)
1006
1064
 
1007
- # comparators
1008
-
1009
- def __lt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1010
- return MicroSeries(super().__lt__(other), weights=self.weights)
1011
-
1012
- def __le__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1013
- return MicroSeries(super().__le__(other), weights=self.weights)
1014
-
1015
- def __eq__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1016
- return MicroSeries(super().__eq__(other), weights=self.weights)
1017
-
1018
- def __ne__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1019
- return MicroSeries(super().__ne__(other), weights=self.weights)
1020
-
1021
- def __ge__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1022
- return MicroSeries(super().__ge__(other), weights=self.weights)
1023
-
1024
- def __gt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
1025
- return MicroSeries(super().__gt__(other), weights=self.weights)
1026
-
1027
1065
  # assignment operators
1028
1066
 
1029
1067
  def __iadd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
@@ -0,0 +1,195 @@
1
+ """Binary operations retain the calling Series' observation weights."""
2
+
3
+ import inspect
4
+
5
+ import numpy as np
6
+ import pandas as pd
7
+ import pytest
8
+
9
+ from microdf import MicroSeries
10
+
11
+
12
+ ARITHMETIC = ["add", "sub", "mul", "truediv", "floordiv", "mod", "pow"]
13
+ LOGICAL = ["and", "or", "xor"]
14
+ COMPARISONS = ["lt", "le", "eq", "ne", "ge", "gt"]
15
+
16
+
17
+ def assert_weighted_result(result, expected, source):
18
+ assert isinstance(result, MicroSeries)
19
+ pd.testing.assert_series_equal(pd.Series(result), expected)
20
+ expected_weights = (
21
+ source.weights
22
+ if source.index.equals(expected.index)
23
+ else source.weights.reindex(expected.index)
24
+ )
25
+ pd.testing.assert_series_equal(result.weights, expected_weights)
26
+ assert result.weights is not source.weights
27
+ assert result.sum() == expected.multiply(expected_weights).sum()
28
+
29
+
30
+ @pytest.mark.parametrize(
31
+ "method",
32
+ [f"__{prefix}{op}__" for op in ARITHMETIC + LOGICAL for prefix in ["", "r"]],
33
+ )
34
+ @pytest.mark.parametrize("weighted_other", [False, True])
35
+ def test_binary_operators_align_weights_with_labels(method, weighted_other):
36
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x")
37
+ other = pd.Series([2, 1], index=["a", "b"], name="x")
38
+ if weighted_other:
39
+ other = MicroSeries(other, weights=[9, 1])
40
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
41
+
42
+ result = getattr(source, method)(other)
43
+
44
+ assert_weighted_result(result, expected, source)
45
+ # Addition is 22 * 9 + 11 * 1 = 209, rather than the positional 121.
46
+ if method in ["__add__", "__radd__"]:
47
+ assert result.sum() == 209
48
+ result.weights.iloc[0] = 100
49
+ np.testing.assert_array_equal(source.weights, [1, 9])
50
+ if weighted_other:
51
+ np.testing.assert_array_equal(other.weights, [9, 1])
52
+ source.weights.iloc[1] = 200
53
+ assert result.weights.iloc[0] == 100
54
+
55
+
56
+ @pytest.mark.parametrize(
57
+ "method",
58
+ ARITHMETIC + [f"r{op}" for op in ARITHMETIC] + COMPARISONS + ["div", "rdiv"],
59
+ )
60
+ @pytest.mark.parametrize("permuted", [False, True])
61
+ def test_named_binary_methods_use_calling_series_weights(method, permuted):
62
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x")
63
+ other = MicroSeries(
64
+ [2, 1], index=["a", "b"] if permuted else ["b", "a"], weights=[5, 7], name="x"
65
+ )
66
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
67
+
68
+ result = getattr(source, method)(other)
69
+
70
+ assert_weighted_result(result, expected, source)
71
+ # Inherited public methods keep the installed pandas API signatures.
72
+ assert inspect.signature(getattr(MicroSeries, method)) == inspect.signature(
73
+ getattr(pd.Series, method)
74
+ )
75
+
76
+
77
+ @pytest.mark.parametrize("method", [f"__{op}__" for op in COMPARISONS])
78
+ def test_comparison_operators_keep_calling_series_weights_and_pandas_errors(method):
79
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
80
+ other = MicroSeries([20, 10], index=source.index, weights=[5, 7])
81
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
82
+ assert_weighted_result(getattr(source, method)(other), expected, source)
83
+
84
+ other.index = ["a", "b"]
85
+ with pytest.raises(ValueError) as pandas_error:
86
+ getattr(pd.Series(source), method)(pd.Series(other))
87
+ with pytest.raises(ValueError) as microdf_error:
88
+ getattr(source, method)(other)
89
+ assert str(microdf_error.value) == str(pandas_error.value)
90
+
91
+
92
+ @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
93
+ @pytest.mark.parametrize("operand", [3, [2, 1], np.array([2, 1])])
94
+ def test_scalar_and_array_binary_operands_keep_weights(method, operand):
95
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
96
+ expected = getattr(pd.Series(source), method)(operand)
97
+ assert_weighted_result(getattr(source, method)(operand), expected, source)
98
+
99
+
100
+ @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
101
+ def test_matching_duplicate_indexes_keep_positional_weights(method):
102
+ source = MicroSeries([10, 20, 30], index=["a", "a", "b"], weights=[1, 9, 3])
103
+ other = MicroSeries([2, 1, 4], index=source.index, weights=[5, 7, 11])
104
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
105
+ assert_weighted_result(getattr(source, method)(other), expected, source)
106
+
107
+
108
+ @pytest.mark.parametrize("method", ["__divmod__", "__rdivmod__", "divmod", "rdivmod"])
109
+ def test_divmod_results_keep_calling_series_weights(method):
110
+ source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9])
111
+ other = MicroSeries([3, 4], index=["a", "b"], weights=[5, 7])
112
+ expected = getattr(pd.Series(source), method)(pd.Series(other))
113
+ result = getattr(source, method)(other)
114
+ assert isinstance(result, tuple)
115
+ for actual, plain in zip(result, expected):
116
+ assert_weighted_result(actual, plain, source)
117
+
118
+
119
+ @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"])
120
+ @pytest.mark.parametrize(
121
+ "left_index,right_index",
122
+ [
123
+ (["b", "a"], ["a", "c"]),
124
+ (["b", "a", "b"], ["a", "b", "b"]),
125
+ ],
126
+ ids=["new-rows", "ambiguous-duplicates"],
127
+ )
128
+ def test_binary_operations_reject_unknown_row_weights(method, left_index, right_index):
129
+ source = MicroSeries(
130
+ range(len(left_index)), index=left_index, weights=range(1, len(left_index) + 1)
131
+ )
132
+ other = MicroSeries(
133
+ range(len(right_index)),
134
+ index=right_index,
135
+ weights=range(4, len(right_index) + 4),
136
+ )
137
+ with pytest.raises(ValueError, match="weights"):
138
+ getattr(source, method)(other)
139
+
140
+
141
+ @pytest.mark.parametrize("method", ["add", "rsub", "lt"])
142
+ def test_named_binary_arguments_preserve_pandas_values_and_errors(method):
143
+ index = pd.MultiIndex.from_tuples([("b", 2), ("a", 1)], names=["group", "row"])
144
+ source = MicroSeries([np.nan, 20], index=index, weights=[1, 9])
145
+ other = pd.Series([2, 1], index=pd.Index(["a", "b"], name="group"))
146
+ kwargs = {"level": "group", "fill_value": 0, "axis": "index"}
147
+ expected = getattr(pd.Series(source), method)(other, **kwargs)
148
+ assert_weighted_result(getattr(source, method)(other, **kwargs), expected, source)
149
+ for args, options in [
150
+ ((other,), {"axis": 1}),
151
+ (([1],), {}),
152
+ ((other,), {"unknown": True}),
153
+ ]:
154
+ with pytest.raises((TypeError, ValueError)) as pandas_error:
155
+ getattr(pd.Series(source), method)(*args, **options)
156
+ with pytest.raises(type(pandas_error.value)) as microdf_error:
157
+ getattr(source, method)(*args, **options)
158
+ # pandas identifies the concrete subclass in invalid-axis messages.
159
+ expected_error = str(pandas_error.value).replace(
160
+ "object type Series", "object type MicroSeries"
161
+ )
162
+ assert str(microdf_error.value) == expected_error
163
+
164
+
165
+ @pytest.mark.parametrize(
166
+ "operation",
167
+ [
168
+ lambda plain, weighted: plain + weighted,
169
+ lambda plain, weighted: plain - weighted,
170
+ lambda plain, weighted: plain * weighted,
171
+ lambda plain, weighted: plain / weighted,
172
+ lambda plain, weighted: plain // weighted,
173
+ lambda plain, weighted: plain % weighted,
174
+ lambda plain, weighted: plain**weighted,
175
+ lambda plain, weighted: plain & weighted,
176
+ lambda plain, weighted: plain | weighted,
177
+ lambda plain, weighted: plain ^ weighted,
178
+ ],
179
+ ids=ARITHMETIC + LOGICAL,
180
+ )
181
+ @pytest.mark.parametrize("indexes", ["matching", "permuted", "duplicates"])
182
+ def test_plain_series_left_expressions_preserve_weighted_dispatch(operation, indexes):
183
+ index = ["a", "a"] if indexes == "duplicates" else ["b", "a"]
184
+ source = MicroSeries([10, 20], index=index, weights=[1, 9], name="x")
185
+ other_index = ["a", "b"] if indexes == "permuted" else index
186
+ other = pd.Series([2, 1], index=other_index, name="x")
187
+ expected = operation(other, pd.Series(source))
188
+
189
+ result = operation(other, source)
190
+
191
+ assert_weighted_result(result, expected, source)
192
+ result.weights.iloc[0] = 100
193
+ np.testing.assert_array_equal(source.weights, [1, 9])
194
+ source.weights.iloc[1] = 200
195
+ assert result.weights.iloc[0] == 100
@@ -59,3 +59,50 @@ def test_stored_weight_edits_do_not_change_weight_column(dtype):
59
59
 
60
60
  np.testing.assert_array_equal(df["w"], [1, 2])
61
61
  assert df.sum()["x"] == 10 * 100 + 20 * 2
62
+
63
+
64
+ @pytest.mark.parametrize("drop", [False, True])
65
+ @pytest.mark.parametrize("inplace", [False, True])
66
+ @pytest.mark.parametrize(
67
+ "index,level",
68
+ [
69
+ (pd.Index(["b", "a", "a"], name="row"), None),
70
+ (
71
+ pd.MultiIndex.from_tuples(
72
+ [("b", 2), ("a", 1), ("a", 1)], names=["group", "row"]
73
+ ),
74
+ None,
75
+ ),
76
+ (
77
+ pd.MultiIndex.from_tuples(
78
+ [("b", 2), ("a", 1), ("a", 1)], names=["group", "row"]
79
+ ),
80
+ "group",
81
+ ),
82
+ ],
83
+ )
84
+ def test_reset_index_owns_independently_mutable_weights(index, level, drop, inplace):
85
+ source = mdf.MicroDataFrame({"x": [10, 20, 30]}, index=index, weights=[1, 9, 3])
86
+ original_weights = source.weights
87
+ expected = pd.DataFrame(source).reset_index(level=level, drop=drop)
88
+
89
+ result = source.reset_index(level=level, drop=drop, inplace=inplace)
90
+
91
+ if inplace:
92
+ assert result is None
93
+ result = source
94
+ assert isinstance(result, mdf.MicroDataFrame)
95
+ pd.testing.assert_frame_equal(pd.DataFrame(result), expected)
96
+ pd.testing.assert_series_equal(
97
+ result.weights, pd.Series([1.0, 9.0, 3.0], index=expected.index)
98
+ )
99
+ assert result.x.sum() == 10 * 1 + 20 * 9 + 30 * 3
100
+ assert result.weights is not original_weights
101
+ result.weights.iloc[0] = 100
102
+ np.testing.assert_array_equal(original_weights, [1, 9, 3])
103
+ assert result.x.sum() == 10 * 100 + 20 * 9 + 30 * 3
104
+ if not inplace:
105
+ assert source.x.sum() == 10 * 1 + 20 * 9 + 30 * 3
106
+ original_weights.iloc[1] = 200
107
+ np.testing.assert_array_equal(result.weights, [100, 9, 3])
108
+ assert result.x.sum() == 10 * 100 + 20 * 9 + 30 * 3
@@ -0,0 +1,640 @@
1
+ """Binary NumPy dispatch retains the weights of identifiable observations."""
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+ import pytest
6
+
7
+ from microdf import MicroSeries
8
+
9
+
10
+ def test_maximum_preserves_issue_322_weighted_total():
11
+ weighted = MicroSeries([10, 20], weights=[2, 9])
12
+
13
+ result = np.maximum(weighted, pd.Series([12, 10]))
14
+
15
+ assert isinstance(result, MicroSeries)
16
+ pd.testing.assert_series_equal(pd.Series(result), pd.Series([12, 20]))
17
+ pd.testing.assert_series_equal(result.weights, pd.Series([2.0, 9.0]))
18
+ assert result.sum() == 204 # 12 * 2 + 20 * 9.
19
+
20
+
21
+ def test_plain_series_left_divmod_preserves_issue_322_weighted_totals():
22
+ weighted = MicroSeries([10, 20], weights=[2, 9])
23
+
24
+ quotient, remainder = divmod(pd.Series([23, 41]), weighted)
25
+
26
+ for result, values, total in [
27
+ (quotient, [2, 2], 22), # 2 * 2 + 2 * 9.
28
+ (remainder, [3, 1], 15), # 3 * 2 + 1 * 9.
29
+ ]:
30
+ assert isinstance(result, MicroSeries)
31
+ pd.testing.assert_series_equal(pd.Series(result), pd.Series(values))
32
+ pd.testing.assert_series_equal(result.weights, pd.Series([2.0, 9.0]))
33
+ assert result.sum() == total
34
+
35
+
36
+ @pytest.mark.parametrize("ufunc", [np.maximum, np.minimum, np.fmax])
37
+ @pytest.mark.parametrize("weighted_first", [True, False])
38
+ @pytest.mark.parametrize(
39
+ "source_labels,other_labels,other_values,other_name,expected_weights",
40
+ [
41
+ (["b", "a", "c"], ["b", "a", "c"], [12, 10, 30], "income", [2, 9, 5]),
42
+ (["b", "a", "c"], ["c", "a", "b"], [30, 10, 12], "other", [9, 2, 5]),
43
+ (["a", "a", "b"], ["a", "a", "b"], [12, 10, 30], "income", [2, 9, 5]),
44
+ ],
45
+ ids=["matching", "reordered", "equal-duplicates"],
46
+ )
47
+ def test_binary_ufunc_preserves_values_row_weights_and_independence(
48
+ ufunc,
49
+ weighted_first,
50
+ source_labels,
51
+ other_labels,
52
+ other_values,
53
+ other_name,
54
+ expected_weights,
55
+ ):
56
+ source = pd.Series(
57
+ [10.0, 20.0, np.nan],
58
+ index=pd.Index(source_labels, name="person"),
59
+ name="income",
60
+ )
61
+ other = pd.Series(
62
+ other_values,
63
+ index=pd.Index(other_labels, name="person"),
64
+ name=other_name,
65
+ )
66
+ weighted = MicroSeries(source, weights=[2, 9, 5])
67
+ if weighted_first:
68
+ expected = ufunc(source, other)
69
+ result = ufunc(weighted, other)
70
+ else:
71
+ expected = ufunc(other, source)
72
+ result = ufunc(other, weighted)
73
+ weights = pd.Series(expected_weights, index=expected.index, dtype=float)
74
+
75
+ assert isinstance(result, MicroSeries)
76
+ pd.testing.assert_series_equal(pd.Series(result), expected)
77
+ pd.testing.assert_series_equal(result.weights, weights)
78
+ assert result.sum() == expected.multiply(weights).sum()
79
+
80
+ result.weights.iloc[0] = 100
81
+ pd.testing.assert_series_equal(
82
+ weighted.weights, pd.Series([2.0, 9.0, 5.0], index=source.index)
83
+ )
84
+ weighted.weights.iloc[-1] = 200
85
+ assert result.weights.iloc[-1] == expected_weights[-1]
86
+
87
+
88
+ @pytest.mark.parametrize("operation", [divmod, np.divmod])
89
+ @pytest.mark.parametrize(
90
+ "source_labels,other_labels,other_values,other_name,expected_weights",
91
+ [
92
+ (["b", "a", "c"], ["b", "a", "c"], [23, 41, 95], "income", [2, 9, 5]),
93
+ (["b", "a", "c"], ["c", "a", "b"], [95, 41, 23], "other", [9, 2, 5]),
94
+ (["a", "a", "b"], ["a", "a", "b"], [23, 41, 95], "income", [2, 9, 5]),
95
+ ],
96
+ ids=["matching", "reordered", "equal-duplicates"],
97
+ )
98
+ def test_reverse_divmod_retains_each_members_values_and_independent_weights(
99
+ operation,
100
+ source_labels,
101
+ other_labels,
102
+ other_values,
103
+ other_name,
104
+ expected_weights,
105
+ ):
106
+ source = pd.Series(
107
+ [10, 20, 30],
108
+ index=pd.Index(source_labels, name="person"),
109
+ name="income",
110
+ )
111
+ other = pd.Series(
112
+ other_values,
113
+ index=pd.Index(other_labels, name="person"),
114
+ name=other_name,
115
+ )
116
+ weighted = MicroSeries(source, weights=[2, 9, 5])
117
+ expected = operation(other, source)
118
+
119
+ result = operation(other, weighted)
120
+
121
+ assert isinstance(result, tuple)
122
+ assert len(result) == 2
123
+ for member, plain in zip(result, expected):
124
+ weights = pd.Series(expected_weights, index=plain.index, dtype=float)
125
+ assert isinstance(member, MicroSeries)
126
+ pd.testing.assert_series_equal(pd.Series(member), plain)
127
+ pd.testing.assert_series_equal(member.weights, weights)
128
+ assert member.sum() == plain.multiply(weights).sum()
129
+
130
+ quotient, remainder = result
131
+ quotient.weights.iloc[0] = 100
132
+ assert remainder.weights.iloc[0] == expected_weights[0]
133
+ remainder.weights.iloc[1] = 300
134
+ assert quotient.weights.iloc[1] == expected_weights[1]
135
+ pd.testing.assert_series_equal(
136
+ weighted.weights, pd.Series([2.0, 9.0, 5.0], index=source.index)
137
+ )
138
+ weighted.weights.iloc[-1] = 200
139
+ assert quotient.weights.iloc[-1] == expected_weights[-1]
140
+ assert remainder.weights.iloc[-1] == expected_weights[-1]
141
+
142
+
143
+ @pytest.mark.parametrize(
144
+ "operation,source_labels,other_labels",
145
+ [
146
+ (np.maximum, ["a", "b", "c"], ["a", "b", "unknown"]),
147
+ (divmod, ["a", "b", "c"], ["a", "b", "unknown"]),
148
+ (np.divmod, ["a", "b", "c"], ["a", "b", "unknown"]),
149
+ (divmod, ["b", "a", "a"], ["a", "a", "b"]),
150
+ (np.divmod, ["b", "a", "a"], ["a", "a", "b"]),
151
+ ],
152
+ ids=[
153
+ "maximum-unknown-row",
154
+ "divmod-unknown-row",
155
+ "numpy-divmod-unknown-row",
156
+ "divmod-ambiguous-duplicate-rows",
157
+ "numpy-divmod-ambiguous-duplicate-rows",
158
+ ],
159
+ )
160
+ def test_binary_dispatch_rejects_rows_without_unambiguous_weights(
161
+ operation, source_labels, other_labels
162
+ ):
163
+ source = pd.Series([10, 20, 30], index=source_labels)
164
+ other = pd.Series([23, 41, 95], index=other_labels)
165
+ weighted = MicroSeries(source, weights=[2, 9, 5])
166
+ # Pandas can produce values, but these rows have no unique weight assignment.
167
+ operation(other, source)
168
+
169
+ with pytest.raises(ValueError, match="weights"):
170
+ operation(other, weighted)
171
+
172
+
173
+ @pytest.mark.parametrize("ufunc", [np.maximum, np.minimum, np.fmax])
174
+ def test_binary_ufunc_preserves_pandas_duplicate_alignment_errors(ufunc):
175
+ source = pd.Series([10, 20, 30], index=["b", "a", "a"])
176
+ other = pd.Series([23, 41, 95], index=["a", "a", "b"])
177
+ weighted = MicroSeries(source, weights=[2, 9, 5])
178
+
179
+ with pytest.raises(ValueError) as pandas_error:
180
+ ufunc(other, source)
181
+ with pytest.raises(type(pandas_error.value)) as microdf_error:
182
+ ufunc(other, weighted)
183
+
184
+ assert str(microdf_error.value) == str(pandas_error.value)
185
+
186
+
187
+ @pytest.mark.parametrize("operation", [np.maximum, divmod, np.divmod])
188
+ def test_binary_dispatch_preserves_pandas_invalid_dtype_errors(operation):
189
+ source = pd.Series([10, 20], index=["a", "b"], name="income")
190
+ other = pd.Series(["invalid", "data"], index=source.index, dtype=object)
191
+ weighted = MicroSeries(source, weights=[2, 9])
192
+
193
+ with pytest.raises(TypeError) as pandas_error:
194
+ operation(other, source)
195
+ with pytest.raises(type(pandas_error.value)) as microdf_error:
196
+ operation(other, weighted)
197
+
198
+ assert str(microdf_error.value) == str(pandas_error.value)
199
+
200
+
201
+ @pytest.mark.parametrize("weighted_first", [True, False])
202
+ def test_binary_ufunc_preserves_pandas_ndarray_out_and_where(weighted_first):
203
+ source = pd.Series([10.0, 20.0, 30.0], index=["b", "a", "c"], name="income")
204
+ other = pd.Series([12.0, 10.0, 40.0], index=source.index, name="income")
205
+ weighted = MicroSeries(source, weights=[2, 9, 5])
206
+ out = np.full(3, -99.0)
207
+ plain_out = out.copy()
208
+ plain_inputs = (source, other) if weighted_first else (other, source)
209
+ inputs = (weighted, other) if weighted_first else (other, weighted)
210
+ where = np.array([True, False, True])
211
+ expected = np.maximum(*plain_inputs, out=plain_out, where=where)
212
+
213
+ result = np.maximum(*inputs, out=out, where=where)
214
+
215
+ assert (result is out) == (expected is plain_out)
216
+ np.testing.assert_array_equal(out, [12.0, -99.0, 40.0])
217
+ np.testing.assert_array_equal(out, plain_out)
218
+ pd.testing.assert_series_equal(pd.Series(result), expected)
219
+ assert np.shares_memory(np.asarray(result), out) == np.shares_memory(
220
+ np.asarray(expected), plain_out
221
+ )
222
+
223
+
224
+ @pytest.mark.parametrize("weighted_first", [True, False])
225
+ def test_binary_ufunc_explicit_none_out_retains_weights(weighted_first):
226
+ source = pd.Series([10, 20], index=["b", "a"], name="income")
227
+ other = pd.Series([12, 10], index=source.index, name="income")
228
+ weighted = MicroSeries(source, weights=[2, 9])
229
+ plain_inputs = (source, other) if weighted_first else (other, source)
230
+ inputs = (weighted, other) if weighted_first else (other, weighted)
231
+ expected = np.maximum(*plain_inputs, out=None)
232
+
233
+ result = np.maximum(*inputs, out=None)
234
+
235
+ assert isinstance(result, MicroSeries)
236
+ pd.testing.assert_series_equal(pd.Series(result), expected)
237
+ pd.testing.assert_series_equal(
238
+ result.weights, pd.Series([2.0, 9.0], index=source.index)
239
+ )
240
+ assert result.sum() == 204 # 12 * 2 + 20 * 9.
241
+
242
+
243
+ @pytest.mark.parametrize("ufunc", [np.negative, np.modf])
244
+ def test_unary_ufunc_and_tuple_outputs_keep_existing_weight_behavior(ufunc):
245
+ source = pd.Series([10.25, -20.5], index=["b", "a"], name="income")
246
+ weighted = MicroSeries(source, weights=[2, 9])
247
+ expected = ufunc(source)
248
+
249
+ result = ufunc(weighted)
250
+
251
+ if isinstance(expected, tuple):
252
+ assert isinstance(result, tuple)
253
+ assert len(result) == len(expected)
254
+ else:
255
+ result, expected = (result,), (expected,)
256
+ weights = pd.Series([2.0, 9.0], index=source.index)
257
+ for member, plain in zip(result, expected):
258
+ assert isinstance(member, MicroSeries)
259
+ pd.testing.assert_series_equal(pd.Series(member), plain)
260
+ pd.testing.assert_series_equal(member.weights, weights)
261
+ assert member.sum() == plain.multiply(weights).sum()
262
+
263
+
264
+ def test_add_ufunc_reduction_keeps_weighted_sum_behavior():
265
+ weighted = MicroSeries([10, 20], index=["b", "a"], weights=[2, 9])
266
+
267
+ assert np.add.reduce(weighted) == 200 # 10 * 2 + 20 * 9.
268
+
269
+
270
+ @pytest.mark.parametrize("base", [pd.Series, pd.DataFrame])
271
+ @pytest.mark.parametrize("weighted_first", [True, False])
272
+ @pytest.mark.parametrize("higher_priority", [False, True])
273
+ def test_binary_ufunc_defers_to_foreign_pandas_handlers(
274
+ base, weighted_first, higher_priority
275
+ ):
276
+ sentinel = object()
277
+ calls = []
278
+
279
+ class ForeignPandasObject(base):
280
+ __array_priority__ = MicroSeries.__array_priority__ + higher_priority
281
+
282
+ def __array_ufunc__(self, ufunc, method, *inputs, **kwargs):
283
+ calls.append((ufunc, method, inputs, kwargs))
284
+ return sentinel
285
+
286
+ weighted = MicroSeries([10, 20], weights=[2, 9])
287
+ foreign = ForeignPandasObject([12, 10])
288
+ inputs = (weighted, foreign) if weighted_first else (foreign, weighted)
289
+
290
+ result = np.maximum(*inputs)
291
+
292
+ assert result is sentinel
293
+ assert len(calls) == 1
294
+ ufunc, method, received_inputs, kwargs = calls[0]
295
+ assert ufunc is np.maximum
296
+ assert method == "__call__"
297
+ assert len(received_inputs) == len(inputs)
298
+ assert all(
299
+ received is original for received, original in zip(received_inputs, inputs)
300
+ )
301
+ assert kwargs == {}
302
+
303
+
304
+ @pytest.mark.parametrize("weighted_first", [True, False])
305
+ def test_binary_ufunc_defers_to_higher_priority_dataframe_subclass(weighted_first):
306
+ class HigherPriorityObject(pd.DataFrame):
307
+ __array_priority__ = MicroSeries.__array_priority__ + 1
308
+
309
+ source = pd.Series([10, 20])
310
+ weighted = MicroSeries(source, weights=[2, 9])
311
+ higher = HigherPriorityObject([12, 10])
312
+ plain_inputs = (source, higher) if weighted_first else (higher, source)
313
+ inputs = (weighted, higher) if weighted_first else (higher, weighted)
314
+
315
+ # Test the deferral protocol directly: the foreign handler decides how
316
+ # to handle the operation after MicroSeries returns NotImplemented.
317
+ assert (
318
+ source.__array_ufunc__(np.maximum, "__call__", *plain_inputs) is NotImplemented
319
+ )
320
+ assert weighted.__array_ufunc__(np.maximum, "__call__", *inputs) is NotImplemented
321
+
322
+
323
+ @pytest.mark.parametrize("ufunc", [np.divmod, np.maximum, np.add])
324
+ @pytest.mark.parametrize("weighted_first", [True, False])
325
+ def test_binary_ufunc_preserves_pandas_result_metadata(ufunc, weighted_first):
326
+ source = pd.Series([10, 20], index=["b", "a"], name="income")
327
+ other = pd.Series([23, 41], index=["a", "b"], name="income")
328
+ source.attrs = {"units": {"currency": "USD"}}
329
+ other.attrs = {"units": {"currency": "EUR"}}
330
+ weighted = MicroSeries(source, weights=[2, 9])
331
+ weighted.attrs = source.attrs.copy()
332
+ plain_inputs = (source, other) if weighted_first else (other, source)
333
+ inputs = (weighted, other) if weighted_first else (other, weighted)
334
+ expected = ufunc(*plain_inputs)
335
+
336
+ actual = ufunc(*inputs)
337
+
338
+ if not isinstance(expected, tuple):
339
+ actual, expected = (actual,), (expected,)
340
+ for result, plain in zip(actual, expected):
341
+ pd.testing.assert_series_equal(pd.Series(result), plain)
342
+ assert result.attrs == plain.attrs
343
+ if result.attrs:
344
+ result.attrs["units"]["currency"] = "changed"
345
+ assert source.attrs == weighted.attrs == {"units": {"currency": "USD"}}
346
+ assert other.attrs == {"units": {"currency": "EUR"}}
347
+
348
+
349
+ @pytest.mark.parametrize("weighted_first", [True, False])
350
+ @pytest.mark.parametrize("other_labels", [["b", "a"], ["a", "b"]])
351
+ def test_maximum_with_inherited_higher_priority_series_completes(
352
+ weighted_first, other_labels
353
+ ):
354
+ class HigherPrioritySeries(pd.Series):
355
+ __array_priority__ = MicroSeries.__array_priority__ + 1
356
+
357
+ source = pd.Series(
358
+ [10, 20], index=pd.Index(["b", "a"], name="person"), name="income"
359
+ )
360
+ source.attrs = {"units": {"currency": "USD"}}
361
+ other = HigherPrioritySeries(
362
+ [12.5, 10.5],
363
+ index=pd.Index(other_labels, name="person"),
364
+ name="income",
365
+ )
366
+ other.attrs = {"units": {"currency": "EUR"}}
367
+ weighted = MicroSeries(source, weights=[2, 9])
368
+ weighted.attrs = source.attrs.copy()
369
+ plain_inputs = (source, other) if weighted_first else (other, source)
370
+ inputs = (weighted, other) if weighted_first else (other, weighted)
371
+ expected = np.maximum(*plain_inputs)
372
+ weights = pd.Series([2.0, 9.0], index=source.index).reindex(expected.index)
373
+
374
+ result = np.maximum(*inputs)
375
+
376
+ assert isinstance(result, MicroSeries)
377
+ pd.testing.assert_series_equal(pd.Series(result), expected)
378
+ pd.testing.assert_series_equal(result.weights, weights)
379
+ assert result.attrs == expected.attrs
380
+ assert result.sum() == expected.multiply(weights).sum()
381
+ result.weights.iloc[0] = 100
382
+ pd.testing.assert_series_equal(
383
+ weighted.weights, pd.Series([2.0, 9.0], index=source.index)
384
+ )
385
+ weighted.weights.iloc[-1] = 200
386
+ assert result.weights.iloc[-1] == weights.iloc[-1]
387
+ assert source.attrs == weighted.attrs == {"units": {"currency": "USD"}}
388
+ assert other.attrs == {"units": {"currency": "EUR"}}
389
+
390
+
391
+ @pytest.mark.parametrize("weighted_first", [True, False])
392
+ def test_higher_priority_inherited_series_preserves_masked_out(weighted_first):
393
+ class HigherPrioritySeries(pd.Series):
394
+ __array_priority__ = MicroSeries.__array_priority__ + 1
395
+
396
+ source = pd.Series([10.0, 20.0], index=["b", "a"], name="income")
397
+ other = HigherPrioritySeries([12.0, 10.0], index=source.index, name="income")
398
+ weighted = MicroSeries(source, weights=[2, 9])
399
+ out = np.full(2, -99.0)
400
+ plain_out = out.copy()
401
+ plain_inputs = (source, other) if weighted_first else (other, source)
402
+ inputs = (weighted, other) if weighted_first else (other, weighted)
403
+ expected = np.maximum(*plain_inputs, out=plain_out, where=[True, False])
404
+
405
+ result = np.maximum(*inputs, out=out, where=[True, False])
406
+
407
+ assert (result is out) == (expected is plain_out)
408
+ np.testing.assert_array_equal(out, [12.0, -99.0])
409
+ np.testing.assert_array_equal(out, plain_out)
410
+ pd.testing.assert_series_equal(pd.Series(result), expected)
411
+ assert result.attrs == expected.attrs
412
+ assert np.shares_memory(np.asarray(result), out) == np.shares_memory(
413
+ np.asarray(expected), plain_out
414
+ )
415
+
416
+
417
+ @pytest.mark.parametrize("ufunc", [np.maximum, np.minimum, np.fmax])
418
+ @pytest.mark.parametrize("weighted_first", [True, False])
419
+ def test_inherited_priority_series_preserves_unsorted_dispatch_order(
420
+ ufunc, weighted_first
421
+ ):
422
+ class HigherPrioritySeries(pd.Series):
423
+ __array_priority__ = MicroSeries.__array_priority__ + 1
424
+
425
+ source = pd.Series(
426
+ [6.0, 18.0, 30.0], index=pd.Index(["b", "a", "c"], name="row"), name="amount"
427
+ )
428
+ other = HigherPrioritySeries(
429
+ [9.0, 15.0, 45.0], index=pd.Index(["c", "a", "b"], name="row"), name="amount"
430
+ )
431
+ weighted = MicroSeries(source, weights=[7, 2, 5])
432
+ plain_inputs = (source, other) if weighted_first else (other, source)
433
+ inputs = (weighted, other) if weighted_first else (other, weighted)
434
+ expected = ufunc(*plain_inputs)
435
+ weights = pd.Series([7.0, 2.0, 5.0], index=source.index).reindex(expected.index)
436
+
437
+ result = ufunc(*inputs)
438
+
439
+ assert expected.index.tolist() == (
440
+ ["c", "a", "b"] if weighted_first else ["a", "b", "c"]
441
+ )
442
+ assert isinstance(result, MicroSeries)
443
+ pd.testing.assert_series_equal(pd.Series(result), expected)
444
+ pd.testing.assert_series_equal(result.weights, weights)
445
+ assert result.sum() == expected.multiply(weights).sum()
446
+
447
+
448
+ @pytest.mark.parametrize("weighted_first", [True, False])
449
+ @pytest.mark.parametrize("output_kind", ["array", "series", "micro", "none"])
450
+ def test_inherited_priority_output_preserves_unsorted_dispatch_order(
451
+ weighted_first, output_kind
452
+ ):
453
+ class HigherPrioritySeries(pd.Series):
454
+ __array_priority__ = MicroSeries.__array_priority__ + 1
455
+
456
+ source = pd.Series([6.0, 18.0, 30.0], index=["b", "a", "c"], name="amount")
457
+ other = HigherPrioritySeries(
458
+ [9.0, 15.0, 45.0], index=["c", "a", "b"], name="amount"
459
+ )
460
+ weighted = MicroSeries(source, weights=[7, 2, 5])
461
+ plain_inputs = (source, other) if weighted_first else (other, source)
462
+ inputs = (weighted, other) if weighted_first else (other, weighted)
463
+ kwargs = {}
464
+ if output_kind == "array":
465
+ out = np.full(3, -77.0)
466
+ plain_out = out.copy()
467
+ kwargs["where"] = [True, False, True]
468
+ elif output_kind in ("series", "micro"):
469
+ plain_out = pd.Series([-77.0] * 3, index=source.index, name="destination")
470
+ out = (
471
+ MicroSeries(plain_out, weights=[19, 23, 29])
472
+ if output_kind == "micro"
473
+ else plain_out.copy()
474
+ )
475
+ else:
476
+ out = plain_out = None
477
+ expected = np.maximum(*plain_inputs, out=plain_out, **kwargs)
478
+
479
+ result = np.maximum(*inputs, out=out, **kwargs)
480
+
481
+ assert expected.index.tolist() == (
482
+ ["c", "a", "b"] if weighted_first else ["a", "b", "c"]
483
+ )
484
+ pd.testing.assert_series_equal(pd.Series(result), expected)
485
+ if out is not None:
486
+ np.testing.assert_array_equal(np.asarray(out), np.asarray(plain_out))
487
+ assert (result is out) == (expected is plain_out)
488
+ assert np.shares_memory(
489
+ np.asarray(result), np.asarray(out)
490
+ ) == np.shares_memory(np.asarray(expected), np.asarray(plain_out))
491
+ if output_kind == "array":
492
+ np.testing.assert_array_equal(
493
+ out, [30.0, -77.0, 45.0] if weighted_first else [18.0, -77.0, 30.0]
494
+ )
495
+ if output_kind == "micro":
496
+ np.testing.assert_array_equal(out.weights, [19, 23, 29])
497
+ np.testing.assert_array_equal(weighted.weights, [7, 2, 5])
498
+
499
+
500
+ @pytest.mark.parametrize("weighted_first", [True, False])
501
+ def test_inherited_priority_series_rejects_unknown_row_weights(weighted_first):
502
+ class HigherPrioritySeries(pd.Series):
503
+ __array_priority__ = MicroSeries.__array_priority__ + 1
504
+
505
+ source = pd.Series([6, 18, 30], index=["b", "a", "c"])
506
+ other = HigherPrioritySeries([9, 15, 45], index=["c", "a", "unknown"])
507
+ weighted = MicroSeries(source, weights=[7, 2, 5])
508
+ plain_inputs = (source, other) if weighted_first else (other, source)
509
+ inputs = (weighted, other) if weighted_first else (other, weighted)
510
+ # Values are defined by pandas, but the extra row has no observation weight.
511
+ np.maximum(*plain_inputs)
512
+
513
+ with pytest.raises(ValueError, match="weights"):
514
+ np.maximum(*inputs)
515
+
516
+
517
+ @pytest.mark.parametrize("weighted_first", [True, False])
518
+ @pytest.mark.parametrize(
519
+ "row_error,output_kind,masked",
520
+ [
521
+ ("unknown-input", "input", False),
522
+ ("unknown-input", "array", False),
523
+ ("unknown-input", "array", True),
524
+ ("unknown-input", "series", False),
525
+ ("unknown-input", "micro", False),
526
+ ("unknown-output", "micro", False),
527
+ ("ambiguous-output", "micro", False),
528
+ ],
529
+ )
530
+ def test_inherited_priority_row_rejection_preserves_inputs_and_outputs(
531
+ weighted_first, row_error, output_kind, masked
532
+ ):
533
+ class HigherPrioritySeries(pd.Series):
534
+ __array_priority__ = MicroSeries.__array_priority__ + 1
535
+
536
+ source = pd.Series(
537
+ [1.0, 2.0, 4.0], index=pd.Index(["b", "a", "c"], name="row"), name="amount"
538
+ )
539
+ weighted = MicroSeries(source, weights=[3, 7, 11])
540
+ weighted.attrs = {"units": {"currency": "USD"}}
541
+ other_labels = (
542
+ ["b", "a", "unknown"] if row_error == "unknown-input" else source.index
543
+ )
544
+ higher = HigherPrioritySeries([9.0, 8.0, 6.0], index=other_labels, name="amount")
545
+ inputs = (weighted, higher) if weighted_first else (higher, weighted)
546
+ result_index = higher.index if weighted_first else source.index.union(higher.index)
547
+ if output_kind == "input":
548
+ out = weighted
549
+ elif output_kind == "array":
550
+ out = np.full(len(result_index), -99.0)
551
+ else:
552
+ output_labels = (
553
+ ["b", "a", "unknown"]
554
+ if row_error == "unknown-output"
555
+ else ["b", "a", "a"]
556
+ if row_error == "ambiguous-output"
557
+ else result_index
558
+ )
559
+ destination = pd.Series(-99.0, index=output_labels, name="destination")
560
+ out = (
561
+ MicroSeries(destination, weights=np.arange(len(destination)) + 19)
562
+ if output_kind == "micro"
563
+ else destination
564
+ )
565
+ out.attrs = {"purpose": "unchanged"}
566
+ original_output = np.array(out, copy=True)
567
+ original_weights = weighted.weights.copy()
568
+ output_weights = out.weights.copy() if isinstance(out, MicroSeries) else None
569
+ original_index = out.index.copy() if isinstance(out, pd.Series) else None
570
+ original_attrs = out.attrs.copy() if isinstance(out, pd.Series) else None
571
+ kwargs = {"where": np.arange(len(out)) % 2 == 0} if masked else {}
572
+
573
+ with pytest.raises(ValueError):
574
+ np.maximum(*inputs, out=out, **kwargs)
575
+
576
+ pd.testing.assert_series_equal(pd.Series(weighted), source)
577
+ pd.testing.assert_series_equal(weighted.weights, original_weights)
578
+ assert weighted.attrs == {"units": {"currency": "USD"}}
579
+ assert weighted.sum() == 61 # 1 * 3 + 2 * 7 + 4 * 11.
580
+ np.testing.assert_array_equal(np.asarray(out), original_output)
581
+ if isinstance(out, pd.Series):
582
+ pd.testing.assert_index_equal(out.index, original_index)
583
+ assert out.attrs == original_attrs
584
+ assert out.name == ("amount" if output_kind == "input" else "destination")
585
+ if output_weights is not None:
586
+ pd.testing.assert_series_equal(out.weights, output_weights)
587
+
588
+
589
+ @pytest.mark.parametrize("weighted_first", [True, False])
590
+ def test_inherited_priority_valid_aliased_output_matches_pandas(weighted_first):
591
+ class HigherPrioritySeries(pd.Series):
592
+ __array_priority__ = MicroSeries.__array_priority__ + 1
593
+
594
+ source = pd.Series([6.0, 18.0, 30.0], index=["b", "a", "c"], name="amount")
595
+ higher = HigherPrioritySeries(
596
+ [9.0, 15.0, 45.0], index=["c", "a", "b"], name="amount"
597
+ )
598
+ weighted = MicroSeries(source.copy(), weights=[7, 2, 5])
599
+ plain_inputs = (source, higher) if weighted_first else (higher, source)
600
+ inputs = (weighted, higher) if weighted_first else (higher, weighted)
601
+ expected = np.maximum(*plain_inputs, out=source)
602
+
603
+ result = np.maximum(*inputs, out=weighted)
604
+
605
+ pd.testing.assert_series_equal(pd.Series(weighted), source)
606
+ pd.testing.assert_series_equal(pd.Series(result), expected)
607
+ pd.testing.assert_series_equal(
608
+ weighted.weights, pd.Series([7.0, 2.0, 5.0], index=source.index)
609
+ )
610
+ assert (result is weighted) == (expected is source)
611
+ assert np.shares_memory(
612
+ np.asarray(result), np.asarray(weighted)
613
+ ) == np.shares_memory(np.asarray(expected), np.asarray(source))
614
+
615
+
616
+ @pytest.mark.parametrize("weighted_first", [True, False])
617
+ def test_foreign_ufunc_handler_receives_out_before_weight_validation(weighted_first):
618
+ calls = []
619
+ sentinel = object()
620
+
621
+ class ForeignSeries(pd.Series):
622
+ __array_priority__ = MicroSeries.__array_priority__ + 1
623
+
624
+ def __array_ufunc__(self, ufunc, method, *inputs, **kwargs):
625
+ calls.append((ufunc, method, inputs, kwargs))
626
+ return sentinel
627
+
628
+ weighted = MicroSeries([1.0, 2.0], index=["a", "b"], weights=[3, 7])
629
+ other = ForeignSeries([9.0, 8.0], index=["a", "unknown"])
630
+ inputs = (weighted, other) if weighted_first else (other, weighted)
631
+ out = np.full(2, -99.0)
632
+
633
+ assert np.maximum(*inputs, out=out) is sentinel
634
+
635
+ assert len(calls) == 1
636
+ ufunc, method, received_inputs, kwargs = calls[0]
637
+ assert ufunc is np.maximum and method == "__call__"
638
+ assert all(actual is expected for actual, expected in zip(received_inputs, inputs))
639
+ assert kwargs["out"][0] is out
640
+ np.testing.assert_array_equal(out, [-99.0, -99.0])
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.5.0
3
+ Version: 1.5.2
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -8,6 +8,7 @@ microdf/microseries.py
8
8
  microdf/replication.py
9
9
  microdf/tests/conftest.py
10
10
  microdf/tests/test_aggregation_errors.py
11
+ microdf/tests/test_binary_weight_alignment.py
11
12
  microdf/tests/test_dataframe_weight_storage.py
12
13
  microdf/tests/test_microseries_dataframe.py
13
14
  microdf/tests/test_nullify_weights_index.py
@@ -16,6 +17,7 @@ microdf/tests/test_quantile_missing_values.py
16
17
  microdf/tests/test_replication.py
17
18
  microdf/tests/test_serialization.py
18
19
  microdf/tests/test_sum_axes.py
20
+ microdf/tests/test_ufunc_weight_dispatch.py
19
21
  microdf/tests/test_version_metadata.py
20
22
  microdf/tests/test_weight_propagation.py
21
23
  microdf/tests/test_weighted_cov_corr.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "microdf-python"
7
- version = "1.5.0"
7
+ version = "1.5.2"
8
8
  description = "Weighted pandas DataFrames and Series for survey microdata"
9
9
  readme = "README.md"
10
10
  authors = [
File without changes
File without changes
File without changes