microdf-python 1.4.0__tar.gz → 1.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {microdf_python-1.4.0/microdf_python.egg-info → microdf_python-1.4.1}/PKG-INFO +1 -1
  2. microdf_python-1.4.1/microdf/_weights.py +257 -0
  3. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/microdataframe.py +67 -186
  4. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/microseries.py +39 -26
  5. microdf_python-1.4.1/microdf/tests/test_weight_propagation.py +478 -0
  6. {microdf_python-1.4.0 → microdf_python-1.4.1/microdf_python.egg-info}/PKG-INFO +1 -1
  7. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf_python.egg-info/SOURCES.txt +2 -0
  8. {microdf_python-1.4.0 → microdf_python-1.4.1}/pyproject.toml +1 -1
  9. {microdf_python-1.4.0 → microdf_python-1.4.1}/LICENSE +0 -0
  10. {microdf_python-1.4.0 → microdf_python-1.4.1}/README.md +0 -0
  11. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/__init__.py +0 -0
  12. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/conftest.py +0 -0
  13. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_aggregation_errors.py +0 -0
  14. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_dataframe_weight_storage.py +0 -0
  15. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_microseries_dataframe.py +0 -0
  16. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_nullify_weights_index.py +0 -0
  17. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_pandas3_compatibility.py +0 -0
  18. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_quantile_missing_values.py +0 -0
  19. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_serialization.py +0 -0
  20. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_sum_axes.py +0 -0
  21. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_version_metadata.py +0 -0
  22. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf/tests/test_weighted_cov_corr.py +0 -0
  23. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf_python.egg-info/dependency_links.txt +0 -0
  24. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf_python.egg-info/requires.txt +0 -0
  25. {microdf_python-1.4.0 → microdf_python-1.4.1}/microdf_python.egg-info/top_level.txt +0 -0
  26. {microdf_python-1.4.0 → microdf_python-1.4.1}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.4.0
3
+ Version: 1.4.1
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -0,0 +1,257 @@
1
+ """Pandas subclass hooks for copying weights and retaining row positions."""
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+
7
+ def weight_series(values, index):
8
+ """Own an independent float array aligned positionally to an index."""
9
+ return pd.Series(np.array(values, dtype=float, copy=True), index=index)
10
+
11
+
12
+ def aligned_weights(source, index):
13
+ """Align only when row identity can be established without guessing."""
14
+ if source.index.equals(index):
15
+ return weight_series(source.weights, index)
16
+ if source.index.is_unique:
17
+ positions = source.index.get_indexer(index)
18
+ if (positions >= 0).all():
19
+ return weight_series(source.weights.iloc[positions], index)
20
+ raise ValueError(
21
+ "Cannot propagate weights unambiguously to these rows. "
22
+ "Use positional selection, or explicitly supply weights for the result."
23
+ )
24
+
25
+
26
+ def finalize_weights(result, source, method, previous_weights):
27
+ """Propagate row weights after pandas has finalized its own metadata."""
28
+ if method == "transpose" and result.ndim == 2:
29
+ raise ValueError(
30
+ "Cannot transpose row weights onto columns. "
31
+ "Convert to a pandas DataFrame first and supply explicit result weights."
32
+ )
33
+ if method == "concat":
34
+ objects = source.objs
35
+ if not all(isinstance(obj, WeightPropagationMixin) for obj in objects):
36
+ raise ValueError(
37
+ "Cannot concatenate weighted and unweighted objects: "
38
+ "provide explicit weights for every input."
39
+ )
40
+ weights = concat_weights(result, objects, getattr(source, "axis", None))
41
+ result.weights = weights
42
+ if result.ndim == 2:
43
+ names = [obj.__dict__.get("weights_col") for obj in objects]
44
+ result.weights_col = (
45
+ names[0] if all(name == names[0] for name in names) else None
46
+ )
47
+ elif isinstance(source, WeightPropagationMixin):
48
+ if method == "reset_index" and source.ndim == result.ndim == 1:
49
+ # Series.reset_index(drop=True) changes labels, never row order.
50
+ # pandas 2 constructs a result; pandas 3 relabels a shallow copy.
51
+ result.weights = weight_series(source.weights, result.index)
52
+ elif method == "rename" and previous_weights is not None:
53
+ result.weights = weight_series(previous_weights, result.index)
54
+ else:
55
+ result.weights = aligned_weights(source, result.index)
56
+ return result
57
+
58
+
59
+ def concat_weights(result, objects, axis):
60
+ """Recover concat row identity without relying on pandas 2-only state."""
61
+ if objects[0].ndim == 1 and result.ndim == 2:
62
+ axis = 1
63
+ row_weights = None
64
+ if axis != 1 and len(result) == sum(len(obj) for obj in objects):
65
+ combined_index = objects[0].index.append([obj.index for obj in objects[1:]])
66
+ same_rows = result.index.equals(combined_index)
67
+ reset_rows = result.index.equals(pd.RangeIndex(len(result)))
68
+ keyed_rows = False
69
+ if (
70
+ isinstance(result.index, pd.MultiIndex)
71
+ and result.index.nlevels > combined_index.nlevels
72
+ ):
73
+ trailing = result.index.droplevel(
74
+ list(range(result.index.nlevels - combined_index.nlevels))
75
+ )
76
+ keyed_rows = trailing.equals(combined_index)
77
+ if same_rows or reset_rows or keyed_rows:
78
+ row_weights = np.concatenate([np.asarray(obj.weights) for obj in objects])
79
+ column_weights = None
80
+ if (
81
+ axis != 0
82
+ and result.ndim == 2
83
+ and len(result.columns)
84
+ == sum(obj.shape[1] if obj.ndim == 2 else 1 for obj in objects)
85
+ ):
86
+ values = np.empty(len(result))
87
+ known = np.zeros(len(result), dtype=bool)
88
+ conflict = False
89
+ for obj in objects:
90
+ if obj.index.equals(result.index):
91
+ positions = np.arange(len(result))
92
+ elif obj.index.is_unique:
93
+ positions = obj.index.get_indexer(result.index)
94
+ else:
95
+ conflict = True
96
+ break
97
+ present = positions >= 0
98
+ incoming = np.asarray(obj.weights)[positions[present]]
99
+ overlap = known[present]
100
+ if not np.array_equal(
101
+ values[present][overlap], incoming[overlap], equal_nan=True
102
+ ):
103
+ conflict = True
104
+ break
105
+ values[present] = incoming
106
+ known[present] = True
107
+ if not conflict and known.all():
108
+ column_weights = values
109
+ if row_weights is not None:
110
+ if column_weights is not None and not np.array_equal(
111
+ row_weights, column_weights, equal_nan=True
112
+ ):
113
+ raise ValueError(
114
+ "Ambiguous weights: pandas did not expose the concat axis."
115
+ )
116
+ return weight_series(row_weights, result.index)
117
+ if column_weights is not None:
118
+ return weight_series(column_weights, result.index)
119
+ raise ValueError(
120
+ "Cannot propagate concat weights: conflicting or ambiguous row weights."
121
+ )
122
+
123
+
124
+ class WeightPropagationMixin:
125
+ """Use positional provenance for pandas operations that select rows."""
126
+
127
+ def _plain(self):
128
+ return (
129
+ pd.Series(self, copy=False)
130
+ if self.ndim == 1
131
+ else pd.DataFrame(self, copy=False)
132
+ )
133
+
134
+ def _weighted_result(self, plain, weights):
135
+ result = type(self)(plain, weights=weight_series(weights, plain.index))
136
+ if self.ndim == 2:
137
+ result.weights_col = self.__dict__.get("weights_col")
138
+ return result
139
+
140
+ def _finish_row_operation(self, plain, positions, inplace=False):
141
+ result = self._weighted_result(plain, self.weights.iloc[positions])
142
+ if inplace:
143
+ self._update_inplace(result)
144
+ self.weights = result.weights
145
+ return None
146
+ return result
147
+
148
+ def take(self, indices, axis=0, **kwargs):
149
+ """Take values and weights using the same positional indexer."""
150
+ axis = self._get_axis_number(axis)
151
+ plain = self._plain().take(indices, axis=axis, **kwargs)
152
+ weights = self.weights.iloc[indices] if axis == 0 else self.weights
153
+ return self._weighted_result(plain, weights)
154
+
155
+ def _slice(self, slobj, axis=0):
156
+ plain = self._plain()._slice(slobj, axis=axis)
157
+ weights = self.weights.iloc[slobj] if axis == 0 else self.weights
158
+ return self._weighted_result(plain, weights)
159
+
160
+ def _reindex_with_indexers(
161
+ self, reindexers, fill_value=None, copy=False, allow_dups=False
162
+ ):
163
+ plain = self._plain()._reindex_with_indexers(
164
+ reindexers, fill_value=fill_value, allow_dups=allow_dups
165
+ )
166
+ if copy:
167
+ plain = plain.copy()
168
+ indexer = reindexers.get(0, (None, None))[1]
169
+ if indexer is not None:
170
+ if (np.asarray(indexer) < 0).any():
171
+ raise ValueError(
172
+ "Cannot invent weights for new rows introduced by reindex."
173
+ )
174
+ weights = self.weights.iloc[indexer]
175
+ else:
176
+ weights = self.weights
177
+ return self._weighted_result(plain, weights)
178
+
179
+ def drop(
180
+ self,
181
+ labels=None,
182
+ axis=0,
183
+ index=None,
184
+ columns=None,
185
+ level=None,
186
+ inplace=False,
187
+ errors="raise",
188
+ ):
189
+ plain = self._plain().drop(
190
+ labels=labels,
191
+ axis=axis,
192
+ index=index,
193
+ columns=columns,
194
+ level=level,
195
+ errors=errors,
196
+ )
197
+ positions = pd.Series(np.arange(len(self)), index=self.index)
198
+ row_labels = (
199
+ index
200
+ if index is not None
201
+ else labels
202
+ if self._get_axis_number(axis) == 0
203
+ else None
204
+ )
205
+ if row_labels is not None:
206
+ positions = positions.drop(row_labels, level=level, errors=errors)
207
+ return self._finish_row_operation(plain, np.asarray(positions), inplace)
208
+
209
+ def sample(self, *args, **kwargs):
210
+ """Sample weights with the selected rows, including replacements."""
211
+ result = super().sample(*args, **kwargs)
212
+ result.weights = weight_series(result.weights, result.index)
213
+ return result
214
+
215
+ def sort_values(self, *args, **kwargs):
216
+ """Sort rows and weights together even when labels are duplicated."""
217
+ inplace = kwargs.pop("inplace", False)
218
+ axis = self._get_axis_number(kwargs.get("axis", 0))
219
+ if self.ndim == 2:
220
+ plain = self._plain()
221
+ if axis == 1:
222
+ plain = plain.sort_values(*args, **kwargs)
223
+ return self._finish_row_operation(plain, np.arange(len(self)), inplace)
224
+ marker = object()
225
+ plain = plain.copy(deep=False)
226
+ plain[marker] = np.arange(len(self))
227
+ plain = plain.sort_values(*args, **kwargs)
228
+ positions = np.asarray(plain.pop(marker), dtype=int)
229
+ else:
230
+ values = self._plain()
231
+ positions = pd.Series(
232
+ np.arange(len(self)), index=self.index, name=self.name
233
+ )
234
+ key = kwargs.pop("key", None)
235
+ positions = positions.sort_values(
236
+ *args,
237
+ key=lambda unused: values if key is None else key(values),
238
+ **kwargs,
239
+ )
240
+ plain = values.iloc[np.asarray(positions)]
241
+ plain.index = positions.index
242
+ positions = np.asarray(positions, dtype=int)
243
+ return self._finish_row_operation(plain, positions, inplace)
244
+
245
+ def sort_index(self, *args, **kwargs):
246
+ """Sort the index while preserving positional weight provenance."""
247
+ inplace = kwargs.pop("inplace", False)
248
+ axis = self._get_axis_number(kwargs.get("axis", 0))
249
+ if axis == 1:
250
+ plain = self._plain().sort_index(*args, **kwargs)
251
+ return self._finish_row_operation(plain, np.arange(len(self)), inplace)
252
+ positions = pd.Series(np.arange(len(self)), index=self.index).sort_index(
253
+ *args, **kwargs
254
+ )
255
+ plain = self._plain().iloc[np.asarray(positions)]
256
+ plain.index = positions.index
257
+ return self._finish_row_operation(plain, np.asarray(positions), inplace)
@@ -8,93 +8,17 @@ import numpy as np
8
8
  import pandas as pd
9
9
 
10
10
  from microdf.microseries import MicroSeries, MicroSeriesGroupBy
11
+ from microdf._weights import (
12
+ WeightPropagationMixin,
13
+ aligned_weights,
14
+ finalize_weights,
15
+ weight_series,
16
+ )
11
17
 
12
18
  logger = logging.getLogger(__name__)
13
19
 
14
20
 
15
- class _MicroLocIndexer:
16
- """Custom loc indexer that returns MicroDataFrame with proper weights."""
17
-
18
- def __init__(self, mdf: "MicroDataFrame"):
19
- self._mdf = mdf
20
- # Get the parent's loc indexer
21
- self._parent_loc = pd.DataFrame.loc.fget(mdf)
22
-
23
- def __getitem__(self, key):
24
- # Use the parent DataFrame's loc indexer
25
- result = self._parent_loc[key]
26
-
27
- if isinstance(result, pd.DataFrame):
28
- # Get the filtered weights based on the result's index
29
- new_weights = self._mdf.weights.reindex(result.index)
30
- return MicroDataFrame(result, weights=new_weights)
31
- elif isinstance(result, pd.Series):
32
- # Single row or column selected
33
- if result.name in self._mdf.columns:
34
- # Column was selected - return MicroSeries with all weights
35
- return MicroSeries(result, weights=self._mdf.weights)
36
- else:
37
- # Row was selected - return as-is (scalar values for each col)
38
- return result
39
- else:
40
- # Scalar value
41
- return result
42
-
43
- def __setitem__(self, key, value):
44
- self._parent_loc[key] = value
45
- self._mdf._link_all_weights()
46
-
47
- def __getattr__(self, name):
48
- """Delegate unknown attributes to the parent loc indexer."""
49
- return getattr(self._parent_loc, name)
50
-
51
-
52
- class _MicroILocIndexer:
53
- """Custom iloc indexer that returns MicroDataFrame with proper weights."""
54
-
55
- def __init__(self, mdf: "MicroDataFrame"):
56
- self._mdf = mdf
57
- # Get the parent's iloc indexer
58
- self._parent_iloc = pd.DataFrame.iloc.fget(mdf)
59
-
60
- def __getitem__(self, key):
61
- # Use the parent DataFrame's iloc indexer
62
- result = self._parent_iloc[key]
63
-
64
- if isinstance(result, pd.DataFrame):
65
- # Get the filtered weights based on the result's index
66
- new_weights = self._mdf.weights.iloc[
67
- self._mdf.index.get_indexer(result.index)
68
- ]
69
- new_weights = pd.Series(new_weights.values, index=result.index)
70
- return MicroDataFrame(result, weights=new_weights)
71
- elif isinstance(result, pd.Series):
72
- # Single row or column selected
73
- if isinstance(key, tuple) and len(key) == 2:
74
- # df.iloc[:, col_idx] - column selection
75
- row_key = key[0]
76
- if isinstance(row_key, slice) and row_key == slice(None):
77
- # All rows selected for a column
78
- return MicroSeries(result, weights=self._mdf.weights)
79
- # Check if this is a column (result index matches mdf index)
80
- if result.index.equals(self._mdf.index):
81
- return MicroSeries(result, weights=self._mdf.weights)
82
- # Row selection - return as-is
83
- return result
84
- else:
85
- # Scalar value
86
- return result
87
-
88
- def __setitem__(self, key, value):
89
- self._parent_iloc[key] = value
90
- self._mdf._link_all_weights()
91
-
92
- def __getattr__(self, name):
93
- """Delegate unknown attributes to the parent iloc indexer."""
94
- return getattr(self._parent_iloc, name)
95
-
96
-
97
- class MicroDataFrame(pd.DataFrame):
21
+ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
98
22
  # Declare weight state as pandas metadata. pandas includes
99
23
  # _metadata attributes in the pickle state, so weights now survive
100
24
  # pickling, to_pickle/read_pickle and copy.deepcopy instead of
@@ -112,20 +36,54 @@ class MicroDataFrame(pd.DataFrame):
112
36
  :type weights: np.array
113
37
  """
114
38
  super().__init__(*args, **kwargs)
115
- self.weights = None
39
+ # pandas normalizes mixed-dimensional concat inputs through this
40
+ # constructor, either as a Series or a one-column mapping. Preserve
41
+ # that Series' row weights before concat loses the original input.
42
+ weight_source = args[0] if args else kwargs.get("data")
43
+ if isinstance(weight_source, dict) and len(weight_source) == 1:
44
+ weight_source = next(iter(weight_source.values()))
45
+ if weights is None and isinstance(weight_source, MicroSeries):
46
+ weights = aligned_weights(weight_source, self.index)
47
+ self.weights = weight_series(np.ones(len(self)), self.index)
48
+ self.weights_col = None
116
49
  self.set_weights(weights)
117
50
  self._link_all_weights()
118
51
  self.override_df_functions()
119
52
 
120
- def __finalize__(self, other, method=None, **kwargs) -> "MicroDataFrame":
121
- """Retain copied weights when pandas finalizes a renamed result."""
122
- copied_weights = getattr(self, "weights", None) if method == "rename" else None
53
+ @property
54
+ def _constructor(self):
55
+ return MicroDataFrame
56
+
57
+ # A row or a column-summary Series has no per-observation weights.
58
+ # _ixs wraps column selections using their unambiguous row provenance.
59
+ _constructor_sliced = pd.Series
60
+
61
+ def _ixs(self, i, axis=0):
62
+ result = pd.DataFrame(self, copy=False)._ixs(i, axis=axis)
63
+ if axis == 1:
64
+ return MicroSeries(result, weights=self.weights)
65
+ return result
66
+
67
+ def _get_item_cache(self, item):
68
+ # Weight arrays are independently mutable; cached column wrappers would
69
+ # retain stale copies after an in-place edit to frame.weights.
70
+ return self._ixs(self.columns.get_loc(item), axis=1)
71
+
72
+ def __finalize__(self, other, method=None, **kwargs):
73
+ previous = self.__dict__.get("weights")
123
74
  super().__finalize__(other, method=method, **kwargs)
124
- if copied_weights is not None:
125
- # rename already called copy(); metadata propagation must not
126
- # replace those weights with the source's mutable Series.
127
- self.weights = copied_weights
128
- return self
75
+ return finalize_weights(self, other, method, previous)
76
+
77
+ @wraps(pd.DataFrame.cov)
78
+ def cov(self, *args, **kwargs) -> pd.DataFrame:
79
+ # Column summaries have no observation weights, even if labels match.
80
+ result = pd.DataFrame(self, copy=False).cov(*args, **kwargs)
81
+ return result.__finalize__(self, method="cov")
82
+
83
+ @wraps(pd.DataFrame.corr)
84
+ def corr(self, *args, **kwargs) -> pd.DataFrame:
85
+ result = pd.DataFrame(self, copy=False).corr(*args, **kwargs)
86
+ return result.__finalize__(self, method="corr")
129
87
 
130
88
  def __setstate__(self, state) -> None:
131
89
  """Restore a pickled MicroDataFrame.
@@ -141,23 +99,6 @@ class MicroDataFrame(pd.DataFrame):
141
99
  self._link_all_weights()
142
100
  self.override_df_functions()
143
101
 
144
- @property
145
- def loc(self) -> _MicroLocIndexer:
146
- """Label-based indexer that preserves MicroDataFrame type and weights.
147
-
148
- :return: Custom loc indexer for MicroDataFrame
149
- """
150
- return _MicroLocIndexer(self)
151
-
152
- @property
153
- def iloc(self) -> _MicroILocIndexer:
154
- """Integer-based indexer that preserves MicroDataFrame type and
155
- weights.
156
-
157
- :return: Custom iloc indexer for MicroDataFrame
158
- """
159
- return _MicroILocIndexer(self)
160
-
161
102
  def override_df_functions(self) -> None:
162
103
  """Override DataFrame functions to work with weighted operations."""
163
104
  for name in MicroSeries.FUNCTIONS:
@@ -378,7 +319,7 @@ class MicroDataFrame(pd.DataFrame):
378
319
  pass
379
320
 
380
321
  def _link_all_weights(self) -> None:
381
- if self.weights is None:
322
+ if self.weights is None or len(self.weights) == 0:
382
323
  if len(self) > 0:
383
324
  self.set_weights(np.ones((len(self))))
384
325
  # In pandas 3.0+, columns are wrapped as MicroSeries on access via
@@ -472,33 +413,19 @@ class MicroDataFrame(pd.DataFrame):
472
413
  # that treats it as a Series (equals(), reindex() in __getitem__).
473
414
  self.set_weights(np.ones(len(self)))
474
415
 
475
- def __getitem__(
476
- self, key: Union[str, List]
477
- ) -> Union[MicroSeries, "MicroDataFrame"]:
478
- # Let pandas handle the initial slicing
479
- result = super().__getitem__(key)
480
-
481
- # If the result is a DataFrame, re-synchronize the weights
482
- if isinstance(result, pd.DataFrame):
483
- new_weights = self.weights.reindex(result.index)
484
- return MicroDataFrame(result, weights=new_weights)
485
-
486
- # If the result is a Series (single column), wrap as MicroSeries
487
- if isinstance(result, pd.Series):
488
- return MicroSeries(result, weights=self.weights)
489
-
490
- # Otherwise, the result is a scalar, so just return it
491
- return result
416
+ def __getitem__(self, key):
417
+ return super().__getitem__(key)
492
418
 
493
419
  def catch_series_relapse(self) -> None:
494
420
  # In pandas 3.0+, we don't need to track series class changes since
495
421
  # __getitem__ always wraps columns as MicroSeries on access.
496
422
  pass
497
423
 
498
- def __setattr__(self, key, value) -> None:
424
+ def __setattr__(self, key, value):
425
+ weights = self.__dict__.get("weights") if key == "index" else None
499
426
  super().__setattr__(key, value)
500
- # No need to call catch_series_relapse in pandas 3.0+ since we wrap
501
- # on access rather than store MicroSeries internally.
427
+ if weights is not None and len(weights) == len(self.index):
428
+ self.weights = weight_series(weights, self.index)
502
429
 
503
430
  def reset_index(
504
431
  self,
@@ -567,13 +494,7 @@ class MicroDataFrame(pd.DataFrame):
567
494
  return out
568
495
 
569
496
  def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame":
570
- res = super().copy(deep)
571
- # super().copy() corrupts self's column types to plain Series.
572
- # Restore them in O(N) instead of O(N²) by calling
573
- # _link_all_weights once rather than per-column __setitem__.
574
- self._link_all_weights()
575
- res = MicroDataFrame(res, weights=self.weights.copy(deep))
576
- return res
497
+ return super().copy(deep)
577
498
 
578
499
  def drop(
579
500
  self,
@@ -605,55 +526,15 @@ class MicroDataFrame(pd.DataFrame):
605
526
  dropped.
606
527
  :return: MicroDataFrame or None if inplace=True.
607
528
  """
608
- row_drop = axis in (0, "index") or index is not None
609
- if inplace:
610
- # Snapshot the pre-drop weights keyed by the pre-drop index so
611
- # we can reindex to the surviving rows after the drop.
612
- pre_drop_weights = pd.Series(self.weights.values, index=self.index.copy())
613
- # Perform in-place drop on the parent DataFrame
614
- super().drop(
615
- labels=labels,
616
- axis=axis,
617
- index=index,
618
- columns=columns,
619
- level=level,
620
- inplace=True,
621
- errors=errors,
622
- )
623
- if row_drop:
624
- surviving = pre_drop_weights.reindex(self.index)
625
- self.weights = pd.Series(
626
- surviving.values, index=self.index, dtype=float
627
- )
628
- else:
629
- self.weights = pd.Series(
630
- pre_drop_weights.values, index=self.index, dtype=float
631
- )
632
- self._link_all_weights()
633
- return None
634
- else:
635
- res = super().drop(
636
- labels=labels,
637
- axis=axis,
638
- index=index,
639
- columns=columns,
640
- level=level,
641
- inplace=False,
642
- errors=errors,
643
- )
644
- if row_drop:
645
- # Row drop: keep only the weights for surviving rows,
646
- # in the order of the resulting DataFrame.
647
- pre_drop_weights = pd.Series(self.weights.values, index=self.index)
648
- new_weights = pre_drop_weights.reindex(res.index).values
649
- else:
650
- new_weights = self.weights.values
651
- out = MicroDataFrame(res, weights=new_weights)
652
- # Guard against the set_weights path building weights with a
653
- # default RangeIndex, which would misalign against res.index
654
- # and silently zero weighted aggregations.
655
- out.weights = pd.Series(new_weights, index=out.index, dtype=float)
656
- return out
529
+ return super().drop(
530
+ labels=labels,
531
+ axis=axis,
532
+ index=index,
533
+ columns=columns,
534
+ level=level,
535
+ inplace=inplace,
536
+ errors=errors,
537
+ )
657
538
 
658
539
  def merge(
659
540
  self,
@@ -6,6 +6,8 @@ from typing import Callable, List, Optional, Union
6
6
  import numpy as np
7
7
  import pandas as pd
8
8
 
9
+ from microdf._weights import WeightPropagationMixin, finalize_weights, weight_series
10
+
9
11
  logger = logging.getLogger(__name__)
10
12
 
11
13
 
@@ -84,7 +86,7 @@ def _weighted_top_share(
84
86
  return top_sum / total_sum
85
87
 
86
88
 
87
- class MicroSeries(pd.Series):
89
+ class MicroSeries(WeightPropagationMixin, pd.Series):
88
90
  # Declare ``weights`` as pandas metadata. pandas includes
89
91
  # _metadata attributes in the pickle state, so weights now survive
90
92
  # pickling, to_pickle/read_pickle and copy.deepcopy instead of
@@ -103,15 +105,26 @@ class MicroSeries(pd.Series):
103
105
  super().__init__(*args, **kwargs)
104
106
  self.set_weights(weights)
105
107
 
106
- def __finalize__(self, other, method=None, **kwargs) -> "MicroSeries":
107
- """Retain copied weights when pandas finalizes a renamed result."""
108
- copied_weights = getattr(self, "weights", None) if method == "rename" else None
108
+ @property
109
+ def _constructor(self):
110
+ return MicroSeries
111
+
112
+ @property
113
+ def _constructor_expanddim(self):
114
+ from microdf.microdataframe import MicroDataFrame
115
+
116
+ return MicroDataFrame
117
+
118
+ def __finalize__(self, other, method=None, **kwargs):
119
+ previous = self.__dict__.get("weights")
109
120
  super().__finalize__(other, method=method, **kwargs)
110
- if copied_weights is not None:
111
- # rename already called copy(); metadata propagation must not
112
- # replace those weights with the source's mutable Series.
113
- self.weights = copied_weights
114
- return self
121
+ return finalize_weights(self, other, method, previous)
122
+
123
+ def __setattr__(self, name, value):
124
+ weights = self.__dict__.get("weights") if name == "index" else None
125
+ super().__setattr__(name, value)
126
+ if weights is not None and len(weights) == len(self.index):
127
+ self.weights = weight_series(weights, self.index)
115
128
 
116
129
  @property
117
130
  def _values(self):
@@ -178,12 +191,7 @@ class MicroSeries(pd.Series):
178
191
  :type weights: np.array.
179
192
  """
180
193
  if weights is None:
181
- if len(self) > 0:
182
- self.weights = pd.Series(
183
- np.ones_like(self._values),
184
- index=self.index,
185
- dtype=float,
186
- )
194
+ self.weights = weight_series(np.ones(len(self)), self.index)
187
195
  else:
188
196
  if len(weights) != len(self):
189
197
  raise ValueError(
@@ -201,7 +209,7 @@ class MicroSeries(pd.Series):
201
209
  # its index first so we position-align rather than label-align.
202
210
  if isinstance(weights, pd.Series):
203
211
  weights = weights.values
204
- self.weights = pd.Series(np.asarray(weights), index=self.index, dtype=float)
212
+ self.weights = weight_series(weights, self.index)
205
213
 
206
214
  def nullify_weights(self) -> None:
207
215
  """Set all weights to 1, effectively making the Series unweighted.
@@ -824,16 +832,14 @@ class MicroSeries(pd.Series):
824
832
  )
825
833
 
826
834
  def groupby(self, *args, **kwargs) -> "MicroSeriesGroupBy":
827
- gb = super().groupby(*args, **kwargs)
835
+ gb = pd.Series(self, copy=False).groupby(*args, **kwargs)
828
836
  gb.__class__ = MicroSeriesGroupBy
829
837
  gb._init()
830
838
  gb.weights = pd.Series(self.weights).groupby(*args, **kwargs)
831
839
  return gb
832
840
 
833
841
  def copy(self, deep: Optional[bool] = True):
834
- res = super().copy(deep)
835
- res = MicroSeries(res, weights=self.weights.copy(deep))
836
- return res
842
+ return super().copy(deep)
837
843
 
838
844
  def clip(
839
845
  self,
@@ -865,15 +871,22 @@ class MicroSeries(pd.Series):
865
871
  equal_weights = self.weights.equals(other.weights)
866
872
  return equal_values and equal_weights
867
873
 
868
- def __getitem__(
869
- self, key: Union[str, int, slice, List, np.ndarray]
870
- ) -> Union["MicroSeries", pd.Series]:
871
- result = super().__getitem__(key)
874
+ def __getitem__(self, key):
875
+ if callable(key):
876
+ key = key(self)
877
+ result = pd.Series(self, copy=False).__getitem__(key)
872
878
  if isinstance(result, pd.Series):
873
- weights = self.weights.__getitem__(key)
874
- return MicroSeries(result, weights=weights)
879
+ positions = pd.Series(np.arange(len(self)), index=self.index).__getitem__(
880
+ key
881
+ )
882
+ return MicroSeries(result, weights=self.weights.iloc[np.asarray(positions)])
875
883
  return result
876
884
 
885
+ def repeat(self, repeats, axis=None):
886
+ # Use pandas to validate the repeat counts and axis argument.
887
+ positions = pd.Series(np.arange(len(self))).repeat(repeats, axis=axis)
888
+ return self.take(np.asarray(positions))
889
+
877
890
  def __getattr__(self, name: str) -> "MicroSeries":
878
891
  return MicroSeries(super().__getattr__(name), weights=self.weights)
879
892
 
@@ -0,0 +1,478 @@
1
+ """Weighted pandas operations preserve row identity, not just row labels."""
2
+
3
+ import copy
4
+ import pickle
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ import pytest
9
+
10
+ import microdf as mdf
11
+
12
+
13
+ def make_data(kind):
14
+ values = pd.Series([30.0, np.nan, 10.0, 20.0], index=[7, 7, 3, 7], name="x")
15
+ weights = [2.0, 5.0, 11.0, 17.0]
16
+ if kind == "series":
17
+ return mdf.MicroSeries(values, weights=weights)
18
+ return mdf.MicroDataFrame(values.to_frame(), weights=weights)
19
+
20
+
21
+ def assert_rows(result, original, positions, values=None):
22
+ assert type(result) is type(original)
23
+ expected = original.weights.to_numpy()[positions]
24
+ np.testing.assert_array_equal(result.weights.to_numpy(), expected)
25
+ assert result.weights.index.equals(result.index)
26
+ assert result.weights is not original.weights
27
+ data = np.asarray(result if isinstance(result, mdf.MicroSeries) else result["x"])
28
+ if values is not None:
29
+ np.testing.assert_allclose(data, values, equal_nan=True)
30
+ total = np.nansum(data * expected)
31
+ assert (
32
+ result.sum() if isinstance(result, mdf.MicroSeries) else result.sum()["x"]
33
+ ) == total
34
+
35
+
36
+ @pytest.mark.parametrize("kind", ["series", "frame"])
37
+ @pytest.mark.parametrize("ascending", [True, False])
38
+ @pytest.mark.parametrize("ignore_index", [True, False])
39
+ def test_sort_values_preserves_positional_weights(kind, ascending, ignore_index):
40
+ original = make_data(kind)
41
+ args = () if kind == "series" else ("x",)
42
+ result = original.sort_values(*args, ascending=ascending, ignore_index=ignore_index)
43
+ positions = [2, 3, 0, 1] if ascending else [0, 3, 2, 1]
44
+ assert_rows(result, original, positions)
45
+ if ignore_index:
46
+ assert result.index.equals(pd.RangeIndex(4))
47
+ assert_rows(original, original.copy(), np.arange(4))
48
+
49
+
50
+ @pytest.mark.parametrize("kind", ["series", "frame"])
51
+ def test_fillna_and_row_transforms_preserve_weights(kind):
52
+ original = make_data(kind)
53
+ filled = original.fillna(4)
54
+ assert_rows(filled, original, np.arange(4), [30, 4, 10, 20])
55
+ assert_rows(filled.replace(30, 40), filled, np.arange(4), [40, 4, 10, 20])
56
+ assert_rows(filled.astype(float), filled, np.arange(4))
57
+ assert_rows(filled + 1, filled, np.arange(4), [31, 5, 11, 21])
58
+ filled.weights.iloc[0] = 100
59
+ assert original.weights.iloc[0] == 2
60
+
61
+
62
+ @pytest.mark.parametrize("kind", ["series", "frame"])
63
+ @pytest.mark.parametrize("ignore_index", [True, False])
64
+ def test_sample_with_replacement_and_duplicate_labels(kind, ignore_index):
65
+ original = make_data(kind)
66
+ positions = (
67
+ pd.Series(np.arange(4)).sample(n=12, replace=True, random_state=17).to_numpy()
68
+ )
69
+ result = original.sample(
70
+ n=12, replace=True, random_state=17, ignore_index=ignore_index
71
+ )
72
+ assert_rows(result, original, positions)
73
+ assert len(set(positions)) < len(positions)
74
+
75
+
76
+ @pytest.mark.parametrize("kind", ["series", "frame"])
77
+ def test_duplicate_label_selections_and_sort_index(kind):
78
+ original = make_data(kind)
79
+ assert_rows(original.iloc[[3, 0, 3]], original, [3, 0, 3])
80
+ assert_rows(original.iloc[1:3], original, [1, 2])
81
+ assert_rows(original.loc[[3, 7]], original, [2, 0, 1, 3])
82
+ assert_rows(original.sort_index(kind="stable"), original, [2, 0, 1, 3])
83
+ if kind == "frame":
84
+ column = original.loc[[3, 7], "x"]
85
+ assert isinstance(column, mdf.MicroSeries)
86
+ np.testing.assert_array_equal(column.weights, [11, 2, 5, 17])
87
+ row = original.iloc[0]
88
+ assert type(row) is pd.Series
89
+
90
+
91
+ @pytest.mark.parametrize("kind", ["series", "frame"])
92
+ @pytest.mark.parametrize("options", [{}, {"ignore_index": True}, {"keys": ["a", "b"]}])
93
+ def test_concat_rows_preserves_duplicate_and_repeated_weights(kind, options):
94
+ original = make_data(kind)
95
+ pieces = [original.iloc[[3, 0]], original.iloc[[2, 3]]]
96
+ result = pd.concat(pieces, **options)
97
+ assert_rows(result, original, [3, 0, 2, 3])
98
+
99
+
100
+ def test_concat_columns_requires_consistent_row_weights():
101
+ first = mdf.MicroDataFrame({"x": [1, 2]}, index=[9, 3], weights=[4, 5])
102
+ second = mdf.MicroDataFrame({"y": [6, 7]}, index=[3, 9], weights=[5, 4])
103
+ result = pd.concat([first, second], axis=1)
104
+ assert isinstance(result, mdf.MicroDataFrame)
105
+ np.testing.assert_array_equal(result.weights, [4, 5])
106
+ assert result.sum().to_dict() == {"x": 14, "y": 58}
107
+ second.set_weights([50, 40])
108
+ with pytest.raises(ValueError, match="weights"):
109
+ pd.concat([first, second], axis=1)
110
+
111
+
112
+ @pytest.mark.parametrize("kind", ["series", "frame"])
113
+ def test_concat_with_unweighted_objects_is_explicit(kind):
114
+ weighted = make_data(kind)
115
+ plain = pd.Series([1, 2]) if kind == "series" else pd.DataFrame({"x": [1, 2]})
116
+ with pytest.raises(ValueError, match="weights"):
117
+ pd.concat([weighted, plain])
118
+
119
+
120
+ @pytest.mark.parametrize("kind", ["series", "frame"])
121
+ def test_reindex_new_rows_cannot_invent_weights(kind):
122
+ weighted = make_data(kind).iloc[:1]
123
+ with pytest.raises(ValueError, match="weights"):
124
+ weighted.reindex([7, 99])
125
+
126
+
127
+ @pytest.mark.parametrize("kind", ["series", "frame"])
128
+ def test_propagated_objects_keep_serialization_and_rename_isolation(kind):
129
+ original = make_data(kind).fillna(4)
130
+ renamed = (
131
+ original.rename("income")
132
+ if kind == "series"
133
+ else original.rename(columns={"x": "income"})
134
+ )
135
+ renamed.weights.iloc[0] = 99
136
+ assert original.weights.iloc[0] == 2
137
+ for result in [copy.deepcopy(original), pickle.loads(pickle.dumps(original))]:
138
+ assert_rows(result, original, np.arange(4))
139
+ if kind == "series":
140
+ assert original.name == "x"
141
+
142
+
143
+ @pytest.mark.parametrize("kind", ["series", "frame"])
144
+ def test_inplace_sort_and_fillna_preserve_weights(kind):
145
+ original = make_data(kind)
146
+ expected = original.copy()
147
+ args = () if kind == "series" else ("x",)
148
+ assert original.sort_values(*args, inplace=True, ascending=False) is None
149
+ assert_rows(original, expected, [0, 3, 2, 1])
150
+ original.fillna(4, inplace=True)
151
+ assert_rows(original, expected, [0, 3, 2, 1], [30, 20, 10, 4])
152
+
153
+
154
+ @pytest.mark.parametrize("kind", ["series", "frame"])
155
+ def test_boolean_selection_and_drop_with_duplicate_indexes(kind):
156
+ original = make_data(kind)
157
+ mask = np.array([True, False, True, False])
158
+ assert_rows(original[mask], original, [0, 2])
159
+ assert_rows(original.drop(index=7), original, [2])
160
+ assert_rows(original.loc[mask], original, [0, 2])
161
+ if kind == "series":
162
+ assert_rows(original.repeat(2), original, [0, 0, 1, 1, 2, 2, 3, 3])
163
+
164
+
165
+ def test_column_access_uses_current_weights_without_mutating_earlier_series():
166
+ frame = mdf.MicroDataFrame({"x": [1, 2]}, weights=[3, 4])
167
+ earlier = frame["x"]
168
+ assert earlier.sum() == 11
169
+ frame.weights.iloc[0] = 10
170
+ assert frame["x"].sum() == 18
171
+ assert frame.loc[:, "x"].sum() == 18
172
+ assert earlier.sum() == 11
173
+
174
+
175
+ def test_transpose_cannot_reinterpret_row_weights_as_column_weights():
176
+ frame = mdf.MicroDataFrame(
177
+ [[1, 2], [3, 4]], index=[0, 1], columns=[0, 1], weights=[5, 7]
178
+ )
179
+ with pytest.raises(ValueError, match="weights"):
180
+ frame.transpose()
181
+
182
+
183
+ @pytest.mark.parametrize("inplace", [False, True])
184
+ @pytest.mark.parametrize(
185
+ "index,level",
186
+ [
187
+ (pd.Index(["a", "a", "b"], name="row"), None),
188
+ (
189
+ pd.MultiIndex.from_tuples(
190
+ [("a", 1), ("a", 1), ("b", 2)], names=["group", "row"]
191
+ ),
192
+ None,
193
+ ),
194
+ (
195
+ pd.MultiIndex.from_tuples(
196
+ [("a", 1), ("a", 1), ("b", 2)], names=["group", "row"]
197
+ ),
198
+ "group",
199
+ ),
200
+ ],
201
+ )
202
+ def test_series_reset_index_drop_preserves_positional_weights(index, level, inplace):
203
+ original = mdf.MicroSeries(
204
+ [10, 20, 30], index=index, name="income", weights=[2, 3, 5]
205
+ )
206
+ source_weights = original.weights
207
+ expected = pd.Series(original).reset_index(level=level, drop=True, name="ignored")
208
+ result = original.reset_index(
209
+ level=level, drop=True, name="ignored", inplace=inplace
210
+ )
211
+ if inplace:
212
+ assert result is None
213
+ result = original
214
+ assert isinstance(result, mdf.MicroSeries)
215
+ pd.testing.assert_series_equal(pd.Series(result), expected)
216
+ pd.testing.assert_series_equal(
217
+ result.weights, pd.Series([2.0, 3.0, 5.0], index=expected.index)
218
+ )
219
+ assert result.sum() == 230
220
+ assert result.weights is not source_weights
221
+ result.weights.iloc[0] = 99
222
+ assert source_weights.iloc[0] == 2
223
+ source_weights.iloc[1] = 88
224
+ assert result.weights.iloc[1] == 3
225
+
226
+
227
+ def test_series_reset_index_preserves_dataframe_and_invalid_inplace_behavior():
228
+ original = mdf.MicroSeries(
229
+ [10, 20], index=pd.Index(["a", "b"], name="row"), name="income", weights=[2, 3]
230
+ )
231
+ result = original.reset_index(name="amount")
232
+ assert isinstance(result, mdf.MicroDataFrame)
233
+ pd.testing.assert_frame_equal(
234
+ pd.DataFrame(result), pd.Series(original).reset_index(name="amount")
235
+ )
236
+ np.testing.assert_array_equal(result.weights, [2, 3])
237
+ result.weights.iloc[0] = 99
238
+ assert original.weights.iloc[0] == 2
239
+ with pytest.raises(TypeError, match="inplace"):
240
+ original.reset_index(inplace=True)
241
+
242
+
243
+ @pytest.mark.parametrize(
244
+ "select,positions",
245
+ [
246
+ (lambda frame: frame.iloc[1:], [1, 2]),
247
+ (lambda frame: frame[["income"]], [0, 1, 2]),
248
+ (lambda frame: frame.iloc[:, :1], [0, 1, 2]),
249
+ (lambda frame: frame.loc[:, ["income"]], [0, 1, 2]),
250
+ (lambda frame: frame.iloc[[2, 0, 2]], [2, 0, 2]),
251
+ (lambda frame: frame.reindex(columns=["income"]), [0, 1, 2]),
252
+ ],
253
+ ids=["row-slice", "columns", "iloc-columns", "loc-columns", "repeated", "reindex"],
254
+ )
255
+ def test_selected_dataframe_weights_are_independently_mutable(select, positions):
256
+ source = mdf.MicroDataFrame(
257
+ {"income": [10.0, 20.0, 30.0], "other": [1, 2, 3]},
258
+ index=[7, 7, 3],
259
+ weights=[2, 3, 5],
260
+ )
261
+ selected = select(source)
262
+ expected_weights = np.array([2.0, 3.0, 5.0])[positions]
263
+ np.testing.assert_array_equal(selected.weights, expected_weights)
264
+ assert selected.weights.index.equals(selected.index)
265
+
266
+ selected.weights.iloc[0] = 100
267
+ expected_weights[0] = 100
268
+ np.testing.assert_array_equal(source.weights, [2, 3, 5])
269
+ assert source.income.sum() == 230 # 10 * 2 + 20 * 3 + 30 * 5.
270
+ assert selected.income.sum() == np.dot(
271
+ np.array([10.0, 20.0, 30.0])[positions], expected_weights
272
+ )
273
+
274
+ source.weights.iloc[-1] = 200
275
+ np.testing.assert_array_equal(selected.weights, expected_weights)
276
+ assert selected.income.sum() == np.dot(
277
+ np.array([10.0, 20.0, 30.0])[positions], expected_weights
278
+ )
279
+
280
+
281
+ @pytest.mark.parametrize("series_first", [False, True])
282
+ @pytest.mark.parametrize("options", [{}, {"ignore_index": True}, {"keys": ["a", "b"]}])
283
+ @pytest.mark.parametrize("series_name", ["income", None])
284
+ def test_concat_mixed_dimensions_preserves_series_weights(
285
+ series_first, options, series_name
286
+ ):
287
+ frame = mdf.MicroDataFrame({"income": [10.0, 20.0]}, index=[7, 7], weights=[2, 3])
288
+ series = mdf.MicroSeries(
289
+ [30.0, 40.0], index=[7, 3], name=series_name, weights=[5, 7]
290
+ )
291
+ parts = [series, frame] if series_first else [frame, series]
292
+ plain_parts = [
293
+ pd.Series(part) if part.ndim == 1 else pd.DataFrame(part) for part in parts
294
+ ]
295
+ expected = pd.concat(plain_parts, **options)
296
+ expected_weights = [5, 7, 2, 3] if series_first else [2, 3, 5, 7]
297
+
298
+ result = pd.concat(parts, **options)
299
+ assert isinstance(result, mdf.MicroDataFrame)
300
+ pd.testing.assert_frame_equal(pd.DataFrame(result), expected)
301
+ np.testing.assert_array_equal(result.weights, expected_weights)
302
+ assert result.weights.index.equals(result.index)
303
+ for column in expected:
304
+ assert result[column].sum() == np.nansum(
305
+ expected[column].to_numpy() * expected_weights
306
+ )
307
+
308
+ result.weights.iloc[0] = 100
309
+ np.testing.assert_array_equal(frame.weights, [2, 3])
310
+ np.testing.assert_array_equal(series.weights, [5, 7])
311
+ series.weights.iloc[-1] = 200
312
+ series_last_position = 1 if series_first else 3
313
+ assert result.weights.iloc[series_last_position] == 7
314
+
315
+
316
+ @pytest.mark.parametrize("series_first", [False, True])
317
+ @pytest.mark.parametrize("conflicting", [False, True])
318
+ def test_concat_mixed_dimensions_columns_aligns_or_rejects_weights(
319
+ series_first, conflicting
320
+ ):
321
+ frame = mdf.MicroDataFrame({"income": [10.0, 20.0]}, index=[7, 3], weights=[2, 5])
322
+ series = mdf.MicroSeries(
323
+ [30.0, 40.0],
324
+ index=[3, 7],
325
+ name="other",
326
+ weights=[50, 2] if conflicting else [5, 2],
327
+ )
328
+ parts = [series, frame] if series_first else [frame, series]
329
+ if conflicting:
330
+ with pytest.raises(ValueError, match="weights"):
331
+ pd.concat(parts, axis=1)
332
+ else:
333
+ result = pd.concat(parts, axis=1)
334
+ np.testing.assert_array_equal(
335
+ result.weights, [5, 2] if series_first else [2, 5]
336
+ )
337
+ assert result.income.sum() == 120 # 10 * 2 + 20 * 5.
338
+ assert result.other.sum() == 230 # 30 * 5 + 40 * 2.
339
+
340
+
341
+ @pytest.mark.parametrize("mapping", [False, True])
342
+ def test_series_to_dataframe_weights_follow_aligned_rows_and_explicit_override(mapping):
343
+ series = mdf.MicroSeries([10.0, 20.0], index=[7, 3], name="income", weights=[2, 5])
344
+ data = {"income": series} if mapping else series
345
+ result = mdf.MicroDataFrame(data, index=[3, 7])
346
+ np.testing.assert_array_equal(result.weights, [5, 2])
347
+ assert result.income.sum() == 120 # 20 * 5 + 10 * 2.
348
+ result.weights.iloc[0] = 100
349
+ np.testing.assert_array_equal(series.weights, [2, 5])
350
+
351
+ explicit = mdf.MicroDataFrame(data, index=[3, 7], weights=[11, 13])
352
+ np.testing.assert_array_equal(explicit.weights, [11, 13])
353
+ assert explicit.income.sum() == 350 # 20 * 11 + 10 * 13.
354
+ with pytest.raises(ValueError, match="weights"):
355
+ mdf.MicroDataFrame(data, index=[3, 99])
356
+
357
+
358
+ @pytest.mark.parametrize("method", ["cov", "corr"])
359
+ @pytest.mark.parametrize("coincident_labels", [False, True])
360
+ def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_labels):
361
+ if coincident_labels:
362
+ frame = mdf.MicroDataFrame(
363
+ {"x": [10.0, 20.0], "y": [4.0, 8.0]},
364
+ index=["x", "y"],
365
+ weights=[2, 3],
366
+ )
367
+ # Sample covariance divides centered cross-products by n - 1.
368
+ covariance = [[50.0, 20.0], [20.0, 8.0]]
369
+ correlation = [[1.0, 1.0], [1.0, 1.0]]
370
+ else:
371
+ frame = mdf.MicroDataFrame(
372
+ {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]},
373
+ weights=[2, 3, 5],
374
+ )
375
+ # Centered x = [-10, 0, 10], y = [-2, 2, 0]; n - 1 = 2.
376
+ covariance = [[100.0, 10.0], [10.0, 4.0]]
377
+ correlation = [[1.0, 0.5], [0.5, 1.0]]
378
+ expected = pd.DataFrame(
379
+ covariance if method == "cov" else correlation,
380
+ index=frame.columns,
381
+ columns=frame.columns,
382
+ )
383
+
384
+ result = getattr(frame, method)()
385
+
386
+ assert type(result) is pd.DataFrame
387
+ pd.testing.assert_frame_equal(result, expected)
388
+ # Chaining a sum must not apply observation weights to column summaries.
389
+ pd.testing.assert_series_equal(result.sum(), expected.sum())
390
+ assert not hasattr(result, "weights")
391
+
392
+
393
+ @pytest.mark.parametrize(
394
+ "method,args,kwargs",
395
+ [
396
+ ("cov", (), {}),
397
+ ("cov", (2, 0), {}),
398
+ ("cov", (), {"min_periods": 4, "ddof": 2}),
399
+ ("cov", (), {"min_periods": 2, "ddof": 0, "numeric_only": True}),
400
+ ("corr", (), {}),
401
+ ("corr", ("pearson", 2, True), {}),
402
+ ("corr", (), {"method": "spearman", "min_periods": 2}),
403
+ ("corr", (), {"min_periods": 4}),
404
+ ("corr", (), {"method": lambda x, y: np.dot(x, y), "min_periods": 2}),
405
+ ],
406
+ )
407
+ @pytest.mark.parametrize("missing", [False, True])
408
+ def test_dataframe_matrix_summaries_preserve_pandas_arguments(
409
+ method, args, kwargs, missing
410
+ ):
411
+ data = {
412
+ "x": [10.0, 20.0, 30.0, 40.0],
413
+ "y": [4.0, 8.0, np.nan if missing else 6.0, 9.0],
414
+ "flag": [True, False, True, True],
415
+ }
416
+ frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=[2, 3, 5, 7])
417
+ expected = getattr(pd.DataFrame(data, index=frame.index), method)(*args, **kwargs)
418
+
419
+ result = getattr(frame, method)(*args, **kwargs)
420
+
421
+ assert type(result) is pd.DataFrame
422
+ pd.testing.assert_frame_equal(result, expected)
423
+ pd.testing.assert_series_equal(result.sum(), expected.sum())
424
+
425
+
426
+ @pytest.mark.parametrize("method", ["cov", "corr"])
427
+ def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method):
428
+ data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0], "label": ["a", "b", "c"]}
429
+ frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
430
+ plain = pd.DataFrame(data)
431
+ expected = getattr(plain, method)(numeric_only=True)
432
+
433
+ result = getattr(frame, method)(numeric_only=True)
434
+
435
+ assert type(result) is pd.DataFrame
436
+ pd.testing.assert_frame_equal(result, expected)
437
+ for kwargs in [{}, {"numeric_only": False}]:
438
+ with pytest.raises((TypeError, ValueError)) as pandas_error:
439
+ getattr(plain, method)(**kwargs)
440
+ with pytest.raises(type(pandas_error.value)) as microdf_error:
441
+ getattr(frame, method)(**kwargs)
442
+ assert str(microdf_error.value) == str(pandas_error.value)
443
+
444
+
445
+ def test_dataframe_correlation_preserves_optional_kendall_support():
446
+ data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
447
+ frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
448
+ try:
449
+ expected = pd.DataFrame(data).corr(method="kendall")
450
+ except ImportError as pandas_error:
451
+ # Kendall requires scipy; delegation preserves pandas' dependency error.
452
+ with pytest.raises(type(pandas_error)) as microdf_error:
453
+ frame.corr(method="kendall")
454
+ assert str(microdf_error.value) == str(pandas_error)
455
+ else:
456
+ result = frame.corr(method="kendall")
457
+ assert type(result) is pd.DataFrame
458
+ pd.testing.assert_frame_equal(result, expected)
459
+
460
+
461
+ @pytest.mark.parametrize("method", ["cov", "corr"])
462
+ def test_dataframe_matrix_summaries_preserve_pandas_metadata(method):
463
+ data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
464
+ frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
465
+ plain = pd.DataFrame(data)
466
+ for source in [frame, plain]:
467
+ source.attrs = {"survey": {"year": 2026}}
468
+ source.flags.allows_duplicate_labels = False
469
+ source.columns.name = "measure"
470
+ expected = getattr(plain, method)()
471
+
472
+ result = getattr(frame, method)()
473
+
474
+ assert type(result) is pd.DataFrame
475
+ pd.testing.assert_frame_equal(result, expected)
476
+ assert result.attrs == expected.attrs
477
+ result.attrs["survey"]["year"] = 2025
478
+ assert frame.attrs["survey"]["year"] == 2026
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.4.0
3
+ Version: 1.4.1
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -2,6 +2,7 @@ LICENSE
2
2
  README.md
3
3
  pyproject.toml
4
4
  microdf/__init__.py
5
+ microdf/_weights.py
5
6
  microdf/microdataframe.py
6
7
  microdf/microseries.py
7
8
  microdf/tests/conftest.py
@@ -14,6 +15,7 @@ microdf/tests/test_quantile_missing_values.py
14
15
  microdf/tests/test_serialization.py
15
16
  microdf/tests/test_sum_axes.py
16
17
  microdf/tests/test_version_metadata.py
18
+ microdf/tests/test_weight_propagation.py
17
19
  microdf/tests/test_weighted_cov_corr.py
18
20
  microdf_python.egg-info/PKG-INFO
19
21
  microdf_python.egg-info/SOURCES.txt
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "microdf-python"
7
- version = "1.4.0"
7
+ version = "1.4.1"
8
8
  description = "Weighted pandas DataFrames and Series for survey microdata"
9
9
  readme = "README.md"
10
10
  authors = [
File without changes
File without changes
File without changes