microdf-python 1.4.0__py3-none-any.whl → 1.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- microdf/_weights.py +257 -0
- microdf/microdataframe.py +67 -186
- microdf/microseries.py +39 -26
- microdf/tests/test_weight_propagation.py +478 -0
- {microdf_python-1.4.0.dist-info → microdf_python-1.4.1.dist-info}/METADATA +1 -1
- {microdf_python-1.4.0.dist-info → microdf_python-1.4.1.dist-info}/RECORD +9 -7
- {microdf_python-1.4.0.dist-info → microdf_python-1.4.1.dist-info}/WHEEL +0 -0
- {microdf_python-1.4.0.dist-info → microdf_python-1.4.1.dist-info}/licenses/LICENSE +0 -0
- {microdf_python-1.4.0.dist-info → microdf_python-1.4.1.dist-info}/top_level.txt +0 -0
microdf/_weights.py
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""Pandas subclass hooks for copying weights and retaining row positions."""
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def weight_series(values, index):
|
|
8
|
+
"""Own an independent float array aligned positionally to an index."""
|
|
9
|
+
return pd.Series(np.array(values, dtype=float, copy=True), index=index)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def aligned_weights(source, index):
|
|
13
|
+
"""Align only when row identity can be established without guessing."""
|
|
14
|
+
if source.index.equals(index):
|
|
15
|
+
return weight_series(source.weights, index)
|
|
16
|
+
if source.index.is_unique:
|
|
17
|
+
positions = source.index.get_indexer(index)
|
|
18
|
+
if (positions >= 0).all():
|
|
19
|
+
return weight_series(source.weights.iloc[positions], index)
|
|
20
|
+
raise ValueError(
|
|
21
|
+
"Cannot propagate weights unambiguously to these rows. "
|
|
22
|
+
"Use positional selection, or explicitly supply weights for the result."
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def finalize_weights(result, source, method, previous_weights):
|
|
27
|
+
"""Propagate row weights after pandas has finalized its own metadata."""
|
|
28
|
+
if method == "transpose" and result.ndim == 2:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
"Cannot transpose row weights onto columns. "
|
|
31
|
+
"Convert to a pandas DataFrame first and supply explicit result weights."
|
|
32
|
+
)
|
|
33
|
+
if method == "concat":
|
|
34
|
+
objects = source.objs
|
|
35
|
+
if not all(isinstance(obj, WeightPropagationMixin) for obj in objects):
|
|
36
|
+
raise ValueError(
|
|
37
|
+
"Cannot concatenate weighted and unweighted objects: "
|
|
38
|
+
"provide explicit weights for every input."
|
|
39
|
+
)
|
|
40
|
+
weights = concat_weights(result, objects, getattr(source, "axis", None))
|
|
41
|
+
result.weights = weights
|
|
42
|
+
if result.ndim == 2:
|
|
43
|
+
names = [obj.__dict__.get("weights_col") for obj in objects]
|
|
44
|
+
result.weights_col = (
|
|
45
|
+
names[0] if all(name == names[0] for name in names) else None
|
|
46
|
+
)
|
|
47
|
+
elif isinstance(source, WeightPropagationMixin):
|
|
48
|
+
if method == "reset_index" and source.ndim == result.ndim == 1:
|
|
49
|
+
# Series.reset_index(drop=True) changes labels, never row order.
|
|
50
|
+
# pandas 2 constructs a result; pandas 3 relabels a shallow copy.
|
|
51
|
+
result.weights = weight_series(source.weights, result.index)
|
|
52
|
+
elif method == "rename" and previous_weights is not None:
|
|
53
|
+
result.weights = weight_series(previous_weights, result.index)
|
|
54
|
+
else:
|
|
55
|
+
result.weights = aligned_weights(source, result.index)
|
|
56
|
+
return result
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def concat_weights(result, objects, axis):
|
|
60
|
+
"""Recover concat row identity without relying on pandas 2-only state."""
|
|
61
|
+
if objects[0].ndim == 1 and result.ndim == 2:
|
|
62
|
+
axis = 1
|
|
63
|
+
row_weights = None
|
|
64
|
+
if axis != 1 and len(result) == sum(len(obj) for obj in objects):
|
|
65
|
+
combined_index = objects[0].index.append([obj.index for obj in objects[1:]])
|
|
66
|
+
same_rows = result.index.equals(combined_index)
|
|
67
|
+
reset_rows = result.index.equals(pd.RangeIndex(len(result)))
|
|
68
|
+
keyed_rows = False
|
|
69
|
+
if (
|
|
70
|
+
isinstance(result.index, pd.MultiIndex)
|
|
71
|
+
and result.index.nlevels > combined_index.nlevels
|
|
72
|
+
):
|
|
73
|
+
trailing = result.index.droplevel(
|
|
74
|
+
list(range(result.index.nlevels - combined_index.nlevels))
|
|
75
|
+
)
|
|
76
|
+
keyed_rows = trailing.equals(combined_index)
|
|
77
|
+
if same_rows or reset_rows or keyed_rows:
|
|
78
|
+
row_weights = np.concatenate([np.asarray(obj.weights) for obj in objects])
|
|
79
|
+
column_weights = None
|
|
80
|
+
if (
|
|
81
|
+
axis != 0
|
|
82
|
+
and result.ndim == 2
|
|
83
|
+
and len(result.columns)
|
|
84
|
+
== sum(obj.shape[1] if obj.ndim == 2 else 1 for obj in objects)
|
|
85
|
+
):
|
|
86
|
+
values = np.empty(len(result))
|
|
87
|
+
known = np.zeros(len(result), dtype=bool)
|
|
88
|
+
conflict = False
|
|
89
|
+
for obj in objects:
|
|
90
|
+
if obj.index.equals(result.index):
|
|
91
|
+
positions = np.arange(len(result))
|
|
92
|
+
elif obj.index.is_unique:
|
|
93
|
+
positions = obj.index.get_indexer(result.index)
|
|
94
|
+
else:
|
|
95
|
+
conflict = True
|
|
96
|
+
break
|
|
97
|
+
present = positions >= 0
|
|
98
|
+
incoming = np.asarray(obj.weights)[positions[present]]
|
|
99
|
+
overlap = known[present]
|
|
100
|
+
if not np.array_equal(
|
|
101
|
+
values[present][overlap], incoming[overlap], equal_nan=True
|
|
102
|
+
):
|
|
103
|
+
conflict = True
|
|
104
|
+
break
|
|
105
|
+
values[present] = incoming
|
|
106
|
+
known[present] = True
|
|
107
|
+
if not conflict and known.all():
|
|
108
|
+
column_weights = values
|
|
109
|
+
if row_weights is not None:
|
|
110
|
+
if column_weights is not None and not np.array_equal(
|
|
111
|
+
row_weights, column_weights, equal_nan=True
|
|
112
|
+
):
|
|
113
|
+
raise ValueError(
|
|
114
|
+
"Ambiguous weights: pandas did not expose the concat axis."
|
|
115
|
+
)
|
|
116
|
+
return weight_series(row_weights, result.index)
|
|
117
|
+
if column_weights is not None:
|
|
118
|
+
return weight_series(column_weights, result.index)
|
|
119
|
+
raise ValueError(
|
|
120
|
+
"Cannot propagate concat weights: conflicting or ambiguous row weights."
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class WeightPropagationMixin:
|
|
125
|
+
"""Use positional provenance for pandas operations that select rows."""
|
|
126
|
+
|
|
127
|
+
def _plain(self):
|
|
128
|
+
return (
|
|
129
|
+
pd.Series(self, copy=False)
|
|
130
|
+
if self.ndim == 1
|
|
131
|
+
else pd.DataFrame(self, copy=False)
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
def _weighted_result(self, plain, weights):
|
|
135
|
+
result = type(self)(plain, weights=weight_series(weights, plain.index))
|
|
136
|
+
if self.ndim == 2:
|
|
137
|
+
result.weights_col = self.__dict__.get("weights_col")
|
|
138
|
+
return result
|
|
139
|
+
|
|
140
|
+
def _finish_row_operation(self, plain, positions, inplace=False):
|
|
141
|
+
result = self._weighted_result(plain, self.weights.iloc[positions])
|
|
142
|
+
if inplace:
|
|
143
|
+
self._update_inplace(result)
|
|
144
|
+
self.weights = result.weights
|
|
145
|
+
return None
|
|
146
|
+
return result
|
|
147
|
+
|
|
148
|
+
def take(self, indices, axis=0, **kwargs):
|
|
149
|
+
"""Take values and weights using the same positional indexer."""
|
|
150
|
+
axis = self._get_axis_number(axis)
|
|
151
|
+
plain = self._plain().take(indices, axis=axis, **kwargs)
|
|
152
|
+
weights = self.weights.iloc[indices] if axis == 0 else self.weights
|
|
153
|
+
return self._weighted_result(plain, weights)
|
|
154
|
+
|
|
155
|
+
def _slice(self, slobj, axis=0):
|
|
156
|
+
plain = self._plain()._slice(slobj, axis=axis)
|
|
157
|
+
weights = self.weights.iloc[slobj] if axis == 0 else self.weights
|
|
158
|
+
return self._weighted_result(plain, weights)
|
|
159
|
+
|
|
160
|
+
def _reindex_with_indexers(
|
|
161
|
+
self, reindexers, fill_value=None, copy=False, allow_dups=False
|
|
162
|
+
):
|
|
163
|
+
plain = self._plain()._reindex_with_indexers(
|
|
164
|
+
reindexers, fill_value=fill_value, allow_dups=allow_dups
|
|
165
|
+
)
|
|
166
|
+
if copy:
|
|
167
|
+
plain = plain.copy()
|
|
168
|
+
indexer = reindexers.get(0, (None, None))[1]
|
|
169
|
+
if indexer is not None:
|
|
170
|
+
if (np.asarray(indexer) < 0).any():
|
|
171
|
+
raise ValueError(
|
|
172
|
+
"Cannot invent weights for new rows introduced by reindex."
|
|
173
|
+
)
|
|
174
|
+
weights = self.weights.iloc[indexer]
|
|
175
|
+
else:
|
|
176
|
+
weights = self.weights
|
|
177
|
+
return self._weighted_result(plain, weights)
|
|
178
|
+
|
|
179
|
+
def drop(
|
|
180
|
+
self,
|
|
181
|
+
labels=None,
|
|
182
|
+
axis=0,
|
|
183
|
+
index=None,
|
|
184
|
+
columns=None,
|
|
185
|
+
level=None,
|
|
186
|
+
inplace=False,
|
|
187
|
+
errors="raise",
|
|
188
|
+
):
|
|
189
|
+
plain = self._plain().drop(
|
|
190
|
+
labels=labels,
|
|
191
|
+
axis=axis,
|
|
192
|
+
index=index,
|
|
193
|
+
columns=columns,
|
|
194
|
+
level=level,
|
|
195
|
+
errors=errors,
|
|
196
|
+
)
|
|
197
|
+
positions = pd.Series(np.arange(len(self)), index=self.index)
|
|
198
|
+
row_labels = (
|
|
199
|
+
index
|
|
200
|
+
if index is not None
|
|
201
|
+
else labels
|
|
202
|
+
if self._get_axis_number(axis) == 0
|
|
203
|
+
else None
|
|
204
|
+
)
|
|
205
|
+
if row_labels is not None:
|
|
206
|
+
positions = positions.drop(row_labels, level=level, errors=errors)
|
|
207
|
+
return self._finish_row_operation(plain, np.asarray(positions), inplace)
|
|
208
|
+
|
|
209
|
+
def sample(self, *args, **kwargs):
|
|
210
|
+
"""Sample weights with the selected rows, including replacements."""
|
|
211
|
+
result = super().sample(*args, **kwargs)
|
|
212
|
+
result.weights = weight_series(result.weights, result.index)
|
|
213
|
+
return result
|
|
214
|
+
|
|
215
|
+
def sort_values(self, *args, **kwargs):
|
|
216
|
+
"""Sort rows and weights together even when labels are duplicated."""
|
|
217
|
+
inplace = kwargs.pop("inplace", False)
|
|
218
|
+
axis = self._get_axis_number(kwargs.get("axis", 0))
|
|
219
|
+
if self.ndim == 2:
|
|
220
|
+
plain = self._plain()
|
|
221
|
+
if axis == 1:
|
|
222
|
+
plain = plain.sort_values(*args, **kwargs)
|
|
223
|
+
return self._finish_row_operation(plain, np.arange(len(self)), inplace)
|
|
224
|
+
marker = object()
|
|
225
|
+
plain = plain.copy(deep=False)
|
|
226
|
+
plain[marker] = np.arange(len(self))
|
|
227
|
+
plain = plain.sort_values(*args, **kwargs)
|
|
228
|
+
positions = np.asarray(plain.pop(marker), dtype=int)
|
|
229
|
+
else:
|
|
230
|
+
values = self._plain()
|
|
231
|
+
positions = pd.Series(
|
|
232
|
+
np.arange(len(self)), index=self.index, name=self.name
|
|
233
|
+
)
|
|
234
|
+
key = kwargs.pop("key", None)
|
|
235
|
+
positions = positions.sort_values(
|
|
236
|
+
*args,
|
|
237
|
+
key=lambda unused: values if key is None else key(values),
|
|
238
|
+
**kwargs,
|
|
239
|
+
)
|
|
240
|
+
plain = values.iloc[np.asarray(positions)]
|
|
241
|
+
plain.index = positions.index
|
|
242
|
+
positions = np.asarray(positions, dtype=int)
|
|
243
|
+
return self._finish_row_operation(plain, positions, inplace)
|
|
244
|
+
|
|
245
|
+
def sort_index(self, *args, **kwargs):
|
|
246
|
+
"""Sort the index while preserving positional weight provenance."""
|
|
247
|
+
inplace = kwargs.pop("inplace", False)
|
|
248
|
+
axis = self._get_axis_number(kwargs.get("axis", 0))
|
|
249
|
+
if axis == 1:
|
|
250
|
+
plain = self._plain().sort_index(*args, **kwargs)
|
|
251
|
+
return self._finish_row_operation(plain, np.arange(len(self)), inplace)
|
|
252
|
+
positions = pd.Series(np.arange(len(self)), index=self.index).sort_index(
|
|
253
|
+
*args, **kwargs
|
|
254
|
+
)
|
|
255
|
+
plain = self._plain().iloc[np.asarray(positions)]
|
|
256
|
+
plain.index = positions.index
|
|
257
|
+
return self._finish_row_operation(plain, np.asarray(positions), inplace)
|
microdf/microdataframe.py
CHANGED
|
@@ -8,93 +8,17 @@ import numpy as np
|
|
|
8
8
|
import pandas as pd
|
|
9
9
|
|
|
10
10
|
from microdf.microseries import MicroSeries, MicroSeriesGroupBy
|
|
11
|
+
from microdf._weights import (
|
|
12
|
+
WeightPropagationMixin,
|
|
13
|
+
aligned_weights,
|
|
14
|
+
finalize_weights,
|
|
15
|
+
weight_series,
|
|
16
|
+
)
|
|
11
17
|
|
|
12
18
|
logger = logging.getLogger(__name__)
|
|
13
19
|
|
|
14
20
|
|
|
15
|
-
class
|
|
16
|
-
"""Custom loc indexer that returns MicroDataFrame with proper weights."""
|
|
17
|
-
|
|
18
|
-
def __init__(self, mdf: "MicroDataFrame"):
|
|
19
|
-
self._mdf = mdf
|
|
20
|
-
# Get the parent's loc indexer
|
|
21
|
-
self._parent_loc = pd.DataFrame.loc.fget(mdf)
|
|
22
|
-
|
|
23
|
-
def __getitem__(self, key):
|
|
24
|
-
# Use the parent DataFrame's loc indexer
|
|
25
|
-
result = self._parent_loc[key]
|
|
26
|
-
|
|
27
|
-
if isinstance(result, pd.DataFrame):
|
|
28
|
-
# Get the filtered weights based on the result's index
|
|
29
|
-
new_weights = self._mdf.weights.reindex(result.index)
|
|
30
|
-
return MicroDataFrame(result, weights=new_weights)
|
|
31
|
-
elif isinstance(result, pd.Series):
|
|
32
|
-
# Single row or column selected
|
|
33
|
-
if result.name in self._mdf.columns:
|
|
34
|
-
# Column was selected - return MicroSeries with all weights
|
|
35
|
-
return MicroSeries(result, weights=self._mdf.weights)
|
|
36
|
-
else:
|
|
37
|
-
# Row was selected - return as-is (scalar values for each col)
|
|
38
|
-
return result
|
|
39
|
-
else:
|
|
40
|
-
# Scalar value
|
|
41
|
-
return result
|
|
42
|
-
|
|
43
|
-
def __setitem__(self, key, value):
|
|
44
|
-
self._parent_loc[key] = value
|
|
45
|
-
self._mdf._link_all_weights()
|
|
46
|
-
|
|
47
|
-
def __getattr__(self, name):
|
|
48
|
-
"""Delegate unknown attributes to the parent loc indexer."""
|
|
49
|
-
return getattr(self._parent_loc, name)
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
class _MicroILocIndexer:
|
|
53
|
-
"""Custom iloc indexer that returns MicroDataFrame with proper weights."""
|
|
54
|
-
|
|
55
|
-
def __init__(self, mdf: "MicroDataFrame"):
|
|
56
|
-
self._mdf = mdf
|
|
57
|
-
# Get the parent's iloc indexer
|
|
58
|
-
self._parent_iloc = pd.DataFrame.iloc.fget(mdf)
|
|
59
|
-
|
|
60
|
-
def __getitem__(self, key):
|
|
61
|
-
# Use the parent DataFrame's iloc indexer
|
|
62
|
-
result = self._parent_iloc[key]
|
|
63
|
-
|
|
64
|
-
if isinstance(result, pd.DataFrame):
|
|
65
|
-
# Get the filtered weights based on the result's index
|
|
66
|
-
new_weights = self._mdf.weights.iloc[
|
|
67
|
-
self._mdf.index.get_indexer(result.index)
|
|
68
|
-
]
|
|
69
|
-
new_weights = pd.Series(new_weights.values, index=result.index)
|
|
70
|
-
return MicroDataFrame(result, weights=new_weights)
|
|
71
|
-
elif isinstance(result, pd.Series):
|
|
72
|
-
# Single row or column selected
|
|
73
|
-
if isinstance(key, tuple) and len(key) == 2:
|
|
74
|
-
# df.iloc[:, col_idx] - column selection
|
|
75
|
-
row_key = key[0]
|
|
76
|
-
if isinstance(row_key, slice) and row_key == slice(None):
|
|
77
|
-
# All rows selected for a column
|
|
78
|
-
return MicroSeries(result, weights=self._mdf.weights)
|
|
79
|
-
# Check if this is a column (result index matches mdf index)
|
|
80
|
-
if result.index.equals(self._mdf.index):
|
|
81
|
-
return MicroSeries(result, weights=self._mdf.weights)
|
|
82
|
-
# Row selection - return as-is
|
|
83
|
-
return result
|
|
84
|
-
else:
|
|
85
|
-
# Scalar value
|
|
86
|
-
return result
|
|
87
|
-
|
|
88
|
-
def __setitem__(self, key, value):
|
|
89
|
-
self._parent_iloc[key] = value
|
|
90
|
-
self._mdf._link_all_weights()
|
|
91
|
-
|
|
92
|
-
def __getattr__(self, name):
|
|
93
|
-
"""Delegate unknown attributes to the parent iloc indexer."""
|
|
94
|
-
return getattr(self._parent_iloc, name)
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
class MicroDataFrame(pd.DataFrame):
|
|
21
|
+
class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
|
|
98
22
|
# Declare weight state as pandas metadata. pandas includes
|
|
99
23
|
# _metadata attributes in the pickle state, so weights now survive
|
|
100
24
|
# pickling, to_pickle/read_pickle and copy.deepcopy instead of
|
|
@@ -112,20 +36,54 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
112
36
|
:type weights: np.array
|
|
113
37
|
"""
|
|
114
38
|
super().__init__(*args, **kwargs)
|
|
115
|
-
|
|
39
|
+
# pandas normalizes mixed-dimensional concat inputs through this
|
|
40
|
+
# constructor, either as a Series or a one-column mapping. Preserve
|
|
41
|
+
# that Series' row weights before concat loses the original input.
|
|
42
|
+
weight_source = args[0] if args else kwargs.get("data")
|
|
43
|
+
if isinstance(weight_source, dict) and len(weight_source) == 1:
|
|
44
|
+
weight_source = next(iter(weight_source.values()))
|
|
45
|
+
if weights is None and isinstance(weight_source, MicroSeries):
|
|
46
|
+
weights = aligned_weights(weight_source, self.index)
|
|
47
|
+
self.weights = weight_series(np.ones(len(self)), self.index)
|
|
48
|
+
self.weights_col = None
|
|
116
49
|
self.set_weights(weights)
|
|
117
50
|
self._link_all_weights()
|
|
118
51
|
self.override_df_functions()
|
|
119
52
|
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
53
|
+
@property
|
|
54
|
+
def _constructor(self):
|
|
55
|
+
return MicroDataFrame
|
|
56
|
+
|
|
57
|
+
# A row or a column-summary Series has no per-observation weights.
|
|
58
|
+
# _ixs wraps column selections using their unambiguous row provenance.
|
|
59
|
+
_constructor_sliced = pd.Series
|
|
60
|
+
|
|
61
|
+
def _ixs(self, i, axis=0):
|
|
62
|
+
result = pd.DataFrame(self, copy=False)._ixs(i, axis=axis)
|
|
63
|
+
if axis == 1:
|
|
64
|
+
return MicroSeries(result, weights=self.weights)
|
|
65
|
+
return result
|
|
66
|
+
|
|
67
|
+
def _get_item_cache(self, item):
|
|
68
|
+
# Weight arrays are independently mutable; cached column wrappers would
|
|
69
|
+
# retain stale copies after an in-place edit to frame.weights.
|
|
70
|
+
return self._ixs(self.columns.get_loc(item), axis=1)
|
|
71
|
+
|
|
72
|
+
def __finalize__(self, other, method=None, **kwargs):
|
|
73
|
+
previous = self.__dict__.get("weights")
|
|
123
74
|
super().__finalize__(other, method=method, **kwargs)
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
75
|
+
return finalize_weights(self, other, method, previous)
|
|
76
|
+
|
|
77
|
+
@wraps(pd.DataFrame.cov)
|
|
78
|
+
def cov(self, *args, **kwargs) -> pd.DataFrame:
|
|
79
|
+
# Column summaries have no observation weights, even if labels match.
|
|
80
|
+
result = pd.DataFrame(self, copy=False).cov(*args, **kwargs)
|
|
81
|
+
return result.__finalize__(self, method="cov")
|
|
82
|
+
|
|
83
|
+
@wraps(pd.DataFrame.corr)
|
|
84
|
+
def corr(self, *args, **kwargs) -> pd.DataFrame:
|
|
85
|
+
result = pd.DataFrame(self, copy=False).corr(*args, **kwargs)
|
|
86
|
+
return result.__finalize__(self, method="corr")
|
|
129
87
|
|
|
130
88
|
def __setstate__(self, state) -> None:
|
|
131
89
|
"""Restore a pickled MicroDataFrame.
|
|
@@ -141,23 +99,6 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
141
99
|
self._link_all_weights()
|
|
142
100
|
self.override_df_functions()
|
|
143
101
|
|
|
144
|
-
@property
|
|
145
|
-
def loc(self) -> _MicroLocIndexer:
|
|
146
|
-
"""Label-based indexer that preserves MicroDataFrame type and weights.
|
|
147
|
-
|
|
148
|
-
:return: Custom loc indexer for MicroDataFrame
|
|
149
|
-
"""
|
|
150
|
-
return _MicroLocIndexer(self)
|
|
151
|
-
|
|
152
|
-
@property
|
|
153
|
-
def iloc(self) -> _MicroILocIndexer:
|
|
154
|
-
"""Integer-based indexer that preserves MicroDataFrame type and
|
|
155
|
-
weights.
|
|
156
|
-
|
|
157
|
-
:return: Custom iloc indexer for MicroDataFrame
|
|
158
|
-
"""
|
|
159
|
-
return _MicroILocIndexer(self)
|
|
160
|
-
|
|
161
102
|
def override_df_functions(self) -> None:
|
|
162
103
|
"""Override DataFrame functions to work with weighted operations."""
|
|
163
104
|
for name in MicroSeries.FUNCTIONS:
|
|
@@ -378,7 +319,7 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
378
319
|
pass
|
|
379
320
|
|
|
380
321
|
def _link_all_weights(self) -> None:
|
|
381
|
-
if self.weights is None:
|
|
322
|
+
if self.weights is None or len(self.weights) == 0:
|
|
382
323
|
if len(self) > 0:
|
|
383
324
|
self.set_weights(np.ones((len(self))))
|
|
384
325
|
# In pandas 3.0+, columns are wrapped as MicroSeries on access via
|
|
@@ -472,33 +413,19 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
472
413
|
# that treats it as a Series (equals(), reindex() in __getitem__).
|
|
473
414
|
self.set_weights(np.ones(len(self)))
|
|
474
415
|
|
|
475
|
-
def __getitem__(
|
|
476
|
-
|
|
477
|
-
) -> Union[MicroSeries, "MicroDataFrame"]:
|
|
478
|
-
# Let pandas handle the initial slicing
|
|
479
|
-
result = super().__getitem__(key)
|
|
480
|
-
|
|
481
|
-
# If the result is a DataFrame, re-synchronize the weights
|
|
482
|
-
if isinstance(result, pd.DataFrame):
|
|
483
|
-
new_weights = self.weights.reindex(result.index)
|
|
484
|
-
return MicroDataFrame(result, weights=new_weights)
|
|
485
|
-
|
|
486
|
-
# If the result is a Series (single column), wrap as MicroSeries
|
|
487
|
-
if isinstance(result, pd.Series):
|
|
488
|
-
return MicroSeries(result, weights=self.weights)
|
|
489
|
-
|
|
490
|
-
# Otherwise, the result is a scalar, so just return it
|
|
491
|
-
return result
|
|
416
|
+
def __getitem__(self, key):
|
|
417
|
+
return super().__getitem__(key)
|
|
492
418
|
|
|
493
419
|
def catch_series_relapse(self) -> None:
|
|
494
420
|
# In pandas 3.0+, we don't need to track series class changes since
|
|
495
421
|
# __getitem__ always wraps columns as MicroSeries on access.
|
|
496
422
|
pass
|
|
497
423
|
|
|
498
|
-
def __setattr__(self, key, value)
|
|
424
|
+
def __setattr__(self, key, value):
|
|
425
|
+
weights = self.__dict__.get("weights") if key == "index" else None
|
|
499
426
|
super().__setattr__(key, value)
|
|
500
|
-
|
|
501
|
-
|
|
427
|
+
if weights is not None and len(weights) == len(self.index):
|
|
428
|
+
self.weights = weight_series(weights, self.index)
|
|
502
429
|
|
|
503
430
|
def reset_index(
|
|
504
431
|
self,
|
|
@@ -567,13 +494,7 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
567
494
|
return out
|
|
568
495
|
|
|
569
496
|
def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame":
|
|
570
|
-
|
|
571
|
-
# super().copy() corrupts self's column types to plain Series.
|
|
572
|
-
# Restore them in O(N) instead of O(N²) by calling
|
|
573
|
-
# _link_all_weights once rather than per-column __setitem__.
|
|
574
|
-
self._link_all_weights()
|
|
575
|
-
res = MicroDataFrame(res, weights=self.weights.copy(deep))
|
|
576
|
-
return res
|
|
497
|
+
return super().copy(deep)
|
|
577
498
|
|
|
578
499
|
def drop(
|
|
579
500
|
self,
|
|
@@ -605,55 +526,15 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
605
526
|
dropped.
|
|
606
527
|
:return: MicroDataFrame or None if inplace=True.
|
|
607
528
|
"""
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
index=index,
|
|
618
|
-
columns=columns,
|
|
619
|
-
level=level,
|
|
620
|
-
inplace=True,
|
|
621
|
-
errors=errors,
|
|
622
|
-
)
|
|
623
|
-
if row_drop:
|
|
624
|
-
surviving = pre_drop_weights.reindex(self.index)
|
|
625
|
-
self.weights = pd.Series(
|
|
626
|
-
surviving.values, index=self.index, dtype=float
|
|
627
|
-
)
|
|
628
|
-
else:
|
|
629
|
-
self.weights = pd.Series(
|
|
630
|
-
pre_drop_weights.values, index=self.index, dtype=float
|
|
631
|
-
)
|
|
632
|
-
self._link_all_weights()
|
|
633
|
-
return None
|
|
634
|
-
else:
|
|
635
|
-
res = super().drop(
|
|
636
|
-
labels=labels,
|
|
637
|
-
axis=axis,
|
|
638
|
-
index=index,
|
|
639
|
-
columns=columns,
|
|
640
|
-
level=level,
|
|
641
|
-
inplace=False,
|
|
642
|
-
errors=errors,
|
|
643
|
-
)
|
|
644
|
-
if row_drop:
|
|
645
|
-
# Row drop: keep only the weights for surviving rows,
|
|
646
|
-
# in the order of the resulting DataFrame.
|
|
647
|
-
pre_drop_weights = pd.Series(self.weights.values, index=self.index)
|
|
648
|
-
new_weights = pre_drop_weights.reindex(res.index).values
|
|
649
|
-
else:
|
|
650
|
-
new_weights = self.weights.values
|
|
651
|
-
out = MicroDataFrame(res, weights=new_weights)
|
|
652
|
-
# Guard against the set_weights path building weights with a
|
|
653
|
-
# default RangeIndex, which would misalign against res.index
|
|
654
|
-
# and silently zero weighted aggregations.
|
|
655
|
-
out.weights = pd.Series(new_weights, index=out.index, dtype=float)
|
|
656
|
-
return out
|
|
529
|
+
return super().drop(
|
|
530
|
+
labels=labels,
|
|
531
|
+
axis=axis,
|
|
532
|
+
index=index,
|
|
533
|
+
columns=columns,
|
|
534
|
+
level=level,
|
|
535
|
+
inplace=inplace,
|
|
536
|
+
errors=errors,
|
|
537
|
+
)
|
|
657
538
|
|
|
658
539
|
def merge(
|
|
659
540
|
self,
|
microdf/microseries.py
CHANGED
|
@@ -6,6 +6,8 @@ from typing import Callable, List, Optional, Union
|
|
|
6
6
|
import numpy as np
|
|
7
7
|
import pandas as pd
|
|
8
8
|
|
|
9
|
+
from microdf._weights import WeightPropagationMixin, finalize_weights, weight_series
|
|
10
|
+
|
|
9
11
|
logger = logging.getLogger(__name__)
|
|
10
12
|
|
|
11
13
|
|
|
@@ -84,7 +86,7 @@ def _weighted_top_share(
|
|
|
84
86
|
return top_sum / total_sum
|
|
85
87
|
|
|
86
88
|
|
|
87
|
-
class MicroSeries(pd.Series):
|
|
89
|
+
class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
88
90
|
# Declare ``weights`` as pandas metadata. pandas includes
|
|
89
91
|
# _metadata attributes in the pickle state, so weights now survive
|
|
90
92
|
# pickling, to_pickle/read_pickle and copy.deepcopy instead of
|
|
@@ -103,15 +105,26 @@ class MicroSeries(pd.Series):
|
|
|
103
105
|
super().__init__(*args, **kwargs)
|
|
104
106
|
self.set_weights(weights)
|
|
105
107
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
108
|
+
@property
|
|
109
|
+
def _constructor(self):
|
|
110
|
+
return MicroSeries
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def _constructor_expanddim(self):
|
|
114
|
+
from microdf.microdataframe import MicroDataFrame
|
|
115
|
+
|
|
116
|
+
return MicroDataFrame
|
|
117
|
+
|
|
118
|
+
def __finalize__(self, other, method=None, **kwargs):
|
|
119
|
+
previous = self.__dict__.get("weights")
|
|
109
120
|
super().__finalize__(other, method=method, **kwargs)
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
121
|
+
return finalize_weights(self, other, method, previous)
|
|
122
|
+
|
|
123
|
+
def __setattr__(self, name, value):
|
|
124
|
+
weights = self.__dict__.get("weights") if name == "index" else None
|
|
125
|
+
super().__setattr__(name, value)
|
|
126
|
+
if weights is not None and len(weights) == len(self.index):
|
|
127
|
+
self.weights = weight_series(weights, self.index)
|
|
115
128
|
|
|
116
129
|
@property
|
|
117
130
|
def _values(self):
|
|
@@ -178,12 +191,7 @@ class MicroSeries(pd.Series):
|
|
|
178
191
|
:type weights: np.array.
|
|
179
192
|
"""
|
|
180
193
|
if weights is None:
|
|
181
|
-
|
|
182
|
-
self.weights = pd.Series(
|
|
183
|
-
np.ones_like(self._values),
|
|
184
|
-
index=self.index,
|
|
185
|
-
dtype=float,
|
|
186
|
-
)
|
|
194
|
+
self.weights = weight_series(np.ones(len(self)), self.index)
|
|
187
195
|
else:
|
|
188
196
|
if len(weights) != len(self):
|
|
189
197
|
raise ValueError(
|
|
@@ -201,7 +209,7 @@ class MicroSeries(pd.Series):
|
|
|
201
209
|
# its index first so we position-align rather than label-align.
|
|
202
210
|
if isinstance(weights, pd.Series):
|
|
203
211
|
weights = weights.values
|
|
204
|
-
self.weights =
|
|
212
|
+
self.weights = weight_series(weights, self.index)
|
|
205
213
|
|
|
206
214
|
def nullify_weights(self) -> None:
|
|
207
215
|
"""Set all weights to 1, effectively making the Series unweighted.
|
|
@@ -824,16 +832,14 @@ class MicroSeries(pd.Series):
|
|
|
824
832
|
)
|
|
825
833
|
|
|
826
834
|
def groupby(self, *args, **kwargs) -> "MicroSeriesGroupBy":
|
|
827
|
-
gb =
|
|
835
|
+
gb = pd.Series(self, copy=False).groupby(*args, **kwargs)
|
|
828
836
|
gb.__class__ = MicroSeriesGroupBy
|
|
829
837
|
gb._init()
|
|
830
838
|
gb.weights = pd.Series(self.weights).groupby(*args, **kwargs)
|
|
831
839
|
return gb
|
|
832
840
|
|
|
833
841
|
def copy(self, deep: Optional[bool] = True):
|
|
834
|
-
|
|
835
|
-
res = MicroSeries(res, weights=self.weights.copy(deep))
|
|
836
|
-
return res
|
|
842
|
+
return super().copy(deep)
|
|
837
843
|
|
|
838
844
|
def clip(
|
|
839
845
|
self,
|
|
@@ -865,15 +871,22 @@ class MicroSeries(pd.Series):
|
|
|
865
871
|
equal_weights = self.weights.equals(other.weights)
|
|
866
872
|
return equal_values and equal_weights
|
|
867
873
|
|
|
868
|
-
def __getitem__(
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
result =
|
|
874
|
+
def __getitem__(self, key):
|
|
875
|
+
if callable(key):
|
|
876
|
+
key = key(self)
|
|
877
|
+
result = pd.Series(self, copy=False).__getitem__(key)
|
|
872
878
|
if isinstance(result, pd.Series):
|
|
873
|
-
|
|
874
|
-
|
|
879
|
+
positions = pd.Series(np.arange(len(self)), index=self.index).__getitem__(
|
|
880
|
+
key
|
|
881
|
+
)
|
|
882
|
+
return MicroSeries(result, weights=self.weights.iloc[np.asarray(positions)])
|
|
875
883
|
return result
|
|
876
884
|
|
|
885
|
+
def repeat(self, repeats, axis=None):
|
|
886
|
+
# Use pandas to validate the repeat counts and axis argument.
|
|
887
|
+
positions = pd.Series(np.arange(len(self))).repeat(repeats, axis=axis)
|
|
888
|
+
return self.take(np.asarray(positions))
|
|
889
|
+
|
|
877
890
|
def __getattr__(self, name: str) -> "MicroSeries":
|
|
878
891
|
return MicroSeries(super().__getattr__(name), weights=self.weights)
|
|
879
892
|
|
|
@@ -0,0 +1,478 @@
|
|
|
1
|
+
"""Weighted pandas operations preserve row identity, not just row labels."""
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import pickle
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
import microdf as mdf
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def make_data(kind):
|
|
14
|
+
values = pd.Series([30.0, np.nan, 10.0, 20.0], index=[7, 7, 3, 7], name="x")
|
|
15
|
+
weights = [2.0, 5.0, 11.0, 17.0]
|
|
16
|
+
if kind == "series":
|
|
17
|
+
return mdf.MicroSeries(values, weights=weights)
|
|
18
|
+
return mdf.MicroDataFrame(values.to_frame(), weights=weights)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def assert_rows(result, original, positions, values=None):
|
|
22
|
+
assert type(result) is type(original)
|
|
23
|
+
expected = original.weights.to_numpy()[positions]
|
|
24
|
+
np.testing.assert_array_equal(result.weights.to_numpy(), expected)
|
|
25
|
+
assert result.weights.index.equals(result.index)
|
|
26
|
+
assert result.weights is not original.weights
|
|
27
|
+
data = np.asarray(result if isinstance(result, mdf.MicroSeries) else result["x"])
|
|
28
|
+
if values is not None:
|
|
29
|
+
np.testing.assert_allclose(data, values, equal_nan=True)
|
|
30
|
+
total = np.nansum(data * expected)
|
|
31
|
+
assert (
|
|
32
|
+
result.sum() if isinstance(result, mdf.MicroSeries) else result.sum()["x"]
|
|
33
|
+
) == total
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
37
|
+
@pytest.mark.parametrize("ascending", [True, False])
|
|
38
|
+
@pytest.mark.parametrize("ignore_index", [True, False])
|
|
39
|
+
def test_sort_values_preserves_positional_weights(kind, ascending, ignore_index):
|
|
40
|
+
original = make_data(kind)
|
|
41
|
+
args = () if kind == "series" else ("x",)
|
|
42
|
+
result = original.sort_values(*args, ascending=ascending, ignore_index=ignore_index)
|
|
43
|
+
positions = [2, 3, 0, 1] if ascending else [0, 3, 2, 1]
|
|
44
|
+
assert_rows(result, original, positions)
|
|
45
|
+
if ignore_index:
|
|
46
|
+
assert result.index.equals(pd.RangeIndex(4))
|
|
47
|
+
assert_rows(original, original.copy(), np.arange(4))
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
51
|
+
def test_fillna_and_row_transforms_preserve_weights(kind):
|
|
52
|
+
original = make_data(kind)
|
|
53
|
+
filled = original.fillna(4)
|
|
54
|
+
assert_rows(filled, original, np.arange(4), [30, 4, 10, 20])
|
|
55
|
+
assert_rows(filled.replace(30, 40), filled, np.arange(4), [40, 4, 10, 20])
|
|
56
|
+
assert_rows(filled.astype(float), filled, np.arange(4))
|
|
57
|
+
assert_rows(filled + 1, filled, np.arange(4), [31, 5, 11, 21])
|
|
58
|
+
filled.weights.iloc[0] = 100
|
|
59
|
+
assert original.weights.iloc[0] == 2
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
63
|
+
@pytest.mark.parametrize("ignore_index", [True, False])
|
|
64
|
+
def test_sample_with_replacement_and_duplicate_labels(kind, ignore_index):
|
|
65
|
+
original = make_data(kind)
|
|
66
|
+
positions = (
|
|
67
|
+
pd.Series(np.arange(4)).sample(n=12, replace=True, random_state=17).to_numpy()
|
|
68
|
+
)
|
|
69
|
+
result = original.sample(
|
|
70
|
+
n=12, replace=True, random_state=17, ignore_index=ignore_index
|
|
71
|
+
)
|
|
72
|
+
assert_rows(result, original, positions)
|
|
73
|
+
assert len(set(positions)) < len(positions)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
77
|
+
def test_duplicate_label_selections_and_sort_index(kind):
|
|
78
|
+
original = make_data(kind)
|
|
79
|
+
assert_rows(original.iloc[[3, 0, 3]], original, [3, 0, 3])
|
|
80
|
+
assert_rows(original.iloc[1:3], original, [1, 2])
|
|
81
|
+
assert_rows(original.loc[[3, 7]], original, [2, 0, 1, 3])
|
|
82
|
+
assert_rows(original.sort_index(kind="stable"), original, [2, 0, 1, 3])
|
|
83
|
+
if kind == "frame":
|
|
84
|
+
column = original.loc[[3, 7], "x"]
|
|
85
|
+
assert isinstance(column, mdf.MicroSeries)
|
|
86
|
+
np.testing.assert_array_equal(column.weights, [11, 2, 5, 17])
|
|
87
|
+
row = original.iloc[0]
|
|
88
|
+
assert type(row) is pd.Series
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
92
|
+
@pytest.mark.parametrize("options", [{}, {"ignore_index": True}, {"keys": ["a", "b"]}])
|
|
93
|
+
def test_concat_rows_preserves_duplicate_and_repeated_weights(kind, options):
|
|
94
|
+
original = make_data(kind)
|
|
95
|
+
pieces = [original.iloc[[3, 0]], original.iloc[[2, 3]]]
|
|
96
|
+
result = pd.concat(pieces, **options)
|
|
97
|
+
assert_rows(result, original, [3, 0, 2, 3])
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_concat_columns_requires_consistent_row_weights():
|
|
101
|
+
first = mdf.MicroDataFrame({"x": [1, 2]}, index=[9, 3], weights=[4, 5])
|
|
102
|
+
second = mdf.MicroDataFrame({"y": [6, 7]}, index=[3, 9], weights=[5, 4])
|
|
103
|
+
result = pd.concat([first, second], axis=1)
|
|
104
|
+
assert isinstance(result, mdf.MicroDataFrame)
|
|
105
|
+
np.testing.assert_array_equal(result.weights, [4, 5])
|
|
106
|
+
assert result.sum().to_dict() == {"x": 14, "y": 58}
|
|
107
|
+
second.set_weights([50, 40])
|
|
108
|
+
with pytest.raises(ValueError, match="weights"):
|
|
109
|
+
pd.concat([first, second], axis=1)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
113
|
+
def test_concat_with_unweighted_objects_is_explicit(kind):
|
|
114
|
+
weighted = make_data(kind)
|
|
115
|
+
plain = pd.Series([1, 2]) if kind == "series" else pd.DataFrame({"x": [1, 2]})
|
|
116
|
+
with pytest.raises(ValueError, match="weights"):
|
|
117
|
+
pd.concat([weighted, plain])
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
121
|
+
def test_reindex_new_rows_cannot_invent_weights(kind):
|
|
122
|
+
weighted = make_data(kind).iloc[:1]
|
|
123
|
+
with pytest.raises(ValueError, match="weights"):
|
|
124
|
+
weighted.reindex([7, 99])
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
128
|
+
def test_propagated_objects_keep_serialization_and_rename_isolation(kind):
|
|
129
|
+
original = make_data(kind).fillna(4)
|
|
130
|
+
renamed = (
|
|
131
|
+
original.rename("income")
|
|
132
|
+
if kind == "series"
|
|
133
|
+
else original.rename(columns={"x": "income"})
|
|
134
|
+
)
|
|
135
|
+
renamed.weights.iloc[0] = 99
|
|
136
|
+
assert original.weights.iloc[0] == 2
|
|
137
|
+
for result in [copy.deepcopy(original), pickle.loads(pickle.dumps(original))]:
|
|
138
|
+
assert_rows(result, original, np.arange(4))
|
|
139
|
+
if kind == "series":
|
|
140
|
+
assert original.name == "x"
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
144
|
+
def test_inplace_sort_and_fillna_preserve_weights(kind):
|
|
145
|
+
original = make_data(kind)
|
|
146
|
+
expected = original.copy()
|
|
147
|
+
args = () if kind == "series" else ("x",)
|
|
148
|
+
assert original.sort_values(*args, inplace=True, ascending=False) is None
|
|
149
|
+
assert_rows(original, expected, [0, 3, 2, 1])
|
|
150
|
+
original.fillna(4, inplace=True)
|
|
151
|
+
assert_rows(original, expected, [0, 3, 2, 1], [30, 20, 10, 4])
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@pytest.mark.parametrize("kind", ["series", "frame"])
|
|
155
|
+
def test_boolean_selection_and_drop_with_duplicate_indexes(kind):
|
|
156
|
+
original = make_data(kind)
|
|
157
|
+
mask = np.array([True, False, True, False])
|
|
158
|
+
assert_rows(original[mask], original, [0, 2])
|
|
159
|
+
assert_rows(original.drop(index=7), original, [2])
|
|
160
|
+
assert_rows(original.loc[mask], original, [0, 2])
|
|
161
|
+
if kind == "series":
|
|
162
|
+
assert_rows(original.repeat(2), original, [0, 0, 1, 1, 2, 2, 3, 3])
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_column_access_uses_current_weights_without_mutating_earlier_series():
|
|
166
|
+
frame = mdf.MicroDataFrame({"x": [1, 2]}, weights=[3, 4])
|
|
167
|
+
earlier = frame["x"]
|
|
168
|
+
assert earlier.sum() == 11
|
|
169
|
+
frame.weights.iloc[0] = 10
|
|
170
|
+
assert frame["x"].sum() == 18
|
|
171
|
+
assert frame.loc[:, "x"].sum() == 18
|
|
172
|
+
assert earlier.sum() == 11
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def test_transpose_cannot_reinterpret_row_weights_as_column_weights():
|
|
176
|
+
frame = mdf.MicroDataFrame(
|
|
177
|
+
[[1, 2], [3, 4]], index=[0, 1], columns=[0, 1], weights=[5, 7]
|
|
178
|
+
)
|
|
179
|
+
with pytest.raises(ValueError, match="weights"):
|
|
180
|
+
frame.transpose()
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
@pytest.mark.parametrize("inplace", [False, True])
|
|
184
|
+
@pytest.mark.parametrize(
|
|
185
|
+
"index,level",
|
|
186
|
+
[
|
|
187
|
+
(pd.Index(["a", "a", "b"], name="row"), None),
|
|
188
|
+
(
|
|
189
|
+
pd.MultiIndex.from_tuples(
|
|
190
|
+
[("a", 1), ("a", 1), ("b", 2)], names=["group", "row"]
|
|
191
|
+
),
|
|
192
|
+
None,
|
|
193
|
+
),
|
|
194
|
+
(
|
|
195
|
+
pd.MultiIndex.from_tuples(
|
|
196
|
+
[("a", 1), ("a", 1), ("b", 2)], names=["group", "row"]
|
|
197
|
+
),
|
|
198
|
+
"group",
|
|
199
|
+
),
|
|
200
|
+
],
|
|
201
|
+
)
|
|
202
|
+
def test_series_reset_index_drop_preserves_positional_weights(index, level, inplace):
|
|
203
|
+
original = mdf.MicroSeries(
|
|
204
|
+
[10, 20, 30], index=index, name="income", weights=[2, 3, 5]
|
|
205
|
+
)
|
|
206
|
+
source_weights = original.weights
|
|
207
|
+
expected = pd.Series(original).reset_index(level=level, drop=True, name="ignored")
|
|
208
|
+
result = original.reset_index(
|
|
209
|
+
level=level, drop=True, name="ignored", inplace=inplace
|
|
210
|
+
)
|
|
211
|
+
if inplace:
|
|
212
|
+
assert result is None
|
|
213
|
+
result = original
|
|
214
|
+
assert isinstance(result, mdf.MicroSeries)
|
|
215
|
+
pd.testing.assert_series_equal(pd.Series(result), expected)
|
|
216
|
+
pd.testing.assert_series_equal(
|
|
217
|
+
result.weights, pd.Series([2.0, 3.0, 5.0], index=expected.index)
|
|
218
|
+
)
|
|
219
|
+
assert result.sum() == 230
|
|
220
|
+
assert result.weights is not source_weights
|
|
221
|
+
result.weights.iloc[0] = 99
|
|
222
|
+
assert source_weights.iloc[0] == 2
|
|
223
|
+
source_weights.iloc[1] = 88
|
|
224
|
+
assert result.weights.iloc[1] == 3
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def test_series_reset_index_preserves_dataframe_and_invalid_inplace_behavior():
|
|
228
|
+
original = mdf.MicroSeries(
|
|
229
|
+
[10, 20], index=pd.Index(["a", "b"], name="row"), name="income", weights=[2, 3]
|
|
230
|
+
)
|
|
231
|
+
result = original.reset_index(name="amount")
|
|
232
|
+
assert isinstance(result, mdf.MicroDataFrame)
|
|
233
|
+
pd.testing.assert_frame_equal(
|
|
234
|
+
pd.DataFrame(result), pd.Series(original).reset_index(name="amount")
|
|
235
|
+
)
|
|
236
|
+
np.testing.assert_array_equal(result.weights, [2, 3])
|
|
237
|
+
result.weights.iloc[0] = 99
|
|
238
|
+
assert original.weights.iloc[0] == 2
|
|
239
|
+
with pytest.raises(TypeError, match="inplace"):
|
|
240
|
+
original.reset_index(inplace=True)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
@pytest.mark.parametrize(
|
|
244
|
+
"select,positions",
|
|
245
|
+
[
|
|
246
|
+
(lambda frame: frame.iloc[1:], [1, 2]),
|
|
247
|
+
(lambda frame: frame[["income"]], [0, 1, 2]),
|
|
248
|
+
(lambda frame: frame.iloc[:, :1], [0, 1, 2]),
|
|
249
|
+
(lambda frame: frame.loc[:, ["income"]], [0, 1, 2]),
|
|
250
|
+
(lambda frame: frame.iloc[[2, 0, 2]], [2, 0, 2]),
|
|
251
|
+
(lambda frame: frame.reindex(columns=["income"]), [0, 1, 2]),
|
|
252
|
+
],
|
|
253
|
+
ids=["row-slice", "columns", "iloc-columns", "loc-columns", "repeated", "reindex"],
|
|
254
|
+
)
|
|
255
|
+
def test_selected_dataframe_weights_are_independently_mutable(select, positions):
|
|
256
|
+
source = mdf.MicroDataFrame(
|
|
257
|
+
{"income": [10.0, 20.0, 30.0], "other": [1, 2, 3]},
|
|
258
|
+
index=[7, 7, 3],
|
|
259
|
+
weights=[2, 3, 5],
|
|
260
|
+
)
|
|
261
|
+
selected = select(source)
|
|
262
|
+
expected_weights = np.array([2.0, 3.0, 5.0])[positions]
|
|
263
|
+
np.testing.assert_array_equal(selected.weights, expected_weights)
|
|
264
|
+
assert selected.weights.index.equals(selected.index)
|
|
265
|
+
|
|
266
|
+
selected.weights.iloc[0] = 100
|
|
267
|
+
expected_weights[0] = 100
|
|
268
|
+
np.testing.assert_array_equal(source.weights, [2, 3, 5])
|
|
269
|
+
assert source.income.sum() == 230 # 10 * 2 + 20 * 3 + 30 * 5.
|
|
270
|
+
assert selected.income.sum() == np.dot(
|
|
271
|
+
np.array([10.0, 20.0, 30.0])[positions], expected_weights
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
source.weights.iloc[-1] = 200
|
|
275
|
+
np.testing.assert_array_equal(selected.weights, expected_weights)
|
|
276
|
+
assert selected.income.sum() == np.dot(
|
|
277
|
+
np.array([10.0, 20.0, 30.0])[positions], expected_weights
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
@pytest.mark.parametrize("series_first", [False, True])
|
|
282
|
+
@pytest.mark.parametrize("options", [{}, {"ignore_index": True}, {"keys": ["a", "b"]}])
|
|
283
|
+
@pytest.mark.parametrize("series_name", ["income", None])
|
|
284
|
+
def test_concat_mixed_dimensions_preserves_series_weights(
|
|
285
|
+
series_first, options, series_name
|
|
286
|
+
):
|
|
287
|
+
frame = mdf.MicroDataFrame({"income": [10.0, 20.0]}, index=[7, 7], weights=[2, 3])
|
|
288
|
+
series = mdf.MicroSeries(
|
|
289
|
+
[30.0, 40.0], index=[7, 3], name=series_name, weights=[5, 7]
|
|
290
|
+
)
|
|
291
|
+
parts = [series, frame] if series_first else [frame, series]
|
|
292
|
+
plain_parts = [
|
|
293
|
+
pd.Series(part) if part.ndim == 1 else pd.DataFrame(part) for part in parts
|
|
294
|
+
]
|
|
295
|
+
expected = pd.concat(plain_parts, **options)
|
|
296
|
+
expected_weights = [5, 7, 2, 3] if series_first else [2, 3, 5, 7]
|
|
297
|
+
|
|
298
|
+
result = pd.concat(parts, **options)
|
|
299
|
+
assert isinstance(result, mdf.MicroDataFrame)
|
|
300
|
+
pd.testing.assert_frame_equal(pd.DataFrame(result), expected)
|
|
301
|
+
np.testing.assert_array_equal(result.weights, expected_weights)
|
|
302
|
+
assert result.weights.index.equals(result.index)
|
|
303
|
+
for column in expected:
|
|
304
|
+
assert result[column].sum() == np.nansum(
|
|
305
|
+
expected[column].to_numpy() * expected_weights
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
result.weights.iloc[0] = 100
|
|
309
|
+
np.testing.assert_array_equal(frame.weights, [2, 3])
|
|
310
|
+
np.testing.assert_array_equal(series.weights, [5, 7])
|
|
311
|
+
series.weights.iloc[-1] = 200
|
|
312
|
+
series_last_position = 1 if series_first else 3
|
|
313
|
+
assert result.weights.iloc[series_last_position] == 7
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
@pytest.mark.parametrize("series_first", [False, True])
|
|
317
|
+
@pytest.mark.parametrize("conflicting", [False, True])
|
|
318
|
+
def test_concat_mixed_dimensions_columns_aligns_or_rejects_weights(
|
|
319
|
+
series_first, conflicting
|
|
320
|
+
):
|
|
321
|
+
frame = mdf.MicroDataFrame({"income": [10.0, 20.0]}, index=[7, 3], weights=[2, 5])
|
|
322
|
+
series = mdf.MicroSeries(
|
|
323
|
+
[30.0, 40.0],
|
|
324
|
+
index=[3, 7],
|
|
325
|
+
name="other",
|
|
326
|
+
weights=[50, 2] if conflicting else [5, 2],
|
|
327
|
+
)
|
|
328
|
+
parts = [series, frame] if series_first else [frame, series]
|
|
329
|
+
if conflicting:
|
|
330
|
+
with pytest.raises(ValueError, match="weights"):
|
|
331
|
+
pd.concat(parts, axis=1)
|
|
332
|
+
else:
|
|
333
|
+
result = pd.concat(parts, axis=1)
|
|
334
|
+
np.testing.assert_array_equal(
|
|
335
|
+
result.weights, [5, 2] if series_first else [2, 5]
|
|
336
|
+
)
|
|
337
|
+
assert result.income.sum() == 120 # 10 * 2 + 20 * 5.
|
|
338
|
+
assert result.other.sum() == 230 # 30 * 5 + 40 * 2.
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
@pytest.mark.parametrize("mapping", [False, True])
|
|
342
|
+
def test_series_to_dataframe_weights_follow_aligned_rows_and_explicit_override(mapping):
|
|
343
|
+
series = mdf.MicroSeries([10.0, 20.0], index=[7, 3], name="income", weights=[2, 5])
|
|
344
|
+
data = {"income": series} if mapping else series
|
|
345
|
+
result = mdf.MicroDataFrame(data, index=[3, 7])
|
|
346
|
+
np.testing.assert_array_equal(result.weights, [5, 2])
|
|
347
|
+
assert result.income.sum() == 120 # 20 * 5 + 10 * 2.
|
|
348
|
+
result.weights.iloc[0] = 100
|
|
349
|
+
np.testing.assert_array_equal(series.weights, [2, 5])
|
|
350
|
+
|
|
351
|
+
explicit = mdf.MicroDataFrame(data, index=[3, 7], weights=[11, 13])
|
|
352
|
+
np.testing.assert_array_equal(explicit.weights, [11, 13])
|
|
353
|
+
assert explicit.income.sum() == 350 # 20 * 11 + 10 * 13.
|
|
354
|
+
with pytest.raises(ValueError, match="weights"):
|
|
355
|
+
mdf.MicroDataFrame(data, index=[3, 99])
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
359
|
+
@pytest.mark.parametrize("coincident_labels", [False, True])
|
|
360
|
+
def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_labels):
|
|
361
|
+
if coincident_labels:
|
|
362
|
+
frame = mdf.MicroDataFrame(
|
|
363
|
+
{"x": [10.0, 20.0], "y": [4.0, 8.0]},
|
|
364
|
+
index=["x", "y"],
|
|
365
|
+
weights=[2, 3],
|
|
366
|
+
)
|
|
367
|
+
# Sample covariance divides centered cross-products by n - 1.
|
|
368
|
+
covariance = [[50.0, 20.0], [20.0, 8.0]]
|
|
369
|
+
correlation = [[1.0, 1.0], [1.0, 1.0]]
|
|
370
|
+
else:
|
|
371
|
+
frame = mdf.MicroDataFrame(
|
|
372
|
+
{"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]},
|
|
373
|
+
weights=[2, 3, 5],
|
|
374
|
+
)
|
|
375
|
+
# Centered x = [-10, 0, 10], y = [-2, 2, 0]; n - 1 = 2.
|
|
376
|
+
covariance = [[100.0, 10.0], [10.0, 4.0]]
|
|
377
|
+
correlation = [[1.0, 0.5], [0.5, 1.0]]
|
|
378
|
+
expected = pd.DataFrame(
|
|
379
|
+
covariance if method == "cov" else correlation,
|
|
380
|
+
index=frame.columns,
|
|
381
|
+
columns=frame.columns,
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
result = getattr(frame, method)()
|
|
385
|
+
|
|
386
|
+
assert type(result) is pd.DataFrame
|
|
387
|
+
pd.testing.assert_frame_equal(result, expected)
|
|
388
|
+
# Chaining a sum must not apply observation weights to column summaries.
|
|
389
|
+
pd.testing.assert_series_equal(result.sum(), expected.sum())
|
|
390
|
+
assert not hasattr(result, "weights")
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
@pytest.mark.parametrize(
|
|
394
|
+
"method,args,kwargs",
|
|
395
|
+
[
|
|
396
|
+
("cov", (), {}),
|
|
397
|
+
("cov", (2, 0), {}),
|
|
398
|
+
("cov", (), {"min_periods": 4, "ddof": 2}),
|
|
399
|
+
("cov", (), {"min_periods": 2, "ddof": 0, "numeric_only": True}),
|
|
400
|
+
("corr", (), {}),
|
|
401
|
+
("corr", ("pearson", 2, True), {}),
|
|
402
|
+
("corr", (), {"method": "spearman", "min_periods": 2}),
|
|
403
|
+
("corr", (), {"min_periods": 4}),
|
|
404
|
+
("corr", (), {"method": lambda x, y: np.dot(x, y), "min_periods": 2}),
|
|
405
|
+
],
|
|
406
|
+
)
|
|
407
|
+
@pytest.mark.parametrize("missing", [False, True])
|
|
408
|
+
def test_dataframe_matrix_summaries_preserve_pandas_arguments(
|
|
409
|
+
method, args, kwargs, missing
|
|
410
|
+
):
|
|
411
|
+
data = {
|
|
412
|
+
"x": [10.0, 20.0, 30.0, 40.0],
|
|
413
|
+
"y": [4.0, 8.0, np.nan if missing else 6.0, 9.0],
|
|
414
|
+
"flag": [True, False, True, True],
|
|
415
|
+
}
|
|
416
|
+
frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=[2, 3, 5, 7])
|
|
417
|
+
expected = getattr(pd.DataFrame(data, index=frame.index), method)(*args, **kwargs)
|
|
418
|
+
|
|
419
|
+
result = getattr(frame, method)(*args, **kwargs)
|
|
420
|
+
|
|
421
|
+
assert type(result) is pd.DataFrame
|
|
422
|
+
pd.testing.assert_frame_equal(result, expected)
|
|
423
|
+
pd.testing.assert_series_equal(result.sum(), expected.sum())
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
427
|
+
def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method):
|
|
428
|
+
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0], "label": ["a", "b", "c"]}
|
|
429
|
+
frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
|
|
430
|
+
plain = pd.DataFrame(data)
|
|
431
|
+
expected = getattr(plain, method)(numeric_only=True)
|
|
432
|
+
|
|
433
|
+
result = getattr(frame, method)(numeric_only=True)
|
|
434
|
+
|
|
435
|
+
assert type(result) is pd.DataFrame
|
|
436
|
+
pd.testing.assert_frame_equal(result, expected)
|
|
437
|
+
for kwargs in [{}, {"numeric_only": False}]:
|
|
438
|
+
with pytest.raises((TypeError, ValueError)) as pandas_error:
|
|
439
|
+
getattr(plain, method)(**kwargs)
|
|
440
|
+
with pytest.raises(type(pandas_error.value)) as microdf_error:
|
|
441
|
+
getattr(frame, method)(**kwargs)
|
|
442
|
+
assert str(microdf_error.value) == str(pandas_error.value)
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def test_dataframe_correlation_preserves_optional_kendall_support():
|
|
446
|
+
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
|
|
447
|
+
frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
|
|
448
|
+
try:
|
|
449
|
+
expected = pd.DataFrame(data).corr(method="kendall")
|
|
450
|
+
except ImportError as pandas_error:
|
|
451
|
+
# Kendall requires scipy; delegation preserves pandas' dependency error.
|
|
452
|
+
with pytest.raises(type(pandas_error)) as microdf_error:
|
|
453
|
+
frame.corr(method="kendall")
|
|
454
|
+
assert str(microdf_error.value) == str(pandas_error)
|
|
455
|
+
else:
|
|
456
|
+
result = frame.corr(method="kendall")
|
|
457
|
+
assert type(result) is pd.DataFrame
|
|
458
|
+
pd.testing.assert_frame_equal(result, expected)
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
462
|
+
def test_dataframe_matrix_summaries_preserve_pandas_metadata(method):
|
|
463
|
+
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
|
|
464
|
+
frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
|
|
465
|
+
plain = pd.DataFrame(data)
|
|
466
|
+
for source in [frame, plain]:
|
|
467
|
+
source.attrs = {"survey": {"year": 2026}}
|
|
468
|
+
source.flags.allows_duplicate_labels = False
|
|
469
|
+
source.columns.name = "measure"
|
|
470
|
+
expected = getattr(plain, method)()
|
|
471
|
+
|
|
472
|
+
result = getattr(frame, method)()
|
|
473
|
+
|
|
474
|
+
assert type(result) is pd.DataFrame
|
|
475
|
+
pd.testing.assert_frame_equal(result, expected)
|
|
476
|
+
assert result.attrs == expected.attrs
|
|
477
|
+
result.attrs["survey"]["year"] = 2025
|
|
478
|
+
assert frame.attrs["survey"]["year"] == 2026
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
microdf/__init__.py,sha256=sddmTcTZFSb1fjZkoLUR5TK7r5PEld4v2F92xBI5Sxs,641
|
|
2
|
-
microdf/
|
|
3
|
-
microdf/
|
|
2
|
+
microdf/_weights.py,sha256=uBOcTZGmVJlzi27sSAALSTb4V9j56tXuaBfJB83x7j0,10303
|
|
3
|
+
microdf/microdataframe.py,sha256=yJNtsm-LNyB9_nyDUpV58nUQ3iZWWT0cZsD9Eu1j22w,39711
|
|
4
|
+
microdf/microseries.py,sha256=H_-y1UOFhbZYxxvf7pKuxP-psK5IBfuLSkyQ6C9qJts,45399
|
|
4
5
|
microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
|
|
5
6
|
microdf/tests/test_aggregation_errors.py,sha256=9jJDiEyxMb2z1Zmj-o8AHDNp8LOputbEkAMefifnWaE,2013
|
|
6
7
|
microdf/tests/test_dataframe_weight_storage.py,sha256=ngIsWa_QcBnnaLpAyIZhTgMxjv_7cK8Nbf7f4n29R4Q,1975
|
|
@@ -11,9 +12,10 @@ microdf/tests/test_quantile_missing_values.py,sha256=lfntDvV2q7KH_CPVrXFJSQFoaGl
|
|
|
11
12
|
microdf/tests/test_serialization.py,sha256=a7pHL2hNiG5iJjRtfx3C1BCmgOZouAekiOwUxouAPfo,5083
|
|
12
13
|
microdf/tests/test_sum_axes.py,sha256=N05ocwI5lLv2OgoaovRIqFIae-70356kZemRRet0ac8,8521
|
|
13
14
|
microdf/tests/test_version_metadata.py,sha256=M1EabzHLKZZw3Djd6Zu2UuMQtDLV6rZ1zDrOU7W_jf0,227
|
|
15
|
+
microdf/tests/test_weight_propagation.py,sha256=3odbufFZ2o1rnyjx6PJ7kn1TPO5K2ho3RErczI7mqO4,18761
|
|
14
16
|
microdf/tests/test_weighted_cov_corr.py,sha256=LTnFhMWnb28f7L_9kOhPiV5UlLMAUbC9OlLIaZ1XgLs,13954
|
|
15
|
-
microdf_python-1.4.
|
|
16
|
-
microdf_python-1.4.
|
|
17
|
-
microdf_python-1.4.
|
|
18
|
-
microdf_python-1.4.
|
|
19
|
-
microdf_python-1.4.
|
|
17
|
+
microdf_python-1.4.1.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
|
|
18
|
+
microdf_python-1.4.1.dist-info/METADATA,sha256=LVzReGsomnwivms2XGOPPZiYBWU_VLsQSuWGkfiNm24,2305
|
|
19
|
+
microdf_python-1.4.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
20
|
+
microdf_python-1.4.1.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
|
|
21
|
+
microdf_python-1.4.1.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|