microdf-python 1.3.9__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {microdf_python-1.3.9/microdf_python.egg-info → microdf_python-1.4.0}/PKG-INFO +1 -1
  2. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/microdataframe.py +48 -1
  3. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/microseries.py +201 -31
  4. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_microseries_dataframe.py +0 -22
  5. microdf_python-1.4.0/microdf/tests/test_sum_axes.py +216 -0
  6. microdf_python-1.4.0/microdf/tests/test_weighted_cov_corr.py +343 -0
  7. {microdf_python-1.3.9 → microdf_python-1.4.0/microdf_python.egg-info}/PKG-INFO +1 -1
  8. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf_python.egg-info/SOURCES.txt +2 -0
  9. {microdf_python-1.3.9 → microdf_python-1.4.0}/pyproject.toml +1 -1
  10. {microdf_python-1.3.9 → microdf_python-1.4.0}/LICENSE +0 -0
  11. {microdf_python-1.3.9 → microdf_python-1.4.0}/README.md +0 -0
  12. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/__init__.py +0 -0
  13. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/conftest.py +0 -0
  14. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_aggregation_errors.py +0 -0
  15. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_dataframe_weight_storage.py +0 -0
  16. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_nullify_weights_index.py +0 -0
  17. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_pandas3_compatibility.py +0 -0
  18. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_quantile_missing_values.py +0 -0
  19. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_serialization.py +0 -0
  20. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf/tests/test_version_metadata.py +0 -0
  21. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf_python.egg-info/dependency_links.txt +0 -0
  22. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf_python.egg-info/requires.txt +0 -0
  23. {microdf_python-1.3.9 → microdf_python-1.4.0}/microdf_python.egg-info/top_level.txt +0 -0
  24. {microdf_python-1.3.9 → microdf_python-1.4.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.3.9
3
+ Version: 1.4.0
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -161,13 +161,60 @@ class MicroDataFrame(pd.DataFrame):
161
161
  def override_df_functions(self) -> None:
162
162
  """Override DataFrame functions to work with weighted operations."""
163
163
  for name in MicroSeries.FUNCTIONS:
164
- if name in MicroSeries.SCALAR_FUNCTIONS:
164
+ if name == "sum":
165
+ # Sum has its own axis-aware signature and result types.
166
+ continue
167
+ elif name in MicroSeries.SCALAR_FUNCTIONS:
165
168
  setattr(self, name, self._create_scalar_function(name))
166
169
  elif name in MicroSeries.VECTOR_FUNCTIONS:
167
170
  setattr(self, name, self._create_vector_function(name))
168
171
  elif name in MicroSeries.AGNOSTIC_FUNCTIONS:
169
172
  setattr(self, name, self._create_agnostic_function(name))
170
173
 
174
+ def sum(
175
+ self,
176
+ axis: Optional[Union[int, str]] = 0,
177
+ skipna: bool = True,
178
+ numeric_only: bool = False,
179
+ min_count: int = 0,
180
+ **kwargs,
181
+ ) -> Union[pd.Series, MicroSeries, float]:
182
+ """Sum numeric columns, weighting reductions across observations.
183
+
184
+ Column sums (axis=0 or 'index') apply observation weights and return a
185
+ plain Series. Row sums (axis=1 or 'columns') do not multiply row values
186
+ by weights; they return a MicroSeries with an independent copy of the
187
+ original weights for subsequent weighted aggregation.
188
+
189
+ Non-numeric columns are excluded, matching other MicroDataFrame
190
+ aggregations. skipna and min_count follow pandas sum semantics.
191
+ Explicit axis=None follows the installed pandas version: column sums in
192
+ pandas 2, and a weighted total over both axes in pandas 3.
193
+ """
194
+ axis_number = None if axis is None else self._get_axis_number(axis)
195
+ values = pd.DataFrame(self)
196
+ numeric_columns = [
197
+ pd.api.types.is_numeric_dtype(dtype) for dtype in values.dtypes
198
+ ]
199
+ values = values.iloc[:, numeric_columns]
200
+ if axis_number != 1 and self.weights is not None:
201
+ values = values.mul(self.weights, axis=0)
202
+ result = values.sum(
203
+ axis=axis,
204
+ skipna=skipna,
205
+ numeric_only=numeric_only,
206
+ min_count=min_count,
207
+ **kwargs,
208
+ )
209
+ if axis_number == 1:
210
+ weights = (
211
+ self.weights.copy()
212
+ if self.weights is not None
213
+ else pd.Series(1.0, index=self.index)
214
+ )
215
+ return MicroSeries(result, weights=weights)
216
+ return result
217
+
171
218
  def _create_scalar_function(self, name: str) -> Callable:
172
219
  """Create a scalar function that returns a Series of results.
173
220
 
@@ -9,6 +9,41 @@ import pandas as pd
9
9
  logger = logging.getLogger(__name__)
10
10
 
11
11
 
12
+ def _weighted_centered_vector(
13
+ values: np.ndarray, weights: np.ndarray
14
+ ) -> tuple[np.ndarray, int]:
15
+ """Return scaled sqrt-weighted deviations and their power-of-two
16
+ exponent."""
17
+ # Center relative to a maximum-weight observation: shifting by a low-weight
18
+ # extreme could erase differences among the influential observations.
19
+ # A relative mean also preserves nearby values at a large common offset.
20
+ reference = values[np.argmax(weights)]
21
+ with np.errstate(over="ignore"):
22
+ shifted = values - reference
23
+ exponent = 0
24
+ if np.isinf(shifted).any():
25
+ # Opposite finite extremes can overflow their difference. Halving is
26
+ # exact for those values; restore that factor in the final exponent.
27
+ shifted = values / 2 - reference / 2
28
+ exponent = 1
29
+ magnitude = np.max(np.abs(shifted))
30
+ if magnitude == 0:
31
+ return shifted, 0
32
+ _, shift = np.frexp(magnitude)
33
+ shifted = np.ldexp(shifted, -shift)
34
+ # Raise tiny mean weights by an exact common power of two so products
35
+ # with the scaled deviations do not underflow. Never scale down: that
36
+ # could discard small weights when frequencies span a wide range.
37
+ _, mean_weight_exponent = np.frexp(np.max(weights))
38
+ mean_weights = np.ldexp(weights, -min(int(mean_weight_exponent), 0))
39
+ shifted -= np.average(shifted, weights=mean_weights)
40
+ # Weight each vector before taking products, then scale again so squared
41
+ # deviations never accumulate raw frequencies at the original value scale.
42
+ shifted *= np.sqrt(weights)
43
+ _, weight_shift = np.frexp(np.max(np.abs(shifted)))
44
+ return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)
45
+
46
+
12
47
  def _weighted_top_share(
13
48
  values: np.ndarray, weights: np.ndarray, top_x_pct: float
14
49
  ) -> float:
@@ -187,16 +222,38 @@ class MicroSeries(pd.Series):
187
222
  :returns: A Series multiplying the MicroSeries by its weight.
188
223
  :rtype: pd.Series
189
224
  """
190
- return self.multiply(self.weights)
225
+ return pd.Series(self, copy=False).multiply(self.weights)
191
226
 
192
227
  @scalar_function
193
- def sum(self) -> float:
228
+ def sum(
229
+ self,
230
+ axis: Optional[Union[int, str]] = 0,
231
+ skipna: bool = True,
232
+ numeric_only: bool = False,
233
+ min_count: int = 0,
234
+ **kwargs,
235
+ ) -> float:
194
236
  """Calculates the weighted sum of the MicroSeries.
195
237
 
238
+ axis may be 0, 'index' or None, as for pandas Series.sum. skipna,
239
+ numeric_only and min_count are applied to the weighted values;
240
+ min_count counts valid observations, not the sum of their weights.
241
+
196
242
  :returns: The weighted sum.
197
243
  :rtype: float
198
244
  """
199
- return self.multiply(self.weights).sum()
245
+ # Keep the intermediate unweighted so subclass constructors cannot
246
+ # apply observation weights a second time during the final reduction.
247
+ values = pd.Series(self)
248
+ if not self.empty:
249
+ values = values.multiply(self.weights)
250
+ return values.sum(
251
+ axis=axis,
252
+ skipna=skipna,
253
+ numeric_only=numeric_only,
254
+ min_count=min_count,
255
+ **kwargs,
256
+ )
200
257
 
201
258
  @scalar_function
202
259
  def count(self, skipna: bool = True) -> float:
@@ -293,39 +350,152 @@ class MicroSeries(pd.Series):
293
350
  v = self._weighted_variance(ddof=ddof, skipna=skipna)
294
351
  return float(np.sqrt(v)) if np.isfinite(v) else v
295
352
 
296
- def cov(self, other, *args, **kwargs):
297
- """Pandas ``cov`` — **unweighted**.
353
+ def _weighted_pair(
354
+ self,
355
+ other: pd.Series,
356
+ min_periods: Optional[int],
357
+ ddof: int,
358
+ skipna: bool,
359
+ ) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
360
+ """Align usable paired observations with their left weights."""
361
+ if not isinstance(other, pd.Series):
362
+ raise TypeError("other must be a pandas Series or MicroSeries")
363
+ if not isinstance(ddof, (int, np.integer)):
364
+ raise TypeError("ddof must be an integer")
365
+ if min_periods is None:
366
+ min_periods = 1
367
+ if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
368
+ raise ValueError("min_periods must be a nonnegative integer")
369
+ if len(self) == 0 or len(other) == 0:
370
+ return None
371
+
372
+ # Align left row positions so values and weights undergo exactly the
373
+ # same join, including pandas' expansion of duplicate index labels.
374
+ positions = pd.Series(np.arange(len(self)), index=self.index)
375
+ positions, right = positions.align(pd.Series(other), join="inner")
376
+ positions = positions.to_numpy(dtype=int)
377
+ x = (
378
+ pd.Series(self._values)
379
+ .iloc[positions]
380
+ .to_numpy(dtype=float, na_value=np.nan)
381
+ )
382
+ y = right.to_numpy(dtype=float, na_value=np.nan)
383
+ weights = np.asarray(self.weights, dtype=float)[positions]
384
+ if not np.isfinite(weights).all() or (weights < 0).any():
385
+ raise ValueError("frequency weights must be finite and nonnegative")
386
+
387
+ # Zero frequency means the row is absent, including for skipna=False.
388
+ positive = weights > 0
389
+ x, y, weights = x[positive], y[positive], weights[positive]
390
+ missing = np.isnan(x) | np.isnan(y)
391
+ if not skipna and missing.any():
392
+ return None
393
+ x, y, weights = x[~missing], y[~missing], weights[~missing]
394
+ total_weight = weights.sum()
395
+ if not np.isfinite(total_weight):
396
+ raise ValueError("the sum of frequency weights must be finite")
397
+ if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
398
+ return None
399
+ return (
400
+ x,
401
+ y,
402
+ weights,
403
+ float(total_weight - ddof),
404
+ )
298
405
 
299
- MicroSeries does not yet compute weighted covariance. Emits a
300
- ``UserWarning`` so callers aren't silently given an unweighted number
301
- after ``.sum()`` and ``.mean()`` worked as expected. See issue tracker
302
- for a weighted implementation.
406
+ def cov(
407
+ self,
408
+ other: pd.Series,
409
+ min_periods: Optional[int] = None,
410
+ ddof: int = 1,
411
+ *,
412
+ skipna: bool = True,
413
+ ) -> float:
414
+ """Calculate frequency-weighted covariance with another Series.
415
+
416
+ Observations align by index as in pandas, including its duplicate-
417
+ label join behavior. Only this Series' weights are used; weights on
418
+ another MicroSeries are ignored. Each aligned left weight must be
419
+ finite and nonnegative. Zero-weight rows are omitted.
420
+
421
+ Uses ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) - ddof)``.
422
+ Integer weights therefore match covariance on the replicated sample.
423
+ Missing values are removed pairwise before computing both means.
424
+
425
+ :param other: A pandas Series or MicroSeries to align by index.
426
+ :param min_periods: Minimum usable aligned row pairs, not the sum of
427
+ frequency weights. Defaults to 1.
428
+ :param ddof: Degrees of freedom subtracted from the weight total.
429
+ :param skipna: Drop pairs with a missing value. If False, any missing
430
+ value in a positive-weight aligned pair produces NaN.
431
+ :returns: Weighted covariance, or NaN for an empty or insufficient
432
+ sample (including a weight total no greater than ddof).
303
433
  """
304
- warnings.warn(
305
- "MicroSeries.cov() falls through to pandas and is "
306
- "unweighted. Use MicroSeries.var()/std() for weighted "
307
- "second moments, or compute covariance manually with the "
308
- "weights.",
309
- UserWarning,
310
- stacklevel=2,
434
+ pair = self._weighted_pair(other, min_periods, ddof, skipna)
435
+ if pair is None:
436
+ return np.nan
437
+ x, y, weights, denominator = pair
438
+ if not np.isfinite(x).all() or not np.isfinite(y).all():
439
+ return np.nan
440
+ x, x_exponent = _weighted_centered_vector(x, weights)
441
+ y, y_exponent = _weighted_centered_vector(y, weights)
442
+ # Combine exponents only after dividing out sum(weights) - ddof.
443
+ # Neither the original squared scale nor raw weighted sum need fit.
444
+ denominator, denominator_exponent = np.frexp(denominator)
445
+ return float(
446
+ np.ldexp(
447
+ np.sum(x * y) / denominator,
448
+ x_exponent + y_exponent - int(denominator_exponent),
449
+ )
311
450
  )
312
- return super().cov(other, *args, **kwargs)
313
451
 
314
- def corr(self, other, *args, **kwargs):
315
- """Pandas ``corr`` — **unweighted**.
316
-
317
- MicroSeries does not yet compute weighted correlation. Emits a
318
- ``UserWarning`` so callers aren't silently given an unweighted number.
319
- See issue tracker for a weighted implementation.
452
+ def corr(
453
+ self,
454
+ other: pd.Series,
455
+ method: str = "pearson",
456
+ min_periods: Optional[int] = None,
457
+ *,
458
+ ddof: int = 1,
459
+ skipna: bool = True,
460
+ ) -> float:
461
+ """Calculate frequency-weighted Pearson correlation.
462
+
463
+ Uses the same aligned pairs and left Series weights for covariance and
464
+ both variances. Weights on another MicroSeries are ignored. Weights
465
+ must be finite and nonnegative; zero-weight rows are omitted. Other
466
+ correlation methods, including callables, are unsupported.
467
+
468
+ :param other: A pandas Series or MicroSeries to align by index.
469
+ :param method: Only "pearson" is supported.
470
+ :param min_periods: Minimum usable aligned row pairs, not frequency
471
+ weight total. Defaults to 1.
472
+ :param ddof: Degrees of freedom for all three moments. It cancels from
473
+ the correlation but the weight total must exceed it.
474
+ :param skipna: Drop pairs with a missing value. If False, any missing
475
+ value in a positive-weight aligned pair produces NaN.
476
+ :returns: Weighted correlation, or NaN for an empty, insufficient, or
477
+ constant sample.
320
478
  """
321
- warnings.warn(
322
- "MicroSeries.corr() falls through to pandas and is "
323
- "unweighted. Compute correlation manually with the weights "
324
- "if you need the survey-weighted value.",
325
- UserWarning,
326
- stacklevel=2,
327
- )
328
- return super().corr(other, *args, **kwargs)
479
+ if method != "pearson":
480
+ raise ValueError("weighted correlation only supports method='pearson'")
481
+ pair = self._weighted_pair(other, min_periods, ddof, skipna)
482
+ if pair is None:
483
+ return np.nan
484
+ x, y, weights, _ = pair
485
+ if not np.isfinite(x).all() or not np.isfinite(y).all():
486
+ return np.nan
487
+ # A weighted mean can round away from identical decimal inputs.
488
+ # Check the retained observations exactly before subtracting it.
489
+ if (x == x[0]).all() or (y == y[0]).all():
490
+ return np.nan
491
+ x, _ = _weighted_centered_vector(x, weights)
492
+ y, _ = _weighted_centered_vector(y, weights)
493
+ x_ss = np.sum(x * x)
494
+ y_ss = np.sum(y * y)
495
+ if x_ss == 0 or y_ss == 0:
496
+ return np.nan
497
+ result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
498
+ return float(np.clip(result, -1.0, 1.0))
329
499
 
330
500
  def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
331
501
  """Calculates weighted quantiles of the MicroSeries.
@@ -744,28 +744,6 @@ def test_std_var_are_weighted() -> None:
744
744
  )
745
745
 
746
746
 
747
- def test_cov_corr_warn_when_fallthrough() -> None:
748
- """Regression: cov/corr silently returned unweighted pandas values.
749
-
750
- They still fall through to pandas (a weighted impl is a separate issue) but
751
- now emit a UserWarning so callers aren't misled.
752
- """
753
- s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
754
- s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1])
755
-
756
- with warnings.catch_warnings(record=True) as w:
757
- warnings.simplefilter("always")
758
- _ = s1.cov(s2)
759
- msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
760
- assert any("unweighted" in m.lower() for m in msgs)
761
-
762
- with warnings.catch_warnings(record=True) as w:
763
- warnings.simplefilter("always")
764
- _ = s1.corr(s2)
765
- msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
766
- assert any("unweighted" in m.lower() for m in msgs)
767
-
768
-
769
747
  def test_count_skips_nan_by_default() -> None:
770
748
  """Regression: ``count()`` included NaN-row weight, contrary to pandas.
771
749
 
@@ -0,0 +1,216 @@
1
+ import warnings
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+ import pytest
6
+
7
+ import microdf as mdf
8
+
9
+
10
+ def test_sum_axis_1() -> None:
11
+ # Test basic row-wise sum
12
+ df = mdf.MicroDataFrame(
13
+ {"A": [1, 2, 3], "B": [4, 5, 6], "C": [7, 8, 9]},
14
+ weights=[0.5, 1.0, 2.0],
15
+ )
16
+
17
+ # Row-wise sum (axis=1) should not use weights
18
+ row_sums = df.sum(axis=1)
19
+ expected = pd.Series([12, 15, 18], index=df.index) # 1+4+7, 2+5+8, 3+6+9
20
+ pd.testing.assert_series_equal(pd.Series(row_sums), expected)
21
+
22
+ # Column-wise sum (axis=0) should use weights
23
+ col_sums = df.sum(axis=0)
24
+ expected_weighted = pd.Series(
25
+ {
26
+ "A": 1 * 0.5 + 2 * 1.0 + 3 * 2.0, # 8.5
27
+ "B": 4 * 0.5 + 5 * 1.0 + 6 * 2.0, # 19.0
28
+ "C": 7 * 0.5 + 8 * 1.0 + 9 * 2.0, # 29.5
29
+ }
30
+ )
31
+ pd.testing.assert_series_equal(col_sums, expected_weighted)
32
+
33
+ # Test with mixed types (non-numeric columns should be ignored)
34
+ df_mixed = mdf.MicroDataFrame(
35
+ {"A": [1, 2, 3], "B": [4, 5, 6], "text": ["a", "b", "c"]},
36
+ weights=[1, 1, 1],
37
+ )
38
+
39
+ row_sums_mixed = df_mixed.sum(axis=1)
40
+ expected_mixed = pd.Series([5, 7, 9], index=df_mixed.index) # Only A+B
41
+ pd.testing.assert_series_equal(pd.Series(row_sums_mixed), expected_mixed)
42
+
43
+ # Test with axis='columns' (string form)
44
+ row_sums_str = df.sum(axis="columns")
45
+ pd.testing.assert_series_equal(pd.Series(row_sums_str), expected)
46
+
47
+ # Test with additional parameters
48
+ df_with_nan = mdf.MicroDataFrame(
49
+ {"A": [1, np.nan, 3], "B": [4, 5, 6], "C": [7, 8, np.nan]},
50
+ weights=[1, 1, 1],
51
+ )
52
+
53
+ # skipna=True (default)
54
+ row_sums_skipna = df_with_nan.sum(axis=1)
55
+ expected_skipna = pd.Series([12.0, 13.0, 9.0]) # NaN values skipped
56
+ pd.testing.assert_series_equal(pd.Series(row_sums_skipna), expected_skipna)
57
+
58
+ # skipna=False
59
+ row_sums_no_skipna = df_with_nan.sum(axis=1, skipna=False)
60
+ expected_no_skipna = pd.Series([12.0, np.nan, np.nan]) # NaN propagates
61
+ pd.testing.assert_series_equal(pd.Series(row_sums_no_skipna), expected_no_skipna)
62
+
63
+ # Test min_count parameter
64
+ row_sums_min_count = df_with_nan.sum(axis=1, min_count=3)
65
+ expected_min_count = pd.Series(
66
+ [12.0, np.nan, np.nan]
67
+ ) # Row 1 and 2 have < 3 non-NA values
68
+ pd.testing.assert_series_equal(pd.Series(row_sums_min_count), expected_min_count)
69
+
70
+
71
+ @pytest.mark.parametrize("axis", [0, "index", 1, "columns"])
72
+ @pytest.mark.parametrize("positional", [False, True])
73
+ def test_sum_binds_positional_and_keyword_axes(axis, positional):
74
+ frame = mdf.MicroDataFrame(
75
+ {"a": [1, 2, 3], "b": [4, 5, 6]}, index=[7, 8, 9], weights=[1, 2, 3]
76
+ )
77
+ result = frame.sum(axis) if positional else frame.sum(axis=axis)
78
+ if axis in (0, "index"):
79
+ assert type(result) is pd.Series
80
+ pd.testing.assert_series_equal(result, pd.Series({"a": 14.0, "b": 32.0}))
81
+ else:
82
+ assert isinstance(result, mdf.MicroSeries)
83
+ pd.testing.assert_series_equal(
84
+ pd.Series(result), pd.Series([5, 7, 9], index=frame.index)
85
+ )
86
+ pd.testing.assert_series_equal(result.weights, frame.weights)
87
+ # Row values are not weighted yet; subsequent aggregation is weighted.
88
+ assert result.sum() == 5 * 1 + 7 * 2 + 9 * 3
89
+ result.weights.iloc[0] = 100
90
+ assert frame.weights.iloc[0] == 1
91
+
92
+
93
+ @pytest.mark.parametrize("skipna,min_count", [(True, 0), (False, 0), (True, 3)])
94
+ @pytest.mark.parametrize("axis", [0, "index", None])
95
+ def test_weighted_column_sum_options(axis, skipna, min_count):
96
+ raw = pd.DataFrame({"a": [1.0, np.nan, 3.0], "b": [4.0, 5.0, 6.0]})
97
+ weights = pd.Series([1.0, 2.0, 3.0])
98
+ frame = mdf.MicroDataFrame(raw, weights=weights)
99
+ # Native sum defines version-specific axis=None and missing-value behavior.
100
+ # The independently weighted entries are [1, NaN, 9] and [4, 10, 18].
101
+ expected_data = pd.DataFrame({"a": [1.0, np.nan, 9.0], "b": [4.0, 10.0, 18.0]})
102
+ with warnings.catch_warnings():
103
+ warnings.simplefilter("ignore", FutureWarning)
104
+ expected = expected_data.sum(axis=axis, skipna=skipna, min_count=min_count)
105
+ actual = frame.sum(axis=axis, skipna=skipna, min_count=min_count)
106
+ if isinstance(expected, pd.Series):
107
+ assert type(actual) is pd.Series
108
+ pd.testing.assert_series_equal(actual, expected)
109
+ else:
110
+ np.testing.assert_allclose(actual, expected, equal_nan=True)
111
+
112
+
113
+ @pytest.mark.parametrize("axis", [None, 0, "index"])
114
+ @pytest.mark.parametrize("skipna,min_count", [(True, 0), (False, 0), (True, 3)])
115
+ def test_microseries_sum_options(axis, skipna, min_count):
116
+ series = mdf.MicroSeries([1.0, np.nan, 3.0], index=[7, 8, 9], weights=[1, 2, 3])
117
+ expected = pd.Series([1.0, np.nan, 9.0]).sum(
118
+ axis=axis, skipna=skipna, min_count=min_count
119
+ )
120
+ np.testing.assert_allclose(
121
+ series.sum(axis, skipna=skipna, min_count=min_count), expected, equal_nan=True
122
+ )
123
+
124
+
125
+ @pytest.mark.parametrize("min_count,expected", [(0, 0.0), (1, np.nan)])
126
+ def test_empty_numeric_row_sum_identity(min_count, expected):
127
+ frame = mdf.MicroDataFrame({"text": ["a", "b"]}, index=[7, 8], weights=[2, 3])
128
+ actual = frame.sum(axis=1, min_count=min_count)
129
+ assert isinstance(actual, mdf.MicroSeries)
130
+ pd.testing.assert_series_equal(
131
+ pd.Series(actual), pd.Series([expected, expected], index=frame.index)
132
+ )
133
+ pd.testing.assert_series_equal(actual.weights, frame.weights)
134
+
135
+
136
+ def test_sum_rejects_invalid_arguments():
137
+ frame = mdf.MicroDataFrame({"a": [1, 2]}, weights=[1, 2])
138
+ with pytest.raises(TypeError):
139
+ frame.sum(1, axis=0)
140
+ with pytest.raises(TypeError):
141
+ frame.sum(bogus=True)
142
+ with pytest.raises(ValueError):
143
+ frame.sum(axis=2)
144
+ with pytest.raises(ValueError):
145
+ frame["a"].sum(axis=1)
146
+
147
+
148
+ def test_sum_handles_boolean_and_nullable_numeric_columns():
149
+ raw = pd.DataFrame(
150
+ {
151
+ "count": pd.Series([1, None, 3], dtype="Int64"),
152
+ "flag": pd.Series([True, False, True], dtype="boolean"),
153
+ "text": ["a", "b", "c"],
154
+ }
155
+ )
156
+ frame = mdf.MicroDataFrame(raw, weights=[1, 2, 3])
157
+ expected = raw[["count", "flag"]].sum(axis=1)
158
+ pd.testing.assert_series_equal(pd.Series(frame.sum(1)), expected)
159
+ totals = frame.sum(numeric_only=True)
160
+ assert list(totals.index) == ["count", "flag"]
161
+ assert totals["count"] == 10
162
+ assert totals["flag"] == 4
163
+
164
+
165
+ def test_sum_preserves_other_scalar_positional_arguments():
166
+ frame = mdf.MicroDataFrame({"a": [-1.0, 2.0, 3.0]}, weights=[1, 2, 3])
167
+ for method, argument in [
168
+ ("gini", "shift"),
169
+ ("top_x_pct_share", 0.25),
170
+ ("mean", False),
171
+ ("var", 0),
172
+ ]:
173
+ actual = getattr(frame, method)(argument)["a"]
174
+ expected = getattr(frame["a"], method)(argument)
175
+ assert actual == expected
176
+
177
+
178
+ @pytest.mark.parametrize("min_count,expected", [(0, 0.0), (1, np.nan)])
179
+ def test_sum_of_empty_inputs(min_count, expected):
180
+ frame = mdf.MicroDataFrame(pd.DataFrame({"a": pd.Series([], dtype=float)}))
181
+ row_sums = frame.sum(axis=1, min_count=min_count)
182
+ assert isinstance(row_sums, mdf.MicroSeries)
183
+ assert row_sums.empty
184
+ pd.testing.assert_series_equal(row_sums.weights, pd.Series([], dtype=float))
185
+ pd.testing.assert_series_equal(
186
+ frame.sum(min_count=min_count), pd.Series({"a": expected})
187
+ )
188
+ series = mdf.MicroSeries([], dtype=float)
189
+ np.testing.assert_allclose(
190
+ series.sum(min_count=min_count), expected, equal_nan=True
191
+ )
192
+ with pytest.raises(ValueError):
193
+ series.sum(axis=1)
194
+
195
+
196
+ @pytest.mark.parametrize("axis", [0, 1])
197
+ @pytest.mark.parametrize("mixed_dtypes", [False, True])
198
+ def test_sum_preserves_duplicate_numeric_column_labels(axis, mixed_dtypes):
199
+ if mixed_dtypes:
200
+ raw = pd.DataFrame([[1.0, "x", 4.0], [2.0, "y", 5.0]], columns=["a", "a", "a"])
201
+ else:
202
+ raw = pd.DataFrame([[1.0, 4.0], [2.0, 5.0]], columns=["a", "a"])
203
+ frame = mdf.MicroDataFrame(raw, weights=[2, 3])
204
+
205
+ result = frame.sum(axis)
206
+
207
+ if axis == 0:
208
+ assert type(result) is pd.Series
209
+ expected = pd.Series([8.0, 23.0], index=["a", "a"])
210
+ pd.testing.assert_series_equal(result, expected)
211
+ else:
212
+ assert isinstance(result, mdf.MicroSeries)
213
+ pd.testing.assert_series_equal(pd.Series(result), pd.Series([5.0, 7.0]))
214
+ pd.testing.assert_series_equal(result.weights, frame.weights)
215
+ assert result.sum() == 31.0
216
+ pd.testing.assert_frame_equal(pd.DataFrame(frame), raw)
@@ -0,0 +1,343 @@
1
+ import warnings
2
+ from decimal import Decimal, localcontext
3
+ from fractions import Fraction
4
+ from itertools import permutations
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ import pytest
9
+
10
+ import microdf as mdf
11
+
12
+
13
+ def replicated_moments(x, y, weights, ddof=1):
14
+ """Independent frequency-weight oracle: expand to an ordinary sample."""
15
+ repeated_x = np.repeat(np.asarray(x, dtype=float), weights)
16
+ repeated_y = np.repeat(np.asarray(y, dtype=float), weights)
17
+ return (
18
+ np.cov(repeated_x, repeated_y, ddof=ddof)[0, 1],
19
+ np.corrcoef(repeated_x, repeated_y)[0, 1],
20
+ )
21
+
22
+
23
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
24
+ def test_cov_corr_match_replicated_frequency_sample(ddof):
25
+ x, y, weights = [1, 4, 8], [5, 2, 9], [1, 3, 2]
26
+ left = mdf.MicroSeries(x, weights=weights)
27
+ right = pd.Series(y)
28
+ expected_cov, expected_corr = replicated_moments(x, y, weights, ddof)
29
+ with warnings.catch_warnings(record=True) as caught:
30
+ warnings.simplefilter("always")
31
+ assert left.cov(right, ddof=ddof) == pytest.approx(expected_cov)
32
+ assert left.corr(right, ddof=ddof) == pytest.approx(expected_corr)
33
+ assert not any("unweighted" in str(item.message).lower() for item in caught)
34
+
35
+
36
+ def test_cov_corr_align_indices_and_use_only_left_weights():
37
+ left = mdf.MicroSeries([1, 4, 8], index=["a", "b", "c"], weights=[1, 3, 2])
38
+ right = mdf.MicroSeries(
39
+ [9, 5, 2, 100], index=["c", "a", "b", "d"], weights=[99, 1, 1, 9]
40
+ )
41
+ expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2])
42
+ assert left.cov(right) == pytest.approx(expected_cov)
43
+ assert left.corr(right) == pytest.approx(expected_corr)
44
+ assert right.weights.tolist() == [99, 1, 1, 9]
45
+
46
+
47
+ def test_cov_corr_use_same_pairwise_nonmissing_sample():
48
+ left = mdf.MicroSeries(
49
+ [1, np.nan, 5, 9, 20], index=list("abcde"), weights=[1, 9, 2, 3, 7]
50
+ )
51
+ right = pd.Series([2, 4, np.nan, 8, 30], index=list("abcdf"))
52
+ expected_cov, expected_corr = replicated_moments([1, 9], [2, 8], [1, 3])
53
+ assert left.cov(right) == pytest.approx(expected_cov)
54
+ assert left.corr(right) == pytest.approx(expected_corr)
55
+ assert np.isnan(left.cov(right, skipna=False))
56
+ assert np.isnan(left.corr(right, skipna=False))
57
+
58
+
59
+ @pytest.mark.parametrize("same_index", [True, False])
60
+ def test_cov_corr_follow_pandas_duplicate_index_alignment(same_index):
61
+ left = mdf.MicroSeries([1, 4, 7], index=["a", "a", "b"], weights=[1, 3, 2])
62
+ if same_index:
63
+ right = pd.Series([2, 3, 8], index=["a", "a", "b"])
64
+ x, y, weights = [1, 4, 7], [2, 3, 8], [1, 3, 2]
65
+ else:
66
+ right = pd.Series([2, 5, 8], index=["a", "b", "b"])
67
+ # The shared a/b labels join, repeating the left row weight per pair.
68
+ x, y, weights = [1, 4, 7, 7], [2, 2, 5, 8], [1, 3, 2, 2]
69
+ expected_cov, expected_corr = replicated_moments(x, y, weights)
70
+ assert left.cov(right) == pytest.approx(expected_cov)
71
+ assert left.corr(right) == pytest.approx(expected_corr)
72
+
73
+
74
+ def test_zero_weight_rows_do_not_enter_pairwise_sample():
75
+ left = mdf.MicroSeries([1, np.nan, 5, 1000], weights=[2, 0, 1, 0])
76
+ right = pd.Series([3, np.nan, 9, -1000])
77
+ expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [2, 1])
78
+ assert left.cov(right, skipna=False) == pytest.approx(expected_cov)
79
+ assert left.corr(right, skipna=False) == pytest.approx(expected_corr)
80
+ assert np.isnan(left.cov(right, min_periods=3))
81
+ assert np.isnan(left.corr(right, min_periods=3))
82
+
83
+
84
+ def test_min_periods_counts_usable_rows_separately_from_frequency_weight():
85
+ left = mdf.MicroSeries([1, 5], weights=[10, 20])
86
+ right = pd.Series([3, 9])
87
+ expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [10, 20], ddof=0)
88
+ # Existing pandas positional arguments retain their order.
89
+ assert left.cov(right, 2, 0) == pytest.approx(expected_cov)
90
+ assert left.corr(right, "pearson", 2, ddof=0) == pytest.approx(expected_corr)
91
+ assert np.isnan(left.cov(right, min_periods=3))
92
+ assert np.isnan(left.corr(right, min_periods=3))
93
+
94
+
95
+ @pytest.mark.parametrize(
96
+ "values,other,weights",
97
+ [([], [], []), ([np.nan], [1], [3]), ([1], [2], [0]), ([1], [2], [1])],
98
+ )
99
+ def test_cov_corr_return_nan_when_sample_is_insufficient(values, other, weights):
100
+ left = mdf.MicroSeries(values, weights=weights, dtype=float)
101
+ right = pd.Series(other, dtype=float)
102
+ assert np.isnan(left.cov(right))
103
+ assert np.isnan(left.corr(right))
104
+
105
+
106
+ def test_cov_corr_no_index_overlap():
107
+ left = mdf.MicroSeries([1, 2], index=["a", "b"], weights=[1, 2])
108
+ right = pd.Series([3, 4], index=["c", "d"])
109
+ assert np.isnan(left.cov(right))
110
+ assert np.isnan(left.corr(right))
111
+
112
+
113
+ def test_cov_corr_constant_and_frequency_singleton():
114
+ left = mdf.MicroSeries([4, 4], weights=[2, 3])
115
+ right = pd.Series([1, 5])
116
+ assert left.cov(right) == 0
117
+ assert np.isnan(left.corr(right))
118
+ singleton = mdf.MicroSeries([4], weights=[3])
119
+ assert singleton.cov(pd.Series([2])) == 0
120
+ assert np.isnan(singleton.corr(pd.Series([2])))
121
+ assert np.isnan(singleton.cov(pd.Series([2]), ddof=3))
122
+ assert np.isnan(singleton.corr(pd.Series([2]), ddof=3))
123
+
124
+
125
+ def test_cov_corr_nullable_numeric_data():
126
+ left = mdf.MicroSeries(pd.Series([1, pd.NA, 4], dtype="Int64"), weights=[2, 9, 3])
127
+ right = pd.Series([3, 8, 7], dtype="Float64")
128
+ expected_cov, expected_corr = replicated_moments([1, 4], [3, 7], [2, 3])
129
+ assert left.cov(right) == pytest.approx(expected_cov)
130
+ assert left.corr(right) == pytest.approx(expected_corr)
131
+
132
+
133
+ @pytest.mark.parametrize("method", ["spearman", "kendall", lambda x, y: 1.0])
134
+ def test_non_pearson_methods_are_explicitly_unsupported(method):
135
+ left = mdf.MicroSeries([1, 2, 3], weights=[1, 2, 3])
136
+ with pytest.raises(ValueError, match="pearson"):
137
+ left.corr(pd.Series([4, 2, 5]), method=method)
138
+
139
+
140
+ @pytest.mark.parametrize("weights", [[1, -1], [1, np.nan], [1, np.inf]])
141
+ def test_cov_corr_reject_invalid_frequency_weights(weights):
142
+ left = mdf.MicroSeries([1, 2], weights=weights)
143
+ for method in (left.cov, left.corr):
144
+ with pytest.raises(ValueError, match="weights"):
145
+ method(pd.Series([2, 4]))
146
+
147
+
148
+ def test_binary_statistics_not_in_dataframe_aggregation_factories():
149
+ assert "cov" not in mdf.MicroSeries.FUNCTIONS
150
+ assert "corr" not in mdf.MicroSeries.FUNCTIONS
151
+
152
+
153
+ @pytest.mark.parametrize(
154
+ "x,y",
155
+ [([0.1, 0.1], [0, 1]), ([0, 1], [0.1, 0.1]), ([0.1, 0.1], [0.1, 0.1])],
156
+ )
157
+ @pytest.mark.parametrize("with_filtered_rows", [False, True])
158
+ def test_corr_exact_decimal_constants_return_nan(x, y, with_filtered_rows):
159
+ # Unequal weights can round the mean away from the identical 0.1 values.
160
+ # Constant detection must inspect usable observations before centering.
161
+ weights = [1, 2]
162
+ if with_filtered_rows:
163
+ x = x + [9, np.nan]
164
+ y = y + [7, 4]
165
+ weights = weights + [0, 3]
166
+ left = mdf.MicroSeries(x, weights=weights)
167
+ assert np.isnan(left.corr(pd.Series(y)))
168
+
169
+
170
+ def test_corr_does_not_treat_nearby_distinct_values_as_constant():
171
+ x = [0.1, np.nextafter(0.1, np.inf)]
172
+ left = mdf.MicroSeries(x, weights=[1, 2])
173
+ assert np.isfinite(left.corr(pd.Series([0, 1])))
174
+ assert np.isfinite(mdf.MicroSeries([0, 1], weights=[1, 2]).corr(pd.Series(x)))
175
+
176
+
177
+ def exact_weighted_moments(x, y, weights, ddof=1):
178
+ """Compute moments of the actual input floats with exact rational
179
+ arithmetic."""
180
+ x, y, weights = [
181
+ [Fraction(float(value)) for value in values] for values in (x, y, weights)
182
+ ]
183
+ total = sum(weights)
184
+ xmean = sum(w * value for w, value in zip(weights, x)) / total
185
+ ymean = sum(w * value for w, value in zip(weights, y)) / total
186
+ xy = sum(w * (a - xmean) * (b - ymean) for a, b, w in zip(x, y, weights))
187
+ xx = sum(w * (value - xmean) ** 2 for value, w in zip(x, weights))
188
+ yy = sum(w * (value - ymean) ** 2 for value, w in zip(y, weights))
189
+ with localcontext() as context:
190
+ context.prec = 100
191
+ product = xx * yy
192
+ correlation = (Decimal(xy.numerator) / Decimal(xy.denominator)) / (
193
+ Decimal(product.numerator) / Decimal(product.denominator)
194
+ ).sqrt()
195
+ return float(xy / (total - ddof)), float(correlation)
196
+
197
+
198
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
199
+ @pytest.mark.parametrize("swap", [False, True])
200
+ def test_cov_corr_preserve_small_differences_at_large_offsets(ddof, swap):
201
+ # Two distinct points are perfectly linear even two float steps apart.
202
+ x = np.array([1e12 - 2**-13, 1e12 + 2**-13])
203
+ y = np.array([0.0, 1.0])
204
+ if swap:
205
+ x, y = y, x
206
+ weights = [1, 2]
207
+ expected = exact_weighted_moments(x, y, weights, ddof)
208
+ assert expected[1] == 1.0
209
+ for shifted_x, shifted_y in [(x, y), (x - x[0], y - y[0])]:
210
+ left = mdf.MicroSeries(shifted_x, weights=weights)
211
+ right = pd.Series(shifted_y)
212
+ np.testing.assert_allclose(
213
+ [left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)],
214
+ expected,
215
+ rtol=2e-15,
216
+ atol=0,
217
+ )
218
+
219
+
220
+ @pytest.mark.parametrize("frequency", [1, 1_000_000])
221
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
222
+ def test_cov_corr_large_finite_values_do_not_overflow_raw_frequencies(frequency, ddof):
223
+ x = np.array([1.0, 2.0, 3.0]) * 1e153
224
+ weights = [frequency] * 3
225
+ expected = exact_weighted_moments(x, x, weights, ddof)
226
+ assert np.isfinite(expected).all()
227
+ assert expected[1] == 1.0
228
+ left = mdf.MicroSeries(x, weights=weights)
229
+ with np.errstate(over="raise", invalid="raise"):
230
+ actual = [left.cov(pd.Series(x), ddof=ddof), left.corr(pd.Series(x), ddof=ddof)]
231
+ np.testing.assert_allclose(actual, expected, rtol=2e-15, atol=0)
232
+
233
+
234
+ @pytest.mark.parametrize("frequency", [0.5, 1, 1_000_000])
235
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
236
+ @pytest.mark.parametrize("scales", [(1.0, 1.0), (1e153, -1e153), (1e200, 1e-200)])
237
+ def test_cov_corr_preserve_frequency_correction_across_value_scales(
238
+ frequency, ddof, scales
239
+ ):
240
+ x = np.array([1.0, 4.0, 8.0]) * scales[0]
241
+ y = np.array([5.0, 2.0, 9.0]) * scales[1]
242
+ weights = np.array([1, 3, 2]) * frequency
243
+ expected = exact_weighted_moments(x, y, weights, ddof)
244
+ left = mdf.MicroSeries(x, weights=weights)
245
+ with np.errstate(over="raise", invalid="raise"):
246
+ actual = [left.cov(pd.Series(y), ddof=ddof), left.corr(pd.Series(y), ddof=ddof)]
247
+ np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
248
+
249
+
250
+ @pytest.mark.parametrize("huge", [1e16, 1e20])
251
+ @pytest.mark.parametrize("order", list(permutations(range(3))))
252
+ def test_cov_corr_low_weight_extreme_does_not_make_result_depend_on_row_order(
253
+ huge, order
254
+ ):
255
+ # The large observation contributes to covariance, but using it as the
256
+ # centering origin must not erase the difference between 1 and 2.
257
+ x = np.array([huge, 1.0, 2.0])
258
+ y = np.array([0.0, 1.0, 2.0])
259
+ weights = np.array([1 / huge, 1.0, 1.0])
260
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
261
+ order = list(order)
262
+ left = mdf.MicroSeries(x[order], weights=weights[order])
263
+ right = pd.Series(y[order])
264
+ np.testing.assert_allclose(
265
+ [left.cov(right, ddof=0), left.corr(right, ddof=0)],
266
+ expected,
267
+ rtol=3e-15,
268
+ atol=0,
269
+ )
270
+
271
+
272
+ def test_cov_corr_smallest_common_positive_weight_cancels_from_population_moments():
273
+ # The common positive weight cancels: xy = 1, xx = yy = 2 for
274
+ # centered observations [-1, 0, 1] and [-1, 1, 0].
275
+ x, y = [1, 2, 3], [3, 5, 4]
276
+ weights = [np.nextafter(0.0, 1.0)] * 3
277
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
278
+ assert expected == (1 / 3, 0.5)
279
+ left = mdf.MicroSeries(x, weights=weights)
280
+ right = pd.Series(y)
281
+ np.testing.assert_allclose(
282
+ [left.cov(right, ddof=0), left.corr(right, ddof=0)],
283
+ expected,
284
+ rtol=3e-15,
285
+ atol=0,
286
+ )
287
+
288
+
289
+ @pytest.mark.parametrize(
290
+ "frequency", [np.nextafter(0.0, 1.0), np.finfo(float).tiny, 1.0]
291
+ )
292
+ @pytest.mark.parametrize("multipliers", [[1, 2, 3], [1, 1, 2]])
293
+ @pytest.mark.parametrize("order", list(permutations(range(3))))
294
+ def test_cov_corr_unequal_tiny_and_normal_weights_match_exact_moments(
295
+ frequency, multipliers, order
296
+ ):
297
+ x, y = np.array([1.0, 2.0, 3.0]), np.array([3.0, 5.0, 4.0])
298
+ weights = np.array(multipliers) * frequency
299
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
300
+ order = list(order)
301
+ left = mdf.MicroSeries(x[order], weights=weights[order])
302
+ right = pd.Series(y[order])
303
+ np.testing.assert_allclose(
304
+ [left.cov(right, ddof=0), left.corr(right, ddof=0)],
305
+ expected,
306
+ rtol=3e-15,
307
+ atol=0,
308
+ )
309
+
310
+
311
+ @pytest.mark.parametrize(
312
+ "x,y",
313
+ [
314
+ (np.array([1, 2, 3]) * 1e153, np.array([3, 5, 4]) * -1e153),
315
+ (np.array([1, 2, 3]) * 1e200, np.array([3, 5, 4]) * 1e-200),
316
+ ([1e12 - 2**-13, 1e12, 1e12 + 2**-13], [3, 5, 4]),
317
+ ],
318
+ )
319
+ def test_cov_corr_subnormal_weights_preserve_extreme_value_scales(x, y):
320
+ weights = np.array([1, 2, 3]) * np.nextafter(0.0, 1.0)
321
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
322
+ left = mdf.MicroSeries(x, weights=weights)
323
+ right = pd.Series(y)
324
+ with np.errstate(over="raise", invalid="raise"):
325
+ actual = [left.cov(right, ddof=0), left.corr(right, ddof=0)]
326
+ np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
327
+
328
+
329
+ @pytest.mark.parametrize("frequency", [np.nextafter(0.0, 1.0), 0.5])
330
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
331
+ def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof(
332
+ frequency, ddof
333
+ ):
334
+ x, y = [1, 2, 3], [3, 5, 4]
335
+ weights = np.array([1, 2, 3]) * frequency
336
+ left = mdf.MicroSeries(x, weights=weights)
337
+ right = pd.Series(y)
338
+ actual = [left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)]
339
+ if weights.sum() <= ddof:
340
+ assert np.isnan(actual).all()
341
+ else:
342
+ expected = exact_weighted_moments(x, y, weights, ddof=ddof)
343
+ np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.3.9
3
+ Version: 1.4.0
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -12,7 +12,9 @@ microdf/tests/test_nullify_weights_index.py
12
12
  microdf/tests/test_pandas3_compatibility.py
13
13
  microdf/tests/test_quantile_missing_values.py
14
14
  microdf/tests/test_serialization.py
15
+ microdf/tests/test_sum_axes.py
15
16
  microdf/tests/test_version_metadata.py
17
+ microdf/tests/test_weighted_cov_corr.py
16
18
  microdf_python.egg-info/PKG-INFO
17
19
  microdf_python.egg-info/SOURCES.txt
18
20
  microdf_python.egg-info/dependency_links.txt
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "microdf-python"
7
- version = "1.3.9"
7
+ version = "1.4.0"
8
8
  description = "Weighted pandas DataFrames and Series for survey microdata"
9
9
  readme = "README.md"
10
10
  authors = [
File without changes
File without changes
File without changes