microdf-python 1.5.6__py3-none-any.whl → 1.5.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
microdf/microdataframe.py CHANGED
@@ -7,7 +7,15 @@ from typing import Callable, List, Optional, Union
7
7
  import numpy as np
8
8
  import pandas as pd
9
9
 
10
- from microdf.microseries import MicroSeries, MicroSeriesGroupBy
10
+ from microdf.microseries import (
11
+ MicroSeries,
12
+ MicroSeriesGroupBy,
13
+ _pair_options,
14
+ _usable_pair,
15
+ _validate_frequency_weights,
16
+ _weighted_correlation,
17
+ _weighted_covariance,
18
+ )
11
19
  from microdf._weights import (
12
20
  WeightPropagationMixin,
13
21
  aligned_weights,
@@ -74,16 +82,99 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
74
82
  super().__finalize__(other, method=method, **kwargs)
75
83
  return finalize_weights(self, other, method, previous)
76
84
 
77
- @wraps(pd.DataFrame.cov)
78
- def cov(self, *args, **kwargs) -> pd.DataFrame:
79
- # Column summaries have no observation weights, even if labels match.
80
- result = pd.DataFrame(self, copy=False).cov(*args, **kwargs)
81
- return result.__finalize__(self, method="cov")
85
+ def cov(
86
+ self,
87
+ min_periods: Optional[int] = None,
88
+ ddof: int = 1,
89
+ numeric_only: bool = False,
90
+ ) -> pd.DataFrame:
91
+ """Pairwise frequency-weighted covariance of the columns.
92
+
93
+ Every cell uses the estimator of :meth:`MicroSeries.cov` with this
94
+ frame's weights: ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) -
95
+ ddof)`` over the rows where both columns are present and the weight
96
+ is positive. Integer weights therefore match ``pandas.DataFrame.cov``
97
+ on the replicated sample. Missing values are removed pairwise, so
98
+ each cell can use a different set of rows, as in pandas.
99
+
100
+ The result is a plain ``pandas.DataFrame``: it summarises columns,
101
+ so it carries no row weights.
102
+
103
+ :param min_periods: Minimum usable row pairs per cell, not the sum
104
+ of frequency weights. Cells with fewer are NaN. Defaults to 1.
105
+ :param ddof: Degrees of freedom subtracted from the weight total.
106
+ :param numeric_only: Use only numeric columns. Otherwise every column
107
+ is converted to float, raising the error pandas raises.
108
+ :returns: Covariance matrix indexed by column in both directions.
109
+ """
110
+ return self._weighted_pairwise(
111
+ "cov", min_periods=min_periods, ddof=ddof, numeric_only=numeric_only
112
+ )
82
113
 
83
- @wraps(pd.DataFrame.corr)
84
- def corr(self, *args, **kwargs) -> pd.DataFrame:
85
- result = pd.DataFrame(self, copy=False).corr(*args, **kwargs)
86
- return result.__finalize__(self, method="corr")
114
+ def corr(
115
+ self,
116
+ method: str = "pearson",
117
+ min_periods: int = 1,
118
+ numeric_only: bool = False,
119
+ ) -> pd.DataFrame:
120
+ """Pairwise frequency-weighted Pearson correlation of the columns.
121
+
122
+ Every cell uses the estimator of :meth:`MicroSeries.corr` with this
123
+ frame's weights over the rows where both columns are present and the
124
+ weight is positive. Constant columns give NaN. Only ``"pearson"`` is
125
+ supported, as on :meth:`MicroSeries.corr`; for an unweighted rank
126
+ correlation convert to ``pandas.DataFrame`` first.
127
+
128
+ The result is a plain ``pandas.DataFrame``: it summarises columns, so
129
+ it carries no row weights.
130
+
131
+ :param method: Only "pearson" is supported.
132
+ :param min_periods: Minimum usable row pairs per cell, not the sum of
133
+ frequency weights. Cells with fewer are NaN.
134
+ :param numeric_only: Use only numeric columns. Otherwise every column
135
+ is converted to float, raising the error pandas raises.
136
+ :returns: Correlation matrix indexed by column in both directions.
137
+ """
138
+ if method != "pearson":
139
+ raise ValueError("weighted correlation only supports method='pearson'")
140
+ return self._weighted_pairwise(
141
+ "corr", min_periods=min_periods, ddof=1, numeric_only=numeric_only
142
+ )
143
+
144
+ def _weighted_pairwise(
145
+ self,
146
+ statistic: str,
147
+ *,
148
+ min_periods: Optional[int],
149
+ ddof: int,
150
+ numeric_only: bool,
151
+ ) -> pd.DataFrame:
152
+ """Fill a symmetric column matrix one usable pair at a time."""
153
+ min_periods, ddof = _pair_options(min_periods, ddof)
154
+ data = self._get_numeric_data() if numeric_only else self
155
+ frame = pd.DataFrame(data, copy=False)
156
+ # The conversion pandas uses, so non-numeric columns raise its error.
157
+ values = frame.to_numpy(dtype=float, na_value=np.nan)
158
+ weights = np.asarray(self.weights, dtype=float)
159
+ _validate_frequency_weights(weights)
160
+ columns = frame.columns
161
+ matrix = np.full((len(columns), len(columns)), np.nan)
162
+ for i in range(len(columns)):
163
+ for j in range(i, len(columns)):
164
+ pair = _usable_pair(
165
+ values[:, i], values[:, j], weights, min_periods, ddof, True
166
+ )
167
+ if pair is None:
168
+ continue
169
+ x, y, pair_weights, denominator = pair
170
+ if statistic == "cov":
171
+ cell = _weighted_covariance(x, y, pair_weights, denominator)
172
+ else:
173
+ cell = _weighted_correlation(x, y, pair_weights)
174
+ matrix[i, j] = matrix[j, i] = cell
175
+ result = pd.DataFrame(matrix, index=columns, columns=columns)
176
+ # Column summaries have no observation weights, even if labels match.
177
+ return pd.DataFrame.__finalize__(result, self, method=statistic)
87
178
 
88
179
  def __setstate__(self, state) -> None:
89
180
  """Restore a pickled MicroDataFrame.
microdf/microseries.py CHANGED
@@ -51,6 +51,84 @@ def _weighted_centered_vector(
51
51
  return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)
52
52
 
53
53
 
54
+ def _pair_options(min_periods: Optional[int], ddof: int) -> tuple[int, int]:
55
+ """Validate the row minimum and degrees of freedom for paired moments."""
56
+ if not isinstance(ddof, (int, np.integer)):
57
+ raise TypeError("ddof must be an integer")
58
+ if min_periods is None:
59
+ min_periods = 1
60
+ if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
61
+ raise ValueError("min_periods must be a nonnegative integer")
62
+ return int(min_periods), int(ddof)
63
+
64
+
65
+ def _validate_frequency_weights(weights: np.ndarray) -> None:
66
+ """Reject weights that cannot be frequencies."""
67
+ if not np.isfinite(weights).all() or (weights < 0).any():
68
+ raise ValueError("frequency weights must be finite and nonnegative")
69
+
70
+
71
+ def _usable_pair(
72
+ x: np.ndarray,
73
+ y: np.ndarray,
74
+ weights: np.ndarray,
75
+ min_periods: int,
76
+ ddof: int,
77
+ skipna: bool,
78
+ ) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
79
+ """Keep the present, positively weighted rows of an aligned pair."""
80
+ # Zero frequency means the row is absent, including for skipna=False.
81
+ positive = weights > 0
82
+ x, y, weights = x[positive], y[positive], weights[positive]
83
+ missing = np.isnan(x) | np.isnan(y)
84
+ if not skipna and missing.any():
85
+ return None
86
+ x, y, weights = x[~missing], y[~missing], weights[~missing]
87
+ total_weight = weights.sum()
88
+ if not np.isfinite(total_weight):
89
+ raise ValueError("the sum of frequency weights must be finite")
90
+ if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
91
+ return None
92
+ return x, y, weights, float(total_weight - ddof)
93
+
94
+
95
+ def _weighted_covariance(
96
+ x: np.ndarray, y: np.ndarray, weights: np.ndarray, denominator: float
97
+ ) -> float:
98
+ """Frequency-weighted covariance of a usable pair."""
99
+ if not np.isfinite(x).all() or not np.isfinite(y).all():
100
+ return np.nan
101
+ x, x_exponent = _weighted_centered_vector(x, weights)
102
+ y, y_exponent = _weighted_centered_vector(y, weights)
103
+ # Combine exponents only after dividing out sum(weights) - ddof.
104
+ # Neither the original squared scale nor raw weighted sum need fit.
105
+ denominator, denominator_exponent = np.frexp(denominator)
106
+ return float(
107
+ np.ldexp(
108
+ np.sum(x * y) / denominator,
109
+ x_exponent + y_exponent - int(denominator_exponent),
110
+ )
111
+ )
112
+
113
+
114
+ def _weighted_correlation(x: np.ndarray, y: np.ndarray, weights: np.ndarray) -> float:
115
+ """Frequency-weighted Pearson correlation of a usable pair."""
116
+ if not np.isfinite(x).all() or not np.isfinite(y).all():
117
+ return np.nan
118
+ # A weighted mean can round away from identical decimal inputs.
119
+ # Check the retained observations exactly before subtracting it.
120
+ if (x == x[0]).all() or (y == y[0]).all():
121
+ return np.nan
122
+ x, _ = _weighted_centered_vector(x, weights)
123
+ y, _ = _weighted_centered_vector(y, weights)
124
+ x_ss = np.sum(x * x)
125
+ y_ss = np.sum(y * y)
126
+ if x_ss == 0 or y_ss == 0:
127
+ return np.nan
128
+ result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
129
+ return float(np.clip(result, -1.0, 1.0))
130
+
131
+
54
132
  def _weighted_top_share(
55
133
  values: np.ndarray, weights: np.ndarray, top_x_pct: float
56
134
  ) -> float:
@@ -456,12 +534,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
456
534
  """Align usable paired observations with their left weights."""
457
535
  if not isinstance(other, pd.Series):
458
536
  raise TypeError("other must be a pandas Series or MicroSeries")
459
- if not isinstance(ddof, (int, np.integer)):
460
- raise TypeError("ddof must be an integer")
461
- if min_periods is None:
462
- min_periods = 1
463
- if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
464
- raise ValueError("min_periods must be a nonnegative integer")
537
+ min_periods, ddof = _pair_options(min_periods, ddof)
465
538
  if len(self) == 0 or len(other) == 0:
466
539
  return None
467
540
 
@@ -477,27 +550,8 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
477
550
  )
478
551
  y = right.to_numpy(dtype=float, na_value=np.nan)
479
552
  weights = np.asarray(self.weights, dtype=float)[positions]
480
- if not np.isfinite(weights).all() or (weights < 0).any():
481
- raise ValueError("frequency weights must be finite and nonnegative")
482
-
483
- # Zero frequency means the row is absent, including for skipna=False.
484
- positive = weights > 0
485
- x, y, weights = x[positive], y[positive], weights[positive]
486
- missing = np.isnan(x) | np.isnan(y)
487
- if not skipna and missing.any():
488
- return None
489
- x, y, weights = x[~missing], y[~missing], weights[~missing]
490
- total_weight = weights.sum()
491
- if not np.isfinite(total_weight):
492
- raise ValueError("the sum of frequency weights must be finite")
493
- if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
494
- return None
495
- return (
496
- x,
497
- y,
498
- weights,
499
- float(total_weight - ddof),
500
- )
553
+ _validate_frequency_weights(weights)
554
+ return _usable_pair(x, y, weights, min_periods, ddof, skipna)
501
555
 
502
556
  def cov(
503
557
  self,
@@ -531,19 +585,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
531
585
  if pair is None:
532
586
  return np.nan
533
587
  x, y, weights, denominator = pair
534
- if not np.isfinite(x).all() or not np.isfinite(y).all():
535
- return np.nan
536
- x, x_exponent = _weighted_centered_vector(x, weights)
537
- y, y_exponent = _weighted_centered_vector(y, weights)
538
- # Combine exponents only after dividing out sum(weights) - ddof.
539
- # Neither the original squared scale nor raw weighted sum need fit.
540
- denominator, denominator_exponent = np.frexp(denominator)
541
- return float(
542
- np.ldexp(
543
- np.sum(x * y) / denominator,
544
- x_exponent + y_exponent - int(denominator_exponent),
545
- )
546
- )
588
+ return _weighted_covariance(x, y, weights, denominator)
547
589
 
548
590
  def corr(
549
591
  self,
@@ -578,20 +620,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
578
620
  if pair is None:
579
621
  return np.nan
580
622
  x, y, weights, _ = pair
581
- if not np.isfinite(x).all() or not np.isfinite(y).all():
582
- return np.nan
583
- # A weighted mean can round away from identical decimal inputs.
584
- # Check the retained observations exactly before subtracting it.
585
- if (x == x[0]).all() or (y == y[0]).all():
586
- return np.nan
587
- x, _ = _weighted_centered_vector(x, weights)
588
- y, _ = _weighted_centered_vector(y, weights)
589
- x_ss = np.sum(x * x)
590
- y_ss = np.sum(y * y)
591
- if x_ss == 0 or y_ss == 0:
592
- return np.nan
593
- result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
594
- return float(np.clip(result, -1.0, 1.0))
623
+ return _weighted_correlation(x, y, weights)
595
624
 
596
625
  def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
597
626
  """Calculates weighted quantiles of the MicroSeries.
@@ -355,31 +355,51 @@ def test_series_to_dataframe_weights_follow_aligned_rows_and_explicit_override(m
355
355
  mdf.MicroDataFrame(data, index=[3, 99])
356
356
 
357
357
 
358
+ def replicated(data, weights, index=None):
359
+ """Frequency-weight oracle: the plain frame with each row repeated."""
360
+ plain = pd.DataFrame(data, index=index)
361
+ return plain.iloc[np.repeat(np.arange(len(plain)), weights)]
362
+
363
+
364
+ def pairwise(oracle, method, *args, **kwargs):
365
+ """Apply a pandas matrix summary one complete pair at a time.
366
+
367
+ pandas honours ``ddof`` only through ``np.cov`` on complete data; once any
368
+ value is missing it falls back to a kernel that ignores it. Dropping the
369
+ missing rows per pair keeps the oracle on the ``np.cov`` path.
370
+ """
371
+ columns = oracle.columns
372
+ out = pd.DataFrame(np.nan, index=columns, columns=columns, dtype=float)
373
+ for i, left in enumerate(columns):
374
+ for right in columns[i:]:
375
+ pair = oracle[[left, right]] if left != right else oracle[[left]]
376
+ cell = getattr(pair.dropna(), method)(*args, **kwargs).iloc[0, -1]
377
+ out.loc[left, right] = out.loc[right, left] = cell
378
+ return out
379
+
380
+
358
381
  @pytest.mark.parametrize("method", ["cov", "corr"])
359
382
  @pytest.mark.parametrize("coincident_labels", [False, True])
360
- def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_labels):
383
+ def test_dataframe_matrix_summaries_are_plain_and_frequency_weighted(
384
+ method, coincident_labels
385
+ ):
361
386
  if coincident_labels:
362
- frame = mdf.MicroDataFrame(
363
- {"x": [10.0, 20.0], "y": [4.0, 8.0]},
364
- index=["x", "y"],
365
- weights=[2, 3],
366
- )
367
- # Sample covariance divides centered cross-products by n - 1.
368
- covariance = [[50.0, 20.0], [20.0, 8.0]]
387
+ data = {"x": [10.0, 20.0], "y": [4.0, 8.0]}
388
+ weights = [2, 3]
389
+ frame = mdf.MicroDataFrame(data, index=["x", "y"], weights=weights)
390
+ # Weighted means x = 16, y = 6.4; sum(w) - 1 = 4 in the denominator.
391
+ covariance = [[30.0, 12.0], [12.0, 4.8]]
369
392
  correlation = [[1.0, 1.0], [1.0, 1.0]]
370
- else:
371
- frame = mdf.MicroDataFrame(
372
- {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]},
373
- weights=[2, 3, 5],
393
+ expected = pd.DataFrame(
394
+ covariance if method == "cov" else correlation,
395
+ index=frame.columns,
396
+ columns=frame.columns,
374
397
  )
375
- # Centered x = [-10, 0, 10], y = [-2, 2, 0]; n - 1 = 2.
376
- covariance = [[100.0, 10.0], [10.0, 4.0]]
377
- correlation = [[1.0, 0.5], [0.5, 1.0]]
378
- expected = pd.DataFrame(
379
- covariance if method == "cov" else correlation,
380
- index=frame.columns,
381
- columns=frame.columns,
382
- )
398
+ else:
399
+ data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
400
+ weights = [2, 3, 5]
401
+ frame = mdf.MicroDataFrame(data, weights=weights)
402
+ expected = getattr(replicated(data, weights), method)()
383
403
 
384
404
  result = getattr(frame, method)()
385
405
 
@@ -395,17 +415,15 @@ def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_
395
415
  [
396
416
  ("cov", (), {}),
397
417
  ("cov", (2, 0), {}),
398
- ("cov", (), {"min_periods": 4, "ddof": 2}),
418
+ ("cov", (), {"min_periods": 2, "ddof": 2}),
399
419
  ("cov", (), {"min_periods": 2, "ddof": 0, "numeric_only": True}),
400
420
  ("corr", (), {}),
401
421
  ("corr", ("pearson", 2, True), {}),
402
- ("corr", (), {"method": "spearman", "min_periods": 2}),
403
- ("corr", (), {"min_periods": 4}),
404
- ("corr", (), {"method": lambda x, y: np.dot(x, y), "min_periods": 2}),
422
+ ("corr", (), {"min_periods": 2}),
405
423
  ],
406
424
  )
407
425
  @pytest.mark.parametrize("missing", [False, True])
408
- def test_dataframe_matrix_summaries_preserve_pandas_arguments(
426
+ def test_dataframe_matrix_summaries_follow_pandas_arguments(
409
427
  method, args, kwargs, missing
410
428
  ):
411
429
  data = {
@@ -413,8 +431,11 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments(
413
431
  "y": [4.0, 8.0, np.nan if missing else 6.0, 9.0],
414
432
  "flag": [True, False, True, True],
415
433
  }
416
- frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=[2, 3, 5, 7])
417
- expected = getattr(pd.DataFrame(data, index=frame.index), method)(*args, **kwargs)
434
+ weights = [2, 3, 5, 7]
435
+ # Duplicate labels: weights follow row position, never labels.
436
+ frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=weights)
437
+ oracle = replicated(data, weights, index=frame.index)
438
+ expected = pairwise(oracle, method, *args, **kwargs)
418
439
 
419
440
  result = getattr(frame, method)(*args, **kwargs)
420
441
 
@@ -423,12 +444,38 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments(
423
444
  pd.testing.assert_series_equal(result.sum(), expected.sum())
424
445
 
425
446
 
447
+ @pytest.mark.parametrize("method", ["cov", "corr"])
448
+ def test_dataframe_matrix_summaries_min_periods_counts_usable_rows(method):
449
+ data = {"x": [10.0, 20.0, 30.0, 40.0], "y": [4.0, 8.0, np.nan, 9.0]}
450
+ frame = mdf.MicroDataFrame(data, weights=[20, 30, 50, 70])
451
+ # x and y share three usable rows, whatever their weight total.
452
+ assert np.isfinite(getattr(frame, method)(min_periods=3).loc["x", "y"])
453
+ result = getattr(frame, method)(min_periods=4)
454
+ assert np.isnan(result.loc["x", "y"]) and np.isnan(result.loc["y", "x"])
455
+ assert np.isfinite(result.loc["x", "x"])
456
+
457
+
458
+ @pytest.mark.parametrize(
459
+ "method", ["spearman", "kendall", lambda x, y: float(np.dot(x, y))]
460
+ )
461
+ def test_dataframe_correlation_rejects_other_methods(method):
462
+ frame = mdf.MicroDataFrame(
463
+ {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}, weights=[2, 3, 5]
464
+ )
465
+ with pytest.raises(ValueError, match="pearson") as frame_error:
466
+ frame.corr(method=method)
467
+ with pytest.raises(ValueError) as series_error:
468
+ frame.x.corr(frame.y, method=method)
469
+ assert str(frame_error.value) == str(series_error.value)
470
+
471
+
426
472
  @pytest.mark.parametrize("method", ["cov", "corr"])
427
473
  def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method):
428
474
  data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0], "label": ["a", "b", "c"]}
429
- frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
475
+ weights = [2, 3, 5]
476
+ frame = mdf.MicroDataFrame(data, weights=weights)
430
477
  plain = pd.DataFrame(data)
431
- expected = getattr(plain, method)(numeric_only=True)
478
+ expected = getattr(replicated(data, weights), method)(numeric_only=True)
432
479
 
433
480
  result = getattr(frame, method)(numeric_only=True)
434
481
 
@@ -442,37 +489,25 @@ def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method):
442
489
  assert str(microdf_error.value) == str(pandas_error.value)
443
490
 
444
491
 
445
- def test_dataframe_correlation_preserves_optional_kendall_support():
446
- data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
447
- frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
448
- try:
449
- expected = pd.DataFrame(data).corr(method="kendall")
450
- except ImportError as pandas_error:
451
- # Kendall requires scipy; delegation preserves pandas' dependency error.
452
- with pytest.raises(type(pandas_error)) as microdf_error:
453
- frame.corr(method="kendall")
454
- assert str(microdf_error.value) == str(pandas_error)
455
- else:
456
- result = frame.corr(method="kendall")
457
- assert type(result) is pd.DataFrame
458
- pd.testing.assert_frame_equal(result, expected)
459
-
460
-
461
492
  @pytest.mark.parametrize("method", ["cov", "corr"])
462
493
  def test_dataframe_matrix_summaries_preserve_pandas_metadata(method):
463
494
  data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
464
- frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
465
- plain = pd.DataFrame(data)
466
- for source in [frame, plain]:
467
- source.attrs = {"survey": {"year": 2026}}
468
- source.flags.allows_duplicate_labels = False
469
- source.columns.name = "measure"
470
- expected = getattr(plain, method)()
495
+ weights = [2, 3, 5]
496
+ frame = mdf.MicroDataFrame(data, weights=weights)
497
+ frame.attrs = {"survey": {"year": 2026}}
498
+ frame.flags.allows_duplicate_labels = False
499
+ frame.columns.name = "measure"
500
+ expected = getattr(replicated(data, weights), method)()
471
501
 
472
502
  result = getattr(frame, method)()
473
503
 
474
504
  assert type(result) is pd.DataFrame
475
- pd.testing.assert_frame_equal(result, expected)
476
- assert result.attrs == expected.attrs
505
+ pd.testing.assert_frame_equal(
506
+ result, expected, check_flags=False, check_names=False
507
+ )
508
+ assert result.attrs == {"survey": {"year": 2026}}
509
+ assert result.flags.allows_duplicate_labels is False
510
+ assert result.index.name == "measure" and result.columns.name == "measure"
511
+ # attrs are copied, not shared, with the source frame.
477
512
  result.attrs["survey"]["year"] = 2025
478
513
  assert frame.attrs["survey"]["year"] == 2026
@@ -341,3 +341,108 @@ def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof(
341
341
  else:
342
342
  expected = exact_weighted_moments(x, y, weights, ddof=ddof)
343
343
  np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
344
+
345
+
346
+ def replicated_frame(data, weights):
347
+ """Frequency-weight oracle for a frame: repeat each row by its weight."""
348
+ plain = pd.DataFrame(data)
349
+ return plain.iloc[np.repeat(np.arange(len(plain)), weights)]
350
+
351
+
352
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
353
+ def test_dataframe_cov_corr_match_replicated_frequency_sample(ddof):
354
+ data = {"x": [1.0, 4.0, 8.0, 2.0], "y": [5.0, 2.0, 9.0, 7.0], "z": [3, 3, 1, 6]}
355
+ weights = [1, 3, 2, 4]
356
+ frame = mdf.MicroDataFrame(data, weights=weights)
357
+ oracle = replicated_frame(data, weights)
358
+ pd.testing.assert_frame_equal(frame.cov(ddof=ddof), oracle.cov(ddof=ddof))
359
+ pd.testing.assert_frame_equal(frame.corr(), oracle.corr())
360
+
361
+
362
+ def test_dataframe_cov_corr_cells_equal_microseries_results():
363
+ frame = mdf.MicroDataFrame(
364
+ {
365
+ "a": [1.0, 4.0, 8.0, 3.0, 6.0],
366
+ "b": [5.0, 2.0, 9.0, np.nan, 1.0],
367
+ "c": [2.0, 2.0, 7.0, 1.0, 1e9],
368
+ },
369
+ weights=[1, 3, 2, 0, 0.5],
370
+ )
371
+ cov, corr = frame.cov(), frame.corr()
372
+ for left in frame.columns:
373
+ for right in frame.columns:
374
+ assert cov.loc[left, right] == frame[left].cov(frame[right])
375
+ assert corr.loc[left, right] == frame[left].corr(frame[right])
376
+ assert cov.loc[left, right] == cov.loc[right, left]
377
+
378
+
379
+ def test_dataframe_cov_corr_use_pairwise_complete_rows():
380
+ frame = mdf.MicroDataFrame(
381
+ {
382
+ "x": [1.0, np.nan, 5.0, 9.0, 20.0],
383
+ "y": [2.0, 4.0, np.nan, 8.0, 30.0],
384
+ "z": [1.0, 1.0, 2.0, 3.0, 5.0],
385
+ },
386
+ weights=[1, 9, 2, 3, 7],
387
+ )
388
+ cov, corr = frame.cov(), frame.corr()
389
+ # x-y keeps rows 0, 3, 4; x-z rows 0, 2, 3, 4; y-z rows 0, 1, 3, 4.
390
+ cells = {
391
+ ("x", "y"): replicated_moments([1, 9, 20], [2, 8, 30], [1, 3, 7]),
392
+ ("x", "z"): replicated_moments([1, 5, 9, 20], [1, 2, 3, 5], [1, 2, 3, 7]),
393
+ ("y", "z"): replicated_moments([2, 4, 8, 30], [1, 1, 3, 5], [1, 9, 3, 7]),
394
+ }
395
+ for (left, right), (expected_cov, expected_corr) in cells.items():
396
+ assert cov.loc[left, right] == pytest.approx(expected_cov)
397
+ assert corr.loc[left, right] == pytest.approx(expected_corr)
398
+
399
+
400
+ def test_dataframe_cov_corr_omit_zero_weight_rows():
401
+ frame = mdf.MicroDataFrame(
402
+ {"x": [1.0, 4.0, 8.0, 1e6], "y": [5.0, 2.0, 9.0, -1e6]}, weights=[1, 3, 2, 0]
403
+ )
404
+ expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2])
405
+ assert frame.cov().loc["x", "y"] == pytest.approx(expected_cov)
406
+ assert frame.corr().loc["x", "y"] == pytest.approx(expected_corr)
407
+
408
+
409
+ @pytest.mark.parametrize("weights", [[1, -1, 2], [1, np.inf, 2], [1, np.nan, 2]])
410
+ def test_dataframe_cov_corr_reject_invalid_frequency_weights(weights):
411
+ frame = mdf.MicroDataFrame(
412
+ {"x": [1.0, 2.0, 3.0], "y": [3.0, 1.0, 2.0]}, weights=weights
413
+ )
414
+ with pytest.raises(ValueError, match="finite and nonnegative"):
415
+ frame.cov()
416
+ with pytest.raises(ValueError, match="finite and nonnegative"):
417
+ frame.corr()
418
+
419
+
420
+ def test_dataframe_cov_corr_diagonal_and_constant_columns():
421
+ frame = mdf.MicroDataFrame(
422
+ {"x": [1.0, 4.0, 8.0], "c": [2.0, 2.0, 2.0]}, weights=[1, 3, 2]
423
+ )
424
+ cov, corr = frame.cov(), frame.corr()
425
+ assert cov.loc["x", "x"] == pytest.approx(frame.x.var())
426
+ assert cov.loc["c", "c"] == 0 and cov.loc["x", "c"] == 0
427
+ assert corr.loc["x", "x"] == 1.0
428
+ assert np.isnan(corr.loc["c", "c"]) and np.isnan(corr.loc["x", "c"])
429
+
430
+
431
+ def test_dataframe_cov_corr_insufficient_weight_total_and_empty_frame():
432
+ frame = mdf.MicroDataFrame({"x": [1.0, 4.0], "y": [5.0, 2.0]}, weights=[1, 1])
433
+ assert np.isnan(frame.cov(ddof=2)).all().all()
434
+ assert np.isfinite(frame.cov(ddof=1)).all().all()
435
+ empty = mdf.MicroDataFrame({"x": [], "y": []}, weights=[])
436
+ for result in (empty.cov(), empty.corr()):
437
+ assert list(result.columns) == ["x", "y"]
438
+ assert np.isnan(result).all().all()
439
+
440
+
441
+ def test_dataframe_cov_corr_validate_options_like_microseries():
442
+ frame = mdf.MicroDataFrame({"x": [1.0, 2.0], "y": [2.0, 1.0]}, weights=[1, 1])
443
+ with pytest.raises(TypeError, match="ddof"):
444
+ frame.cov(ddof=1.5)
445
+ with pytest.raises(ValueError, match="min_periods"):
446
+ frame.cov(min_periods=-1)
447
+ with pytest.raises(ValueError, match="min_periods"):
448
+ frame.corr(min_periods=-1)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.5.6
3
+ Version: 1.5.8
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -1,7 +1,7 @@
1
1
  microdf/__init__.py,sha256=G4m4UDiGePngG1i0DdGYhsTAXqyDELR6Mbq72OXOMG0,790
2
2
  microdf/_weights.py,sha256=uBOcTZGmVJlzi27sSAALSTb4V9j56tXuaBfJB83x7j0,10303
3
- microdf/microdataframe.py,sha256=BY5Knw4WP0TcdYM6qcSpPD1oUI9y087kxHkeyuarJoE,39475
4
- microdf/microseries.py,sha256=v0X05ko_HYnYpIESWsMVqTF3RMVzmPdbUWsT0xLrF3A,49084
3
+ microdf/microdataframe.py,sha256=2HFXbS4gAmOQKX6eYcI5vwafCqv9aBzl3lDLODU9SXg,43436
4
+ microdf/microseries.py,sha256=ZjSiUg88zOWAslUl2QpJWK0BcglbG-l7bU6_preOqNs,50039
5
5
  microdf/replication.py,sha256=3iZ6xG24ucKofVwmXxyd7ME6rZP6iJVnN9JpWcPxg9s,7697
6
6
  microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
7
7
  microdf/tests/test_aggregation_errors.py,sha256=9jJDiEyxMb2z1Zmj-o8AHDNp8LOputbEkAMefifnWaE,2013
@@ -16,10 +16,10 @@ microdf/tests/test_serialization.py,sha256=a7pHL2hNiG5iJjRtfx3C1BCmgOZouAekiOwUx
16
16
  microdf/tests/test_sum_axes.py,sha256=N05ocwI5lLv2OgoaovRIqFIae-70356kZemRRet0ac8,8521
17
17
  microdf/tests/test_ufunc_weight_dispatch.py,sha256=abQ5I4wFRM3poxO76nbR_1dQymHkaQKLlTaI5nVI70c,25664
18
18
  microdf/tests/test_version_metadata.py,sha256=M1EabzHLKZZw3Djd6Zu2UuMQtDLV6rZ1zDrOU7W_jf0,227
19
- microdf/tests/test_weight_propagation.py,sha256=3odbufFZ2o1rnyjx6PJ7kn1TPO5K2ho3RErczI7mqO4,18761
20
- microdf/tests/test_weighted_cov_corr.py,sha256=LTnFhMWnb28f7L_9kOhPiV5UlLMAUbC9OlLIaZ1XgLs,13954
21
- microdf_python-1.5.6.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
22
- microdf_python-1.5.6.dist-info/METADATA,sha256=20EK7L7Zx5cBz1LmDvNN3hUJwwqgw9F4BqnInpGizrY,3535
23
- microdf_python-1.5.6.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
24
- microdf_python-1.5.6.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
25
- microdf_python-1.5.6.dist-info/RECORD,,
19
+ microdf/tests/test_weight_propagation.py,sha256=YCHZRdeWyWEwnHdpoY1jRBgFLQcBE1nLs61LrcT1wao,20359
20
+ microdf/tests/test_weighted_cov_corr.py,sha256=Ag4YJDy0eH3r_DhOaiX44kEZ0GHuQOoQtBVAOAkBXmw,18291
21
+ microdf_python-1.5.8.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
22
+ microdf_python-1.5.8.dist-info/METADATA,sha256=cG5neCZ-WXOvFPEZNt1saWP2WPD4URKohDdmwn_1t-M,3535
23
+ microdf_python-1.5.8.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
24
+ microdf_python-1.5.8.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
25
+ microdf_python-1.5.8.dist-info/RECORD,,