microdf-python 1.3.10__py3-none-any.whl → 1.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
microdf/microseries.py CHANGED
@@ -9,6 +9,41 @@ import pandas as pd
9
9
  logger = logging.getLogger(__name__)
10
10
 
11
11
 
12
+ def _weighted_centered_vector(
13
+ values: np.ndarray, weights: np.ndarray
14
+ ) -> tuple[np.ndarray, int]:
15
+ """Return scaled sqrt-weighted deviations and their power-of-two
16
+ exponent."""
17
+ # Center relative to a maximum-weight observation: shifting by a low-weight
18
+ # extreme could erase differences among the influential observations.
19
+ # A relative mean also preserves nearby values at a large common offset.
20
+ reference = values[np.argmax(weights)]
21
+ with np.errstate(over="ignore"):
22
+ shifted = values - reference
23
+ exponent = 0
24
+ if np.isinf(shifted).any():
25
+ # Opposite finite extremes can overflow their difference. Halving is
26
+ # exact for those values; restore that factor in the final exponent.
27
+ shifted = values / 2 - reference / 2
28
+ exponent = 1
29
+ magnitude = np.max(np.abs(shifted))
30
+ if magnitude == 0:
31
+ return shifted, 0
32
+ _, shift = np.frexp(magnitude)
33
+ shifted = np.ldexp(shifted, -shift)
34
+ # Raise tiny mean weights by an exact common power of two so products
35
+ # with the scaled deviations do not underflow. Never scale down: that
36
+ # could discard small weights when frequencies span a wide range.
37
+ _, mean_weight_exponent = np.frexp(np.max(weights))
38
+ mean_weights = np.ldexp(weights, -min(int(mean_weight_exponent), 0))
39
+ shifted -= np.average(shifted, weights=mean_weights)
40
+ # Weight each vector before taking products, then scale again so squared
41
+ # deviations never accumulate raw frequencies at the original value scale.
42
+ shifted *= np.sqrt(weights)
43
+ _, weight_shift = np.frexp(np.max(np.abs(shifted)))
44
+ return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)
45
+
46
+
12
47
  def _weighted_top_share(
13
48
  values: np.ndarray, weights: np.ndarray, top_x_pct: float
14
49
  ) -> float:
@@ -315,39 +350,152 @@ class MicroSeries(pd.Series):
315
350
  v = self._weighted_variance(ddof=ddof, skipna=skipna)
316
351
  return float(np.sqrt(v)) if np.isfinite(v) else v
317
352
 
318
- def cov(self, other, *args, **kwargs):
319
- """Pandas ``cov`` — **unweighted**.
353
+ def _weighted_pair(
354
+ self,
355
+ other: pd.Series,
356
+ min_periods: Optional[int],
357
+ ddof: int,
358
+ skipna: bool,
359
+ ) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
360
+ """Align usable paired observations with their left weights."""
361
+ if not isinstance(other, pd.Series):
362
+ raise TypeError("other must be a pandas Series or MicroSeries")
363
+ if not isinstance(ddof, (int, np.integer)):
364
+ raise TypeError("ddof must be an integer")
365
+ if min_periods is None:
366
+ min_periods = 1
367
+ if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
368
+ raise ValueError("min_periods must be a nonnegative integer")
369
+ if len(self) == 0 or len(other) == 0:
370
+ return None
371
+
372
+ # Align left row positions so values and weights undergo exactly the
373
+ # same join, including pandas' expansion of duplicate index labels.
374
+ positions = pd.Series(np.arange(len(self)), index=self.index)
375
+ positions, right = positions.align(pd.Series(other), join="inner")
376
+ positions = positions.to_numpy(dtype=int)
377
+ x = (
378
+ pd.Series(self._values)
379
+ .iloc[positions]
380
+ .to_numpy(dtype=float, na_value=np.nan)
381
+ )
382
+ y = right.to_numpy(dtype=float, na_value=np.nan)
383
+ weights = np.asarray(self.weights, dtype=float)[positions]
384
+ if not np.isfinite(weights).all() or (weights < 0).any():
385
+ raise ValueError("frequency weights must be finite and nonnegative")
386
+
387
+ # Zero frequency means the row is absent, including for skipna=False.
388
+ positive = weights > 0
389
+ x, y, weights = x[positive], y[positive], weights[positive]
390
+ missing = np.isnan(x) | np.isnan(y)
391
+ if not skipna and missing.any():
392
+ return None
393
+ x, y, weights = x[~missing], y[~missing], weights[~missing]
394
+ total_weight = weights.sum()
395
+ if not np.isfinite(total_weight):
396
+ raise ValueError("the sum of frequency weights must be finite")
397
+ if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
398
+ return None
399
+ return (
400
+ x,
401
+ y,
402
+ weights,
403
+ float(total_weight - ddof),
404
+ )
320
405
 
321
- MicroSeries does not yet compute weighted covariance. Emits a
322
- ``UserWarning`` so callers aren't silently given an unweighted number
323
- after ``.sum()`` and ``.mean()`` worked as expected. See issue tracker
324
- for a weighted implementation.
406
+ def cov(
407
+ self,
408
+ other: pd.Series,
409
+ min_periods: Optional[int] = None,
410
+ ddof: int = 1,
411
+ *,
412
+ skipna: bool = True,
413
+ ) -> float:
414
+ """Calculate frequency-weighted covariance with another Series.
415
+
416
+ Observations align by index as in pandas, including its duplicate-
417
+ label join behavior. Only this Series' weights are used; weights on
418
+ another MicroSeries are ignored. Each aligned left weight must be
419
+ finite and nonnegative. Zero-weight rows are omitted.
420
+
421
+ Uses ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) - ddof)``.
422
+ Integer weights therefore match covariance on the replicated sample.
423
+ Missing values are removed pairwise before computing both means.
424
+
425
+ :param other: A pandas Series or MicroSeries to align by index.
426
+ :param min_periods: Minimum usable aligned row pairs, not the sum of
427
+ frequency weights. Defaults to 1.
428
+ :param ddof: Degrees of freedom subtracted from the weight total.
429
+ :param skipna: Drop pairs with a missing value. If False, any missing
430
+ value in a positive-weight aligned pair produces NaN.
431
+ :returns: Weighted covariance, or NaN for an empty or insufficient
432
+ sample (including a weight total no greater than ddof).
325
433
  """
326
- warnings.warn(
327
- "MicroSeries.cov() falls through to pandas and is "
328
- "unweighted. Use MicroSeries.var()/std() for weighted "
329
- "second moments, or compute covariance manually with the "
330
- "weights.",
331
- UserWarning,
332
- stacklevel=2,
434
+ pair = self._weighted_pair(other, min_periods, ddof, skipna)
435
+ if pair is None:
436
+ return np.nan
437
+ x, y, weights, denominator = pair
438
+ if not np.isfinite(x).all() or not np.isfinite(y).all():
439
+ return np.nan
440
+ x, x_exponent = _weighted_centered_vector(x, weights)
441
+ y, y_exponent = _weighted_centered_vector(y, weights)
442
+ # Combine exponents only after dividing out sum(weights) - ddof.
443
+ # Neither the original squared scale nor raw weighted sum need fit.
444
+ denominator, denominator_exponent = np.frexp(denominator)
445
+ return float(
446
+ np.ldexp(
447
+ np.sum(x * y) / denominator,
448
+ x_exponent + y_exponent - int(denominator_exponent),
449
+ )
333
450
  )
334
- return super().cov(other, *args, **kwargs)
335
-
336
- def corr(self, other, *args, **kwargs):
337
- """Pandas ``corr`` — **unweighted**.
338
451
 
339
- MicroSeries does not yet compute weighted correlation. Emits a
340
- ``UserWarning`` so callers aren't silently given an unweighted number.
341
- See issue tracker for a weighted implementation.
452
+ def corr(
453
+ self,
454
+ other: pd.Series,
455
+ method: str = "pearson",
456
+ min_periods: Optional[int] = None,
457
+ *,
458
+ ddof: int = 1,
459
+ skipna: bool = True,
460
+ ) -> float:
461
+ """Calculate frequency-weighted Pearson correlation.
462
+
463
+ Uses the same aligned pairs and left Series weights for covariance and
464
+ both variances. Weights on another MicroSeries are ignored. Weights
465
+ must be finite and nonnegative; zero-weight rows are omitted. Other
466
+ correlation methods, including callables, are unsupported.
467
+
468
+ :param other: A pandas Series or MicroSeries to align by index.
469
+ :param method: Only "pearson" is supported.
470
+ :param min_periods: Minimum usable aligned row pairs, not frequency
471
+ weight total. Defaults to 1.
472
+ :param ddof: Degrees of freedom for all three moments. It cancels from
473
+ the correlation but the weight total must exceed it.
474
+ :param skipna: Drop pairs with a missing value. If False, any missing
475
+ value in a positive-weight aligned pair produces NaN.
476
+ :returns: Weighted correlation, or NaN for an empty, insufficient, or
477
+ constant sample.
342
478
  """
343
- warnings.warn(
344
- "MicroSeries.corr() falls through to pandas and is "
345
- "unweighted. Compute correlation manually with the weights "
346
- "if you need the survey-weighted value.",
347
- UserWarning,
348
- stacklevel=2,
349
- )
350
- return super().corr(other, *args, **kwargs)
479
+ if method != "pearson":
480
+ raise ValueError("weighted correlation only supports method='pearson'")
481
+ pair = self._weighted_pair(other, min_periods, ddof, skipna)
482
+ if pair is None:
483
+ return np.nan
484
+ x, y, weights, _ = pair
485
+ if not np.isfinite(x).all() or not np.isfinite(y).all():
486
+ return np.nan
487
+ # A weighted mean can round away from identical decimal inputs.
488
+ # Check the retained observations exactly before subtracting it.
489
+ if (x == x[0]).all() or (y == y[0]).all():
490
+ return np.nan
491
+ x, _ = _weighted_centered_vector(x, weights)
492
+ y, _ = _weighted_centered_vector(y, weights)
493
+ x_ss = np.sum(x * x)
494
+ y_ss = np.sum(y * y)
495
+ if x_ss == 0 or y_ss == 0:
496
+ return np.nan
497
+ result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
498
+ return float(np.clip(result, -1.0, 1.0))
351
499
 
352
500
  def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
353
501
  """Calculates weighted quantiles of the MicroSeries.
@@ -744,28 +744,6 @@ def test_std_var_are_weighted() -> None:
744
744
  )
745
745
 
746
746
 
747
- def test_cov_corr_warn_when_fallthrough() -> None:
748
- """Regression: cov/corr silently returned unweighted pandas values.
749
-
750
- They still fall through to pandas (a weighted impl is a separate issue) but
751
- now emit a UserWarning so callers aren't misled.
752
- """
753
- s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
754
- s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1])
755
-
756
- with warnings.catch_warnings(record=True) as w:
757
- warnings.simplefilter("always")
758
- _ = s1.cov(s2)
759
- msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
760
- assert any("unweighted" in m.lower() for m in msgs)
761
-
762
- with warnings.catch_warnings(record=True) as w:
763
- warnings.simplefilter("always")
764
- _ = s1.corr(s2)
765
- msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
766
- assert any("unweighted" in m.lower() for m in msgs)
767
-
768
-
769
747
  def test_count_skips_nan_by_default() -> None:
770
748
  """Regression: ``count()`` included NaN-row weight, contrary to pandas.
771
749
 
@@ -0,0 +1,343 @@
1
+ import warnings
2
+ from decimal import Decimal, localcontext
3
+ from fractions import Fraction
4
+ from itertools import permutations
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ import pytest
9
+
10
+ import microdf as mdf
11
+
12
+
13
+ def replicated_moments(x, y, weights, ddof=1):
14
+ """Independent frequency-weight oracle: expand to an ordinary sample."""
15
+ repeated_x = np.repeat(np.asarray(x, dtype=float), weights)
16
+ repeated_y = np.repeat(np.asarray(y, dtype=float), weights)
17
+ return (
18
+ np.cov(repeated_x, repeated_y, ddof=ddof)[0, 1],
19
+ np.corrcoef(repeated_x, repeated_y)[0, 1],
20
+ )
21
+
22
+
23
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
24
+ def test_cov_corr_match_replicated_frequency_sample(ddof):
25
+ x, y, weights = [1, 4, 8], [5, 2, 9], [1, 3, 2]
26
+ left = mdf.MicroSeries(x, weights=weights)
27
+ right = pd.Series(y)
28
+ expected_cov, expected_corr = replicated_moments(x, y, weights, ddof)
29
+ with warnings.catch_warnings(record=True) as caught:
30
+ warnings.simplefilter("always")
31
+ assert left.cov(right, ddof=ddof) == pytest.approx(expected_cov)
32
+ assert left.corr(right, ddof=ddof) == pytest.approx(expected_corr)
33
+ assert not any("unweighted" in str(item.message).lower() for item in caught)
34
+
35
+
36
+ def test_cov_corr_align_indices_and_use_only_left_weights():
37
+ left = mdf.MicroSeries([1, 4, 8], index=["a", "b", "c"], weights=[1, 3, 2])
38
+ right = mdf.MicroSeries(
39
+ [9, 5, 2, 100], index=["c", "a", "b", "d"], weights=[99, 1, 1, 9]
40
+ )
41
+ expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2])
42
+ assert left.cov(right) == pytest.approx(expected_cov)
43
+ assert left.corr(right) == pytest.approx(expected_corr)
44
+ assert right.weights.tolist() == [99, 1, 1, 9]
45
+
46
+
47
+ def test_cov_corr_use_same_pairwise_nonmissing_sample():
48
+ left = mdf.MicroSeries(
49
+ [1, np.nan, 5, 9, 20], index=list("abcde"), weights=[1, 9, 2, 3, 7]
50
+ )
51
+ right = pd.Series([2, 4, np.nan, 8, 30], index=list("abcdf"))
52
+ expected_cov, expected_corr = replicated_moments([1, 9], [2, 8], [1, 3])
53
+ assert left.cov(right) == pytest.approx(expected_cov)
54
+ assert left.corr(right) == pytest.approx(expected_corr)
55
+ assert np.isnan(left.cov(right, skipna=False))
56
+ assert np.isnan(left.corr(right, skipna=False))
57
+
58
+
59
+ @pytest.mark.parametrize("same_index", [True, False])
60
+ def test_cov_corr_follow_pandas_duplicate_index_alignment(same_index):
61
+ left = mdf.MicroSeries([1, 4, 7], index=["a", "a", "b"], weights=[1, 3, 2])
62
+ if same_index:
63
+ right = pd.Series([2, 3, 8], index=["a", "a", "b"])
64
+ x, y, weights = [1, 4, 7], [2, 3, 8], [1, 3, 2]
65
+ else:
66
+ right = pd.Series([2, 5, 8], index=["a", "b", "b"])
67
+ # The shared a/b labels join, repeating the left row weight per pair.
68
+ x, y, weights = [1, 4, 7, 7], [2, 2, 5, 8], [1, 3, 2, 2]
69
+ expected_cov, expected_corr = replicated_moments(x, y, weights)
70
+ assert left.cov(right) == pytest.approx(expected_cov)
71
+ assert left.corr(right) == pytest.approx(expected_corr)
72
+
73
+
74
+ def test_zero_weight_rows_do_not_enter_pairwise_sample():
75
+ left = mdf.MicroSeries([1, np.nan, 5, 1000], weights=[2, 0, 1, 0])
76
+ right = pd.Series([3, np.nan, 9, -1000])
77
+ expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [2, 1])
78
+ assert left.cov(right, skipna=False) == pytest.approx(expected_cov)
79
+ assert left.corr(right, skipna=False) == pytest.approx(expected_corr)
80
+ assert np.isnan(left.cov(right, min_periods=3))
81
+ assert np.isnan(left.corr(right, min_periods=3))
82
+
83
+
84
+ def test_min_periods_counts_usable_rows_separately_from_frequency_weight():
85
+ left = mdf.MicroSeries([1, 5], weights=[10, 20])
86
+ right = pd.Series([3, 9])
87
+ expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [10, 20], ddof=0)
88
+ # Existing pandas positional arguments retain their order.
89
+ assert left.cov(right, 2, 0) == pytest.approx(expected_cov)
90
+ assert left.corr(right, "pearson", 2, ddof=0) == pytest.approx(expected_corr)
91
+ assert np.isnan(left.cov(right, min_periods=3))
92
+ assert np.isnan(left.corr(right, min_periods=3))
93
+
94
+
95
+ @pytest.mark.parametrize(
96
+ "values,other,weights",
97
+ [([], [], []), ([np.nan], [1], [3]), ([1], [2], [0]), ([1], [2], [1])],
98
+ )
99
+ def test_cov_corr_return_nan_when_sample_is_insufficient(values, other, weights):
100
+ left = mdf.MicroSeries(values, weights=weights, dtype=float)
101
+ right = pd.Series(other, dtype=float)
102
+ assert np.isnan(left.cov(right))
103
+ assert np.isnan(left.corr(right))
104
+
105
+
106
+ def test_cov_corr_no_index_overlap():
107
+ left = mdf.MicroSeries([1, 2], index=["a", "b"], weights=[1, 2])
108
+ right = pd.Series([3, 4], index=["c", "d"])
109
+ assert np.isnan(left.cov(right))
110
+ assert np.isnan(left.corr(right))
111
+
112
+
113
+ def test_cov_corr_constant_and_frequency_singleton():
114
+ left = mdf.MicroSeries([4, 4], weights=[2, 3])
115
+ right = pd.Series([1, 5])
116
+ assert left.cov(right) == 0
117
+ assert np.isnan(left.corr(right))
118
+ singleton = mdf.MicroSeries([4], weights=[3])
119
+ assert singleton.cov(pd.Series([2])) == 0
120
+ assert np.isnan(singleton.corr(pd.Series([2])))
121
+ assert np.isnan(singleton.cov(pd.Series([2]), ddof=3))
122
+ assert np.isnan(singleton.corr(pd.Series([2]), ddof=3))
123
+
124
+
125
+ def test_cov_corr_nullable_numeric_data():
126
+ left = mdf.MicroSeries(pd.Series([1, pd.NA, 4], dtype="Int64"), weights=[2, 9, 3])
127
+ right = pd.Series([3, 8, 7], dtype="Float64")
128
+ expected_cov, expected_corr = replicated_moments([1, 4], [3, 7], [2, 3])
129
+ assert left.cov(right) == pytest.approx(expected_cov)
130
+ assert left.corr(right) == pytest.approx(expected_corr)
131
+
132
+
133
+ @pytest.mark.parametrize("method", ["spearman", "kendall", lambda x, y: 1.0])
134
+ def test_non_pearson_methods_are_explicitly_unsupported(method):
135
+ left = mdf.MicroSeries([1, 2, 3], weights=[1, 2, 3])
136
+ with pytest.raises(ValueError, match="pearson"):
137
+ left.corr(pd.Series([4, 2, 5]), method=method)
138
+
139
+
140
+ @pytest.mark.parametrize("weights", [[1, -1], [1, np.nan], [1, np.inf]])
141
+ def test_cov_corr_reject_invalid_frequency_weights(weights):
142
+ left = mdf.MicroSeries([1, 2], weights=weights)
143
+ for method in (left.cov, left.corr):
144
+ with pytest.raises(ValueError, match="weights"):
145
+ method(pd.Series([2, 4]))
146
+
147
+
148
+ def test_binary_statistics_not_in_dataframe_aggregation_factories():
149
+ assert "cov" not in mdf.MicroSeries.FUNCTIONS
150
+ assert "corr" not in mdf.MicroSeries.FUNCTIONS
151
+
152
+
153
+ @pytest.mark.parametrize(
154
+ "x,y",
155
+ [([0.1, 0.1], [0, 1]), ([0, 1], [0.1, 0.1]), ([0.1, 0.1], [0.1, 0.1])],
156
+ )
157
+ @pytest.mark.parametrize("with_filtered_rows", [False, True])
158
+ def test_corr_exact_decimal_constants_return_nan(x, y, with_filtered_rows):
159
+ # Unequal weights can round the mean away from the identical 0.1 values.
160
+ # Constant detection must inspect usable observations before centering.
161
+ weights = [1, 2]
162
+ if with_filtered_rows:
163
+ x = x + [9, np.nan]
164
+ y = y + [7, 4]
165
+ weights = weights + [0, 3]
166
+ left = mdf.MicroSeries(x, weights=weights)
167
+ assert np.isnan(left.corr(pd.Series(y)))
168
+
169
+
170
+ def test_corr_does_not_treat_nearby_distinct_values_as_constant():
171
+ x = [0.1, np.nextafter(0.1, np.inf)]
172
+ left = mdf.MicroSeries(x, weights=[1, 2])
173
+ assert np.isfinite(left.corr(pd.Series([0, 1])))
174
+ assert np.isfinite(mdf.MicroSeries([0, 1], weights=[1, 2]).corr(pd.Series(x)))
175
+
176
+
177
+ def exact_weighted_moments(x, y, weights, ddof=1):
178
+ """Compute moments of the actual input floats with exact rational
179
+ arithmetic."""
180
+ x, y, weights = [
181
+ [Fraction(float(value)) for value in values] for values in (x, y, weights)
182
+ ]
183
+ total = sum(weights)
184
+ xmean = sum(w * value for w, value in zip(weights, x)) / total
185
+ ymean = sum(w * value for w, value in zip(weights, y)) / total
186
+ xy = sum(w * (a - xmean) * (b - ymean) for a, b, w in zip(x, y, weights))
187
+ xx = sum(w * (value - xmean) ** 2 for value, w in zip(x, weights))
188
+ yy = sum(w * (value - ymean) ** 2 for value, w in zip(y, weights))
189
+ with localcontext() as context:
190
+ context.prec = 100
191
+ product = xx * yy
192
+ correlation = (Decimal(xy.numerator) / Decimal(xy.denominator)) / (
193
+ Decimal(product.numerator) / Decimal(product.denominator)
194
+ ).sqrt()
195
+ return float(xy / (total - ddof)), float(correlation)
196
+
197
+
198
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
199
+ @pytest.mark.parametrize("swap", [False, True])
200
+ def test_cov_corr_preserve_small_differences_at_large_offsets(ddof, swap):
201
+ # Two distinct points are perfectly linear even two float steps apart.
202
+ x = np.array([1e12 - 2**-13, 1e12 + 2**-13])
203
+ y = np.array([0.0, 1.0])
204
+ if swap:
205
+ x, y = y, x
206
+ weights = [1, 2]
207
+ expected = exact_weighted_moments(x, y, weights, ddof)
208
+ assert expected[1] == 1.0
209
+ for shifted_x, shifted_y in [(x, y), (x - x[0], y - y[0])]:
210
+ left = mdf.MicroSeries(shifted_x, weights=weights)
211
+ right = pd.Series(shifted_y)
212
+ np.testing.assert_allclose(
213
+ [left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)],
214
+ expected,
215
+ rtol=2e-15,
216
+ atol=0,
217
+ )
218
+
219
+
220
+ @pytest.mark.parametrize("frequency", [1, 1_000_000])
221
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
222
+ def test_cov_corr_large_finite_values_do_not_overflow_raw_frequencies(frequency, ddof):
223
+ x = np.array([1.0, 2.0, 3.0]) * 1e153
224
+ weights = [frequency] * 3
225
+ expected = exact_weighted_moments(x, x, weights, ddof)
226
+ assert np.isfinite(expected).all()
227
+ assert expected[1] == 1.0
228
+ left = mdf.MicroSeries(x, weights=weights)
229
+ with np.errstate(over="raise", invalid="raise"):
230
+ actual = [left.cov(pd.Series(x), ddof=ddof), left.corr(pd.Series(x), ddof=ddof)]
231
+ np.testing.assert_allclose(actual, expected, rtol=2e-15, atol=0)
232
+
233
+
234
+ @pytest.mark.parametrize("frequency", [0.5, 1, 1_000_000])
235
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
236
+ @pytest.mark.parametrize("scales", [(1.0, 1.0), (1e153, -1e153), (1e200, 1e-200)])
237
+ def test_cov_corr_preserve_frequency_correction_across_value_scales(
238
+ frequency, ddof, scales
239
+ ):
240
+ x = np.array([1.0, 4.0, 8.0]) * scales[0]
241
+ y = np.array([5.0, 2.0, 9.0]) * scales[1]
242
+ weights = np.array([1, 3, 2]) * frequency
243
+ expected = exact_weighted_moments(x, y, weights, ddof)
244
+ left = mdf.MicroSeries(x, weights=weights)
245
+ with np.errstate(over="raise", invalid="raise"):
246
+ actual = [left.cov(pd.Series(y), ddof=ddof), left.corr(pd.Series(y), ddof=ddof)]
247
+ np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
248
+
249
+
250
+ @pytest.mark.parametrize("huge", [1e16, 1e20])
251
+ @pytest.mark.parametrize("order", list(permutations(range(3))))
252
+ def test_cov_corr_low_weight_extreme_does_not_make_result_depend_on_row_order(
253
+ huge, order
254
+ ):
255
+ # The large observation contributes to covariance, but using it as the
256
+ # centering origin must not erase the difference between 1 and 2.
257
+ x = np.array([huge, 1.0, 2.0])
258
+ y = np.array([0.0, 1.0, 2.0])
259
+ weights = np.array([1 / huge, 1.0, 1.0])
260
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
261
+ order = list(order)
262
+ left = mdf.MicroSeries(x[order], weights=weights[order])
263
+ right = pd.Series(y[order])
264
+ np.testing.assert_allclose(
265
+ [left.cov(right, ddof=0), left.corr(right, ddof=0)],
266
+ expected,
267
+ rtol=3e-15,
268
+ atol=0,
269
+ )
270
+
271
+
272
+ def test_cov_corr_smallest_common_positive_weight_cancels_from_population_moments():
273
+ # The common positive weight cancels: xy = 1, xx = yy = 2 for
274
+ # centered observations [-1, 0, 1] and [-1, 1, 0].
275
+ x, y = [1, 2, 3], [3, 5, 4]
276
+ weights = [np.nextafter(0.0, 1.0)] * 3
277
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
278
+ assert expected == (1 / 3, 0.5)
279
+ left = mdf.MicroSeries(x, weights=weights)
280
+ right = pd.Series(y)
281
+ np.testing.assert_allclose(
282
+ [left.cov(right, ddof=0), left.corr(right, ddof=0)],
283
+ expected,
284
+ rtol=3e-15,
285
+ atol=0,
286
+ )
287
+
288
+
289
+ @pytest.mark.parametrize(
290
+ "frequency", [np.nextafter(0.0, 1.0), np.finfo(float).tiny, 1.0]
291
+ )
292
+ @pytest.mark.parametrize("multipliers", [[1, 2, 3], [1, 1, 2]])
293
+ @pytest.mark.parametrize("order", list(permutations(range(3))))
294
+ def test_cov_corr_unequal_tiny_and_normal_weights_match_exact_moments(
295
+ frequency, multipliers, order
296
+ ):
297
+ x, y = np.array([1.0, 2.0, 3.0]), np.array([3.0, 5.0, 4.0])
298
+ weights = np.array(multipliers) * frequency
299
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
300
+ order = list(order)
301
+ left = mdf.MicroSeries(x[order], weights=weights[order])
302
+ right = pd.Series(y[order])
303
+ np.testing.assert_allclose(
304
+ [left.cov(right, ddof=0), left.corr(right, ddof=0)],
305
+ expected,
306
+ rtol=3e-15,
307
+ atol=0,
308
+ )
309
+
310
+
311
+ @pytest.mark.parametrize(
312
+ "x,y",
313
+ [
314
+ (np.array([1, 2, 3]) * 1e153, np.array([3, 5, 4]) * -1e153),
315
+ (np.array([1, 2, 3]) * 1e200, np.array([3, 5, 4]) * 1e-200),
316
+ ([1e12 - 2**-13, 1e12, 1e12 + 2**-13], [3, 5, 4]),
317
+ ],
318
+ )
319
+ def test_cov_corr_subnormal_weights_preserve_extreme_value_scales(x, y):
320
+ weights = np.array([1, 2, 3]) * np.nextafter(0.0, 1.0)
321
+ expected = exact_weighted_moments(x, y, weights, ddof=0)
322
+ left = mdf.MicroSeries(x, weights=weights)
323
+ right = pd.Series(y)
324
+ with np.errstate(over="raise", invalid="raise"):
325
+ actual = [left.cov(right, ddof=0), left.corr(right, ddof=0)]
326
+ np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
327
+
328
+
329
+ @pytest.mark.parametrize("frequency", [np.nextafter(0.0, 1.0), 0.5])
330
+ @pytest.mark.parametrize("ddof", [0, 1, 2])
331
+ def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof(
332
+ frequency, ddof
333
+ ):
334
+ x, y = [1, 2, 3], [3, 5, 4]
335
+ weights = np.array([1, 2, 3]) * frequency
336
+ left = mdf.MicroSeries(x, weights=weights)
337
+ right = pd.Series(y)
338
+ actual = [left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)]
339
+ if weights.sum() <= ddof:
340
+ assert np.isnan(actual).all()
341
+ else:
342
+ expected = exact_weighted_moments(x, y, weights, ddof=ddof)
343
+ np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.3.10
3
+ Version: 1.4.0
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -1,18 +1,19 @@
1
1
  microdf/__init__.py,sha256=sddmTcTZFSb1fjZkoLUR5TK7r5PEld4v2F92xBI5Sxs,641
2
2
  microdf/microdataframe.py,sha256=yqLUC43WBDT8Z-AQ7OfjKUoizehcUGBHgt7wg_BVlFI,44416
3
- microdf/microseries.py,sha256=uP4KOKluLsSFHRZf8aAyKKP4GvnXP6cyMFksncgowYU,38143
3
+ microdf/microseries.py,sha256=Y2k80dUOy8XxDZ6xN-YiUX04NXajILaLEc2isMahBCA,45004
4
4
  microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
5
5
  microdf/tests/test_aggregation_errors.py,sha256=9jJDiEyxMb2z1Zmj-o8AHDNp8LOputbEkAMefifnWaE,2013
6
6
  microdf/tests/test_dataframe_weight_storage.py,sha256=ngIsWa_QcBnnaLpAyIZhTgMxjv_7cK8Nbf7f4n29R4Q,1975
7
- microdf/tests/test_microseries_dataframe.py,sha256=r9S9z2RFgnFZFMsaT72N-zClT6jEL8YxQ_MUPAr3xi8,30969
7
+ microdf/tests/test_microseries_dataframe.py,sha256=vL0fg_NydU6a5myVVtXyHMr8eQOB_yyAkZYwLmoEnOA,30075
8
8
  microdf/tests/test_nullify_weights_index.py,sha256=kZgzMaZEa_PXbsor2S4E-6VRid3C3rcC9ufk0qa7mgY,341
9
9
  microdf/tests/test_pandas3_compatibility.py,sha256=A34Ni_WQ303sSNv-sqv5CGAQp54zj-ZSGAPEBHZslNI,8573
10
10
  microdf/tests/test_quantile_missing_values.py,sha256=lfntDvV2q7KH_CPVrXFJSQFoaGlc_OkRGhKwpxtDQtY,5327
11
11
  microdf/tests/test_serialization.py,sha256=a7pHL2hNiG5iJjRtfx3C1BCmgOZouAekiOwUxouAPfo,5083
12
12
  microdf/tests/test_sum_axes.py,sha256=N05ocwI5lLv2OgoaovRIqFIae-70356kZemRRet0ac8,8521
13
13
  microdf/tests/test_version_metadata.py,sha256=M1EabzHLKZZw3Djd6Zu2UuMQtDLV6rZ1zDrOU7W_jf0,227
14
- microdf_python-1.3.10.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
15
- microdf_python-1.3.10.dist-info/METADATA,sha256=ozvlZARMKjPYVIMiJZdTL6wgGi_SLv0GKLtLsqgdR_M,2306
16
- microdf_python-1.3.10.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
17
- microdf_python-1.3.10.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
18
- microdf_python-1.3.10.dist-info/RECORD,,
14
+ microdf/tests/test_weighted_cov_corr.py,sha256=LTnFhMWnb28f7L_9kOhPiV5UlLMAUbC9OlLIaZ1XgLs,13954
15
+ microdf_python-1.4.0.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
16
+ microdf_python-1.4.0.dist-info/METADATA,sha256=NjHgMRpJzMIc2G3IOjOOaY3YaLc-gBMStki-4LuY9d4,2305
17
+ microdf_python-1.4.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
18
+ microdf_python-1.4.0.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
19
+ microdf_python-1.4.0.dist-info/RECORD,,