microdf-python 1.3.10__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {microdf_python-1.3.10/microdf_python.egg-info → microdf_python-1.4.0}/PKG-INFO +1 -1
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/microseries.py +176 -28
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_microseries_dataframe.py +0 -22
- microdf_python-1.4.0/microdf/tests/test_weighted_cov_corr.py +343 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0/microdf_python.egg-info}/PKG-INFO +1 -1
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf_python.egg-info/SOURCES.txt +1 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/pyproject.toml +1 -1
- {microdf_python-1.3.10 → microdf_python-1.4.0}/LICENSE +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/README.md +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/__init__.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/microdataframe.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/conftest.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_aggregation_errors.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_dataframe_weight_storage.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_nullify_weights_index.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_pandas3_compatibility.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_quantile_missing_values.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_serialization.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_sum_axes.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_version_metadata.py +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf_python.egg-info/dependency_links.txt +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf_python.egg-info/requires.txt +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/microdf_python.egg-info/top_level.txt +0 -0
- {microdf_python-1.3.10 → microdf_python-1.4.0}/setup.cfg +0 -0
|
@@ -9,6 +9,41 @@ import pandas as pd
|
|
|
9
9
|
logger = logging.getLogger(__name__)
|
|
10
10
|
|
|
11
11
|
|
|
12
|
+
def _weighted_centered_vector(
|
|
13
|
+
values: np.ndarray, weights: np.ndarray
|
|
14
|
+
) -> tuple[np.ndarray, int]:
|
|
15
|
+
"""Return scaled sqrt-weighted deviations and their power-of-two
|
|
16
|
+
exponent."""
|
|
17
|
+
# Center relative to a maximum-weight observation: shifting by a low-weight
|
|
18
|
+
# extreme could erase differences among the influential observations.
|
|
19
|
+
# A relative mean also preserves nearby values at a large common offset.
|
|
20
|
+
reference = values[np.argmax(weights)]
|
|
21
|
+
with np.errstate(over="ignore"):
|
|
22
|
+
shifted = values - reference
|
|
23
|
+
exponent = 0
|
|
24
|
+
if np.isinf(shifted).any():
|
|
25
|
+
# Opposite finite extremes can overflow their difference. Halving is
|
|
26
|
+
# exact for those values; restore that factor in the final exponent.
|
|
27
|
+
shifted = values / 2 - reference / 2
|
|
28
|
+
exponent = 1
|
|
29
|
+
magnitude = np.max(np.abs(shifted))
|
|
30
|
+
if magnitude == 0:
|
|
31
|
+
return shifted, 0
|
|
32
|
+
_, shift = np.frexp(magnitude)
|
|
33
|
+
shifted = np.ldexp(shifted, -shift)
|
|
34
|
+
# Raise tiny mean weights by an exact common power of two so products
|
|
35
|
+
# with the scaled deviations do not underflow. Never scale down: that
|
|
36
|
+
# could discard small weights when frequencies span a wide range.
|
|
37
|
+
_, mean_weight_exponent = np.frexp(np.max(weights))
|
|
38
|
+
mean_weights = np.ldexp(weights, -min(int(mean_weight_exponent), 0))
|
|
39
|
+
shifted -= np.average(shifted, weights=mean_weights)
|
|
40
|
+
# Weight each vector before taking products, then scale again so squared
|
|
41
|
+
# deviations never accumulate raw frequencies at the original value scale.
|
|
42
|
+
shifted *= np.sqrt(weights)
|
|
43
|
+
_, weight_shift = np.frexp(np.max(np.abs(shifted)))
|
|
44
|
+
return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)
|
|
45
|
+
|
|
46
|
+
|
|
12
47
|
def _weighted_top_share(
|
|
13
48
|
values: np.ndarray, weights: np.ndarray, top_x_pct: float
|
|
14
49
|
) -> float:
|
|
@@ -315,39 +350,152 @@ class MicroSeries(pd.Series):
|
|
|
315
350
|
v = self._weighted_variance(ddof=ddof, skipna=skipna)
|
|
316
351
|
return float(np.sqrt(v)) if np.isfinite(v) else v
|
|
317
352
|
|
|
318
|
-
def
|
|
319
|
-
|
|
353
|
+
def _weighted_pair(
|
|
354
|
+
self,
|
|
355
|
+
other: pd.Series,
|
|
356
|
+
min_periods: Optional[int],
|
|
357
|
+
ddof: int,
|
|
358
|
+
skipna: bool,
|
|
359
|
+
) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
|
|
360
|
+
"""Align usable paired observations with their left weights."""
|
|
361
|
+
if not isinstance(other, pd.Series):
|
|
362
|
+
raise TypeError("other must be a pandas Series or MicroSeries")
|
|
363
|
+
if not isinstance(ddof, (int, np.integer)):
|
|
364
|
+
raise TypeError("ddof must be an integer")
|
|
365
|
+
if min_periods is None:
|
|
366
|
+
min_periods = 1
|
|
367
|
+
if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
|
|
368
|
+
raise ValueError("min_periods must be a nonnegative integer")
|
|
369
|
+
if len(self) == 0 or len(other) == 0:
|
|
370
|
+
return None
|
|
371
|
+
|
|
372
|
+
# Align left row positions so values and weights undergo exactly the
|
|
373
|
+
# same join, including pandas' expansion of duplicate index labels.
|
|
374
|
+
positions = pd.Series(np.arange(len(self)), index=self.index)
|
|
375
|
+
positions, right = positions.align(pd.Series(other), join="inner")
|
|
376
|
+
positions = positions.to_numpy(dtype=int)
|
|
377
|
+
x = (
|
|
378
|
+
pd.Series(self._values)
|
|
379
|
+
.iloc[positions]
|
|
380
|
+
.to_numpy(dtype=float, na_value=np.nan)
|
|
381
|
+
)
|
|
382
|
+
y = right.to_numpy(dtype=float, na_value=np.nan)
|
|
383
|
+
weights = np.asarray(self.weights, dtype=float)[positions]
|
|
384
|
+
if not np.isfinite(weights).all() or (weights < 0).any():
|
|
385
|
+
raise ValueError("frequency weights must be finite and nonnegative")
|
|
386
|
+
|
|
387
|
+
# Zero frequency means the row is absent, including for skipna=False.
|
|
388
|
+
positive = weights > 0
|
|
389
|
+
x, y, weights = x[positive], y[positive], weights[positive]
|
|
390
|
+
missing = np.isnan(x) | np.isnan(y)
|
|
391
|
+
if not skipna and missing.any():
|
|
392
|
+
return None
|
|
393
|
+
x, y, weights = x[~missing], y[~missing], weights[~missing]
|
|
394
|
+
total_weight = weights.sum()
|
|
395
|
+
if not np.isfinite(total_weight):
|
|
396
|
+
raise ValueError("the sum of frequency weights must be finite")
|
|
397
|
+
if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
|
|
398
|
+
return None
|
|
399
|
+
return (
|
|
400
|
+
x,
|
|
401
|
+
y,
|
|
402
|
+
weights,
|
|
403
|
+
float(total_weight - ddof),
|
|
404
|
+
)
|
|
320
405
|
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
406
|
+
def cov(
|
|
407
|
+
self,
|
|
408
|
+
other: pd.Series,
|
|
409
|
+
min_periods: Optional[int] = None,
|
|
410
|
+
ddof: int = 1,
|
|
411
|
+
*,
|
|
412
|
+
skipna: bool = True,
|
|
413
|
+
) -> float:
|
|
414
|
+
"""Calculate frequency-weighted covariance with another Series.
|
|
415
|
+
|
|
416
|
+
Observations align by index as in pandas, including its duplicate-
|
|
417
|
+
label join behavior. Only this Series' weights are used; weights on
|
|
418
|
+
another MicroSeries are ignored. Each aligned left weight must be
|
|
419
|
+
finite and nonnegative. Zero-weight rows are omitted.
|
|
420
|
+
|
|
421
|
+
Uses ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) - ddof)``.
|
|
422
|
+
Integer weights therefore match covariance on the replicated sample.
|
|
423
|
+
Missing values are removed pairwise before computing both means.
|
|
424
|
+
|
|
425
|
+
:param other: A pandas Series or MicroSeries to align by index.
|
|
426
|
+
:param min_periods: Minimum usable aligned row pairs, not the sum of
|
|
427
|
+
frequency weights. Defaults to 1.
|
|
428
|
+
:param ddof: Degrees of freedom subtracted from the weight total.
|
|
429
|
+
:param skipna: Drop pairs with a missing value. If False, any missing
|
|
430
|
+
value in a positive-weight aligned pair produces NaN.
|
|
431
|
+
:returns: Weighted covariance, or NaN for an empty or insufficient
|
|
432
|
+
sample (including a weight total no greater than ddof).
|
|
325
433
|
"""
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
434
|
+
pair = self._weighted_pair(other, min_periods, ddof, skipna)
|
|
435
|
+
if pair is None:
|
|
436
|
+
return np.nan
|
|
437
|
+
x, y, weights, denominator = pair
|
|
438
|
+
if not np.isfinite(x).all() or not np.isfinite(y).all():
|
|
439
|
+
return np.nan
|
|
440
|
+
x, x_exponent = _weighted_centered_vector(x, weights)
|
|
441
|
+
y, y_exponent = _weighted_centered_vector(y, weights)
|
|
442
|
+
# Combine exponents only after dividing out sum(weights) - ddof.
|
|
443
|
+
# Neither the original squared scale nor raw weighted sum need fit.
|
|
444
|
+
denominator, denominator_exponent = np.frexp(denominator)
|
|
445
|
+
return float(
|
|
446
|
+
np.ldexp(
|
|
447
|
+
np.sum(x * y) / denominator,
|
|
448
|
+
x_exponent + y_exponent - int(denominator_exponent),
|
|
449
|
+
)
|
|
333
450
|
)
|
|
334
|
-
return super().cov(other, *args, **kwargs)
|
|
335
|
-
|
|
336
|
-
def corr(self, other, *args, **kwargs):
|
|
337
|
-
"""Pandas ``corr`` — **unweighted**.
|
|
338
451
|
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
452
|
+
def corr(
|
|
453
|
+
self,
|
|
454
|
+
other: pd.Series,
|
|
455
|
+
method: str = "pearson",
|
|
456
|
+
min_periods: Optional[int] = None,
|
|
457
|
+
*,
|
|
458
|
+
ddof: int = 1,
|
|
459
|
+
skipna: bool = True,
|
|
460
|
+
) -> float:
|
|
461
|
+
"""Calculate frequency-weighted Pearson correlation.
|
|
462
|
+
|
|
463
|
+
Uses the same aligned pairs and left Series weights for covariance and
|
|
464
|
+
both variances. Weights on another MicroSeries are ignored. Weights
|
|
465
|
+
must be finite and nonnegative; zero-weight rows are omitted. Other
|
|
466
|
+
correlation methods, including callables, are unsupported.
|
|
467
|
+
|
|
468
|
+
:param other: A pandas Series or MicroSeries to align by index.
|
|
469
|
+
:param method: Only "pearson" is supported.
|
|
470
|
+
:param min_periods: Minimum usable aligned row pairs, not frequency
|
|
471
|
+
weight total. Defaults to 1.
|
|
472
|
+
:param ddof: Degrees of freedom for all three moments. It cancels from
|
|
473
|
+
the correlation but the weight total must exceed it.
|
|
474
|
+
:param skipna: Drop pairs with a missing value. If False, any missing
|
|
475
|
+
value in a positive-weight aligned pair produces NaN.
|
|
476
|
+
:returns: Weighted correlation, or NaN for an empty, insufficient, or
|
|
477
|
+
constant sample.
|
|
342
478
|
"""
|
|
343
|
-
|
|
344
|
-
"
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
)
|
|
350
|
-
|
|
479
|
+
if method != "pearson":
|
|
480
|
+
raise ValueError("weighted correlation only supports method='pearson'")
|
|
481
|
+
pair = self._weighted_pair(other, min_periods, ddof, skipna)
|
|
482
|
+
if pair is None:
|
|
483
|
+
return np.nan
|
|
484
|
+
x, y, weights, _ = pair
|
|
485
|
+
if not np.isfinite(x).all() or not np.isfinite(y).all():
|
|
486
|
+
return np.nan
|
|
487
|
+
# A weighted mean can round away from identical decimal inputs.
|
|
488
|
+
# Check the retained observations exactly before subtracting it.
|
|
489
|
+
if (x == x[0]).all() or (y == y[0]).all():
|
|
490
|
+
return np.nan
|
|
491
|
+
x, _ = _weighted_centered_vector(x, weights)
|
|
492
|
+
y, _ = _weighted_centered_vector(y, weights)
|
|
493
|
+
x_ss = np.sum(x * x)
|
|
494
|
+
y_ss = np.sum(y * y)
|
|
495
|
+
if x_ss == 0 or y_ss == 0:
|
|
496
|
+
return np.nan
|
|
497
|
+
result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
|
|
498
|
+
return float(np.clip(result, -1.0, 1.0))
|
|
351
499
|
|
|
352
500
|
def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
|
|
353
501
|
"""Calculates weighted quantiles of the MicroSeries.
|
|
@@ -744,28 +744,6 @@ def test_std_var_are_weighted() -> None:
|
|
|
744
744
|
)
|
|
745
745
|
|
|
746
746
|
|
|
747
|
-
def test_cov_corr_warn_when_fallthrough() -> None:
|
|
748
|
-
"""Regression: cov/corr silently returned unweighted pandas values.
|
|
749
|
-
|
|
750
|
-
They still fall through to pandas (a weighted impl is a separate issue) but
|
|
751
|
-
now emit a UserWarning so callers aren't misled.
|
|
752
|
-
"""
|
|
753
|
-
s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
|
|
754
|
-
s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1])
|
|
755
|
-
|
|
756
|
-
with warnings.catch_warnings(record=True) as w:
|
|
757
|
-
warnings.simplefilter("always")
|
|
758
|
-
_ = s1.cov(s2)
|
|
759
|
-
msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
|
|
760
|
-
assert any("unweighted" in m.lower() for m in msgs)
|
|
761
|
-
|
|
762
|
-
with warnings.catch_warnings(record=True) as w:
|
|
763
|
-
warnings.simplefilter("always")
|
|
764
|
-
_ = s1.corr(s2)
|
|
765
|
-
msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
|
|
766
|
-
assert any("unweighted" in m.lower() for m in msgs)
|
|
767
|
-
|
|
768
|
-
|
|
769
747
|
def test_count_skips_nan_by_default() -> None:
|
|
770
748
|
"""Regression: ``count()`` included NaN-row weight, contrary to pandas.
|
|
771
749
|
|
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
import warnings
|
|
2
|
+
from decimal import Decimal, localcontext
|
|
3
|
+
from fractions import Fraction
|
|
4
|
+
from itertools import permutations
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
import microdf as mdf
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def replicated_moments(x, y, weights, ddof=1):
|
|
14
|
+
"""Independent frequency-weight oracle: expand to an ordinary sample."""
|
|
15
|
+
repeated_x = np.repeat(np.asarray(x, dtype=float), weights)
|
|
16
|
+
repeated_y = np.repeat(np.asarray(y, dtype=float), weights)
|
|
17
|
+
return (
|
|
18
|
+
np.cov(repeated_x, repeated_y, ddof=ddof)[0, 1],
|
|
19
|
+
np.corrcoef(repeated_x, repeated_y)[0, 1],
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
24
|
+
def test_cov_corr_match_replicated_frequency_sample(ddof):
|
|
25
|
+
x, y, weights = [1, 4, 8], [5, 2, 9], [1, 3, 2]
|
|
26
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
27
|
+
right = pd.Series(y)
|
|
28
|
+
expected_cov, expected_corr = replicated_moments(x, y, weights, ddof)
|
|
29
|
+
with warnings.catch_warnings(record=True) as caught:
|
|
30
|
+
warnings.simplefilter("always")
|
|
31
|
+
assert left.cov(right, ddof=ddof) == pytest.approx(expected_cov)
|
|
32
|
+
assert left.corr(right, ddof=ddof) == pytest.approx(expected_corr)
|
|
33
|
+
assert not any("unweighted" in str(item.message).lower() for item in caught)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_cov_corr_align_indices_and_use_only_left_weights():
|
|
37
|
+
left = mdf.MicroSeries([1, 4, 8], index=["a", "b", "c"], weights=[1, 3, 2])
|
|
38
|
+
right = mdf.MicroSeries(
|
|
39
|
+
[9, 5, 2, 100], index=["c", "a", "b", "d"], weights=[99, 1, 1, 9]
|
|
40
|
+
)
|
|
41
|
+
expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2])
|
|
42
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
43
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
44
|
+
assert right.weights.tolist() == [99, 1, 1, 9]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def test_cov_corr_use_same_pairwise_nonmissing_sample():
|
|
48
|
+
left = mdf.MicroSeries(
|
|
49
|
+
[1, np.nan, 5, 9, 20], index=list("abcde"), weights=[1, 9, 2, 3, 7]
|
|
50
|
+
)
|
|
51
|
+
right = pd.Series([2, 4, np.nan, 8, 30], index=list("abcdf"))
|
|
52
|
+
expected_cov, expected_corr = replicated_moments([1, 9], [2, 8], [1, 3])
|
|
53
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
54
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
55
|
+
assert np.isnan(left.cov(right, skipna=False))
|
|
56
|
+
assert np.isnan(left.corr(right, skipna=False))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@pytest.mark.parametrize("same_index", [True, False])
|
|
60
|
+
def test_cov_corr_follow_pandas_duplicate_index_alignment(same_index):
|
|
61
|
+
left = mdf.MicroSeries([1, 4, 7], index=["a", "a", "b"], weights=[1, 3, 2])
|
|
62
|
+
if same_index:
|
|
63
|
+
right = pd.Series([2, 3, 8], index=["a", "a", "b"])
|
|
64
|
+
x, y, weights = [1, 4, 7], [2, 3, 8], [1, 3, 2]
|
|
65
|
+
else:
|
|
66
|
+
right = pd.Series([2, 5, 8], index=["a", "b", "b"])
|
|
67
|
+
# The shared a/b labels join, repeating the left row weight per pair.
|
|
68
|
+
x, y, weights = [1, 4, 7, 7], [2, 2, 5, 8], [1, 3, 2, 2]
|
|
69
|
+
expected_cov, expected_corr = replicated_moments(x, y, weights)
|
|
70
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
71
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_zero_weight_rows_do_not_enter_pairwise_sample():
|
|
75
|
+
left = mdf.MicroSeries([1, np.nan, 5, 1000], weights=[2, 0, 1, 0])
|
|
76
|
+
right = pd.Series([3, np.nan, 9, -1000])
|
|
77
|
+
expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [2, 1])
|
|
78
|
+
assert left.cov(right, skipna=False) == pytest.approx(expected_cov)
|
|
79
|
+
assert left.corr(right, skipna=False) == pytest.approx(expected_corr)
|
|
80
|
+
assert np.isnan(left.cov(right, min_periods=3))
|
|
81
|
+
assert np.isnan(left.corr(right, min_periods=3))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_min_periods_counts_usable_rows_separately_from_frequency_weight():
|
|
85
|
+
left = mdf.MicroSeries([1, 5], weights=[10, 20])
|
|
86
|
+
right = pd.Series([3, 9])
|
|
87
|
+
expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [10, 20], ddof=0)
|
|
88
|
+
# Existing pandas positional arguments retain their order.
|
|
89
|
+
assert left.cov(right, 2, 0) == pytest.approx(expected_cov)
|
|
90
|
+
assert left.corr(right, "pearson", 2, ddof=0) == pytest.approx(expected_corr)
|
|
91
|
+
assert np.isnan(left.cov(right, min_periods=3))
|
|
92
|
+
assert np.isnan(left.corr(right, min_periods=3))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@pytest.mark.parametrize(
|
|
96
|
+
"values,other,weights",
|
|
97
|
+
[([], [], []), ([np.nan], [1], [3]), ([1], [2], [0]), ([1], [2], [1])],
|
|
98
|
+
)
|
|
99
|
+
def test_cov_corr_return_nan_when_sample_is_insufficient(values, other, weights):
|
|
100
|
+
left = mdf.MicroSeries(values, weights=weights, dtype=float)
|
|
101
|
+
right = pd.Series(other, dtype=float)
|
|
102
|
+
assert np.isnan(left.cov(right))
|
|
103
|
+
assert np.isnan(left.corr(right))
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_cov_corr_no_index_overlap():
|
|
107
|
+
left = mdf.MicroSeries([1, 2], index=["a", "b"], weights=[1, 2])
|
|
108
|
+
right = pd.Series([3, 4], index=["c", "d"])
|
|
109
|
+
assert np.isnan(left.cov(right))
|
|
110
|
+
assert np.isnan(left.corr(right))
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_cov_corr_constant_and_frequency_singleton():
|
|
114
|
+
left = mdf.MicroSeries([4, 4], weights=[2, 3])
|
|
115
|
+
right = pd.Series([1, 5])
|
|
116
|
+
assert left.cov(right) == 0
|
|
117
|
+
assert np.isnan(left.corr(right))
|
|
118
|
+
singleton = mdf.MicroSeries([4], weights=[3])
|
|
119
|
+
assert singleton.cov(pd.Series([2])) == 0
|
|
120
|
+
assert np.isnan(singleton.corr(pd.Series([2])))
|
|
121
|
+
assert np.isnan(singleton.cov(pd.Series([2]), ddof=3))
|
|
122
|
+
assert np.isnan(singleton.corr(pd.Series([2]), ddof=3))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def test_cov_corr_nullable_numeric_data():
|
|
126
|
+
left = mdf.MicroSeries(pd.Series([1, pd.NA, 4], dtype="Int64"), weights=[2, 9, 3])
|
|
127
|
+
right = pd.Series([3, 8, 7], dtype="Float64")
|
|
128
|
+
expected_cov, expected_corr = replicated_moments([1, 4], [3, 7], [2, 3])
|
|
129
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
130
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@pytest.mark.parametrize("method", ["spearman", "kendall", lambda x, y: 1.0])
|
|
134
|
+
def test_non_pearson_methods_are_explicitly_unsupported(method):
|
|
135
|
+
left = mdf.MicroSeries([1, 2, 3], weights=[1, 2, 3])
|
|
136
|
+
with pytest.raises(ValueError, match="pearson"):
|
|
137
|
+
left.corr(pd.Series([4, 2, 5]), method=method)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@pytest.mark.parametrize("weights", [[1, -1], [1, np.nan], [1, np.inf]])
|
|
141
|
+
def test_cov_corr_reject_invalid_frequency_weights(weights):
|
|
142
|
+
left = mdf.MicroSeries([1, 2], weights=weights)
|
|
143
|
+
for method in (left.cov, left.corr):
|
|
144
|
+
with pytest.raises(ValueError, match="weights"):
|
|
145
|
+
method(pd.Series([2, 4]))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def test_binary_statistics_not_in_dataframe_aggregation_factories():
|
|
149
|
+
assert "cov" not in mdf.MicroSeries.FUNCTIONS
|
|
150
|
+
assert "corr" not in mdf.MicroSeries.FUNCTIONS
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@pytest.mark.parametrize(
|
|
154
|
+
"x,y",
|
|
155
|
+
[([0.1, 0.1], [0, 1]), ([0, 1], [0.1, 0.1]), ([0.1, 0.1], [0.1, 0.1])],
|
|
156
|
+
)
|
|
157
|
+
@pytest.mark.parametrize("with_filtered_rows", [False, True])
|
|
158
|
+
def test_corr_exact_decimal_constants_return_nan(x, y, with_filtered_rows):
|
|
159
|
+
# Unequal weights can round the mean away from the identical 0.1 values.
|
|
160
|
+
# Constant detection must inspect usable observations before centering.
|
|
161
|
+
weights = [1, 2]
|
|
162
|
+
if with_filtered_rows:
|
|
163
|
+
x = x + [9, np.nan]
|
|
164
|
+
y = y + [7, 4]
|
|
165
|
+
weights = weights + [0, 3]
|
|
166
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
167
|
+
assert np.isnan(left.corr(pd.Series(y)))
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_corr_does_not_treat_nearby_distinct_values_as_constant():
|
|
171
|
+
x = [0.1, np.nextafter(0.1, np.inf)]
|
|
172
|
+
left = mdf.MicroSeries(x, weights=[1, 2])
|
|
173
|
+
assert np.isfinite(left.corr(pd.Series([0, 1])))
|
|
174
|
+
assert np.isfinite(mdf.MicroSeries([0, 1], weights=[1, 2]).corr(pd.Series(x)))
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def exact_weighted_moments(x, y, weights, ddof=1):
|
|
178
|
+
"""Compute moments of the actual input floats with exact rational
|
|
179
|
+
arithmetic."""
|
|
180
|
+
x, y, weights = [
|
|
181
|
+
[Fraction(float(value)) for value in values] for values in (x, y, weights)
|
|
182
|
+
]
|
|
183
|
+
total = sum(weights)
|
|
184
|
+
xmean = sum(w * value for w, value in zip(weights, x)) / total
|
|
185
|
+
ymean = sum(w * value for w, value in zip(weights, y)) / total
|
|
186
|
+
xy = sum(w * (a - xmean) * (b - ymean) for a, b, w in zip(x, y, weights))
|
|
187
|
+
xx = sum(w * (value - xmean) ** 2 for value, w in zip(x, weights))
|
|
188
|
+
yy = sum(w * (value - ymean) ** 2 for value, w in zip(y, weights))
|
|
189
|
+
with localcontext() as context:
|
|
190
|
+
context.prec = 100
|
|
191
|
+
product = xx * yy
|
|
192
|
+
correlation = (Decimal(xy.numerator) / Decimal(xy.denominator)) / (
|
|
193
|
+
Decimal(product.numerator) / Decimal(product.denominator)
|
|
194
|
+
).sqrt()
|
|
195
|
+
return float(xy / (total - ddof)), float(correlation)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
199
|
+
@pytest.mark.parametrize("swap", [False, True])
|
|
200
|
+
def test_cov_corr_preserve_small_differences_at_large_offsets(ddof, swap):
|
|
201
|
+
# Two distinct points are perfectly linear even two float steps apart.
|
|
202
|
+
x = np.array([1e12 - 2**-13, 1e12 + 2**-13])
|
|
203
|
+
y = np.array([0.0, 1.0])
|
|
204
|
+
if swap:
|
|
205
|
+
x, y = y, x
|
|
206
|
+
weights = [1, 2]
|
|
207
|
+
expected = exact_weighted_moments(x, y, weights, ddof)
|
|
208
|
+
assert expected[1] == 1.0
|
|
209
|
+
for shifted_x, shifted_y in [(x, y), (x - x[0], y - y[0])]:
|
|
210
|
+
left = mdf.MicroSeries(shifted_x, weights=weights)
|
|
211
|
+
right = pd.Series(shifted_y)
|
|
212
|
+
np.testing.assert_allclose(
|
|
213
|
+
[left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)],
|
|
214
|
+
expected,
|
|
215
|
+
rtol=2e-15,
|
|
216
|
+
atol=0,
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@pytest.mark.parametrize("frequency", [1, 1_000_000])
|
|
221
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
222
|
+
def test_cov_corr_large_finite_values_do_not_overflow_raw_frequencies(frequency, ddof):
|
|
223
|
+
x = np.array([1.0, 2.0, 3.0]) * 1e153
|
|
224
|
+
weights = [frequency] * 3
|
|
225
|
+
expected = exact_weighted_moments(x, x, weights, ddof)
|
|
226
|
+
assert np.isfinite(expected).all()
|
|
227
|
+
assert expected[1] == 1.0
|
|
228
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
229
|
+
with np.errstate(over="raise", invalid="raise"):
|
|
230
|
+
actual = [left.cov(pd.Series(x), ddof=ddof), left.corr(pd.Series(x), ddof=ddof)]
|
|
231
|
+
np.testing.assert_allclose(actual, expected, rtol=2e-15, atol=0)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
@pytest.mark.parametrize("frequency", [0.5, 1, 1_000_000])
|
|
235
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
236
|
+
@pytest.mark.parametrize("scales", [(1.0, 1.0), (1e153, -1e153), (1e200, 1e-200)])
|
|
237
|
+
def test_cov_corr_preserve_frequency_correction_across_value_scales(
|
|
238
|
+
frequency, ddof, scales
|
|
239
|
+
):
|
|
240
|
+
x = np.array([1.0, 4.0, 8.0]) * scales[0]
|
|
241
|
+
y = np.array([5.0, 2.0, 9.0]) * scales[1]
|
|
242
|
+
weights = np.array([1, 3, 2]) * frequency
|
|
243
|
+
expected = exact_weighted_moments(x, y, weights, ddof)
|
|
244
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
245
|
+
with np.errstate(over="raise", invalid="raise"):
|
|
246
|
+
actual = [left.cov(pd.Series(y), ddof=ddof), left.corr(pd.Series(y), ddof=ddof)]
|
|
247
|
+
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
@pytest.mark.parametrize("huge", [1e16, 1e20])
|
|
251
|
+
@pytest.mark.parametrize("order", list(permutations(range(3))))
|
|
252
|
+
def test_cov_corr_low_weight_extreme_does_not_make_result_depend_on_row_order(
|
|
253
|
+
huge, order
|
|
254
|
+
):
|
|
255
|
+
# The large observation contributes to covariance, but using it as the
|
|
256
|
+
# centering origin must not erase the difference between 1 and 2.
|
|
257
|
+
x = np.array([huge, 1.0, 2.0])
|
|
258
|
+
y = np.array([0.0, 1.0, 2.0])
|
|
259
|
+
weights = np.array([1 / huge, 1.0, 1.0])
|
|
260
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
261
|
+
order = list(order)
|
|
262
|
+
left = mdf.MicroSeries(x[order], weights=weights[order])
|
|
263
|
+
right = pd.Series(y[order])
|
|
264
|
+
np.testing.assert_allclose(
|
|
265
|
+
[left.cov(right, ddof=0), left.corr(right, ddof=0)],
|
|
266
|
+
expected,
|
|
267
|
+
rtol=3e-15,
|
|
268
|
+
atol=0,
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def test_cov_corr_smallest_common_positive_weight_cancels_from_population_moments():
|
|
273
|
+
# The common positive weight cancels: xy = 1, xx = yy = 2 for
|
|
274
|
+
# centered observations [-1, 0, 1] and [-1, 1, 0].
|
|
275
|
+
x, y = [1, 2, 3], [3, 5, 4]
|
|
276
|
+
weights = [np.nextafter(0.0, 1.0)] * 3
|
|
277
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
278
|
+
assert expected == (1 / 3, 0.5)
|
|
279
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
280
|
+
right = pd.Series(y)
|
|
281
|
+
np.testing.assert_allclose(
|
|
282
|
+
[left.cov(right, ddof=0), left.corr(right, ddof=0)],
|
|
283
|
+
expected,
|
|
284
|
+
rtol=3e-15,
|
|
285
|
+
atol=0,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
@pytest.mark.parametrize(
|
|
290
|
+
"frequency", [np.nextafter(0.0, 1.0), np.finfo(float).tiny, 1.0]
|
|
291
|
+
)
|
|
292
|
+
@pytest.mark.parametrize("multipliers", [[1, 2, 3], [1, 1, 2]])
|
|
293
|
+
@pytest.mark.parametrize("order", list(permutations(range(3))))
|
|
294
|
+
def test_cov_corr_unequal_tiny_and_normal_weights_match_exact_moments(
|
|
295
|
+
frequency, multipliers, order
|
|
296
|
+
):
|
|
297
|
+
x, y = np.array([1.0, 2.0, 3.0]), np.array([3.0, 5.0, 4.0])
|
|
298
|
+
weights = np.array(multipliers) * frequency
|
|
299
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
300
|
+
order = list(order)
|
|
301
|
+
left = mdf.MicroSeries(x[order], weights=weights[order])
|
|
302
|
+
right = pd.Series(y[order])
|
|
303
|
+
np.testing.assert_allclose(
|
|
304
|
+
[left.cov(right, ddof=0), left.corr(right, ddof=0)],
|
|
305
|
+
expected,
|
|
306
|
+
rtol=3e-15,
|
|
307
|
+
atol=0,
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
@pytest.mark.parametrize(
|
|
312
|
+
"x,y",
|
|
313
|
+
[
|
|
314
|
+
(np.array([1, 2, 3]) * 1e153, np.array([3, 5, 4]) * -1e153),
|
|
315
|
+
(np.array([1, 2, 3]) * 1e200, np.array([3, 5, 4]) * 1e-200),
|
|
316
|
+
([1e12 - 2**-13, 1e12, 1e12 + 2**-13], [3, 5, 4]),
|
|
317
|
+
],
|
|
318
|
+
)
|
|
319
|
+
def test_cov_corr_subnormal_weights_preserve_extreme_value_scales(x, y):
|
|
320
|
+
weights = np.array([1, 2, 3]) * np.nextafter(0.0, 1.0)
|
|
321
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
322
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
323
|
+
right = pd.Series(y)
|
|
324
|
+
with np.errstate(over="raise", invalid="raise"):
|
|
325
|
+
actual = [left.cov(right, ddof=0), left.corr(right, ddof=0)]
|
|
326
|
+
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
@pytest.mark.parametrize("frequency", [np.nextafter(0.0, 1.0), 0.5])
|
|
330
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
331
|
+
def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof(
|
|
332
|
+
frequency, ddof
|
|
333
|
+
):
|
|
334
|
+
x, y = [1, 2, 3], [3, 5, 4]
|
|
335
|
+
weights = np.array([1, 2, 3]) * frequency
|
|
336
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
337
|
+
right = pd.Series(y)
|
|
338
|
+
actual = [left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)]
|
|
339
|
+
if weights.sum() <= ddof:
|
|
340
|
+
assert np.isnan(actual).all()
|
|
341
|
+
else:
|
|
342
|
+
expected = exact_weighted_moments(x, y, weights, ddof=ddof)
|
|
343
|
+
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
@@ -14,6 +14,7 @@ microdf/tests/test_quantile_missing_values.py
|
|
|
14
14
|
microdf/tests/test_serialization.py
|
|
15
15
|
microdf/tests/test_sum_axes.py
|
|
16
16
|
microdf/tests/test_version_metadata.py
|
|
17
|
+
microdf/tests/test_weighted_cov_corr.py
|
|
17
18
|
microdf_python.egg-info/PKG-INFO
|
|
18
19
|
microdf_python.egg-info/SOURCES.txt
|
|
19
20
|
microdf_python.egg-info/dependency_links.txt
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_dataframe_weight_storage.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{microdf_python-1.3.10 → microdf_python-1.4.0}/microdf/tests/test_quantile_missing_values.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|