microdf-python 1.5.6__py3-none-any.whl → 1.5.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- microdf/microdataframe.py +101 -10
- microdf/microseries.py +83 -54
- microdf/tests/test_weight_propagation.py +89 -54
- microdf/tests/test_weighted_cov_corr.py +105 -0
- {microdf_python-1.5.6.dist-info → microdf_python-1.5.8.dist-info}/METADATA +1 -1
- {microdf_python-1.5.6.dist-info → microdf_python-1.5.8.dist-info}/RECORD +9 -9
- {microdf_python-1.5.6.dist-info → microdf_python-1.5.8.dist-info}/WHEEL +0 -0
- {microdf_python-1.5.6.dist-info → microdf_python-1.5.8.dist-info}/licenses/LICENSE +0 -0
- {microdf_python-1.5.6.dist-info → microdf_python-1.5.8.dist-info}/top_level.txt +0 -0
microdf/microdataframe.py
CHANGED
|
@@ -7,7 +7,15 @@ from typing import Callable, List, Optional, Union
|
|
|
7
7
|
import numpy as np
|
|
8
8
|
import pandas as pd
|
|
9
9
|
|
|
10
|
-
from microdf.microseries import
|
|
10
|
+
from microdf.microseries import (
|
|
11
|
+
MicroSeries,
|
|
12
|
+
MicroSeriesGroupBy,
|
|
13
|
+
_pair_options,
|
|
14
|
+
_usable_pair,
|
|
15
|
+
_validate_frequency_weights,
|
|
16
|
+
_weighted_correlation,
|
|
17
|
+
_weighted_covariance,
|
|
18
|
+
)
|
|
11
19
|
from microdf._weights import (
|
|
12
20
|
WeightPropagationMixin,
|
|
13
21
|
aligned_weights,
|
|
@@ -74,16 +82,99 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
|
|
|
74
82
|
super().__finalize__(other, method=method, **kwargs)
|
|
75
83
|
return finalize_weights(self, other, method, previous)
|
|
76
84
|
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
85
|
+
def cov(
|
|
86
|
+
self,
|
|
87
|
+
min_periods: Optional[int] = None,
|
|
88
|
+
ddof: int = 1,
|
|
89
|
+
numeric_only: bool = False,
|
|
90
|
+
) -> pd.DataFrame:
|
|
91
|
+
"""Pairwise frequency-weighted covariance of the columns.
|
|
92
|
+
|
|
93
|
+
Every cell uses the estimator of :meth:`MicroSeries.cov` with this
|
|
94
|
+
frame's weights: ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) -
|
|
95
|
+
ddof)`` over the rows where both columns are present and the weight
|
|
96
|
+
is positive. Integer weights therefore match ``pandas.DataFrame.cov``
|
|
97
|
+
on the replicated sample. Missing values are removed pairwise, so
|
|
98
|
+
each cell can use a different set of rows, as in pandas.
|
|
99
|
+
|
|
100
|
+
The result is a plain ``pandas.DataFrame``: it summarises columns,
|
|
101
|
+
so it carries no row weights.
|
|
102
|
+
|
|
103
|
+
:param min_periods: Minimum usable row pairs per cell, not the sum
|
|
104
|
+
of frequency weights. Cells with fewer are NaN. Defaults to 1.
|
|
105
|
+
:param ddof: Degrees of freedom subtracted from the weight total.
|
|
106
|
+
:param numeric_only: Use only numeric columns. Otherwise every column
|
|
107
|
+
is converted to float, raising the error pandas raises.
|
|
108
|
+
:returns: Covariance matrix indexed by column in both directions.
|
|
109
|
+
"""
|
|
110
|
+
return self._weighted_pairwise(
|
|
111
|
+
"cov", min_periods=min_periods, ddof=ddof, numeric_only=numeric_only
|
|
112
|
+
)
|
|
82
113
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
114
|
+
def corr(
|
|
115
|
+
self,
|
|
116
|
+
method: str = "pearson",
|
|
117
|
+
min_periods: int = 1,
|
|
118
|
+
numeric_only: bool = False,
|
|
119
|
+
) -> pd.DataFrame:
|
|
120
|
+
"""Pairwise frequency-weighted Pearson correlation of the columns.
|
|
121
|
+
|
|
122
|
+
Every cell uses the estimator of :meth:`MicroSeries.corr` with this
|
|
123
|
+
frame's weights over the rows where both columns are present and the
|
|
124
|
+
weight is positive. Constant columns give NaN. Only ``"pearson"`` is
|
|
125
|
+
supported, as on :meth:`MicroSeries.corr`; for an unweighted rank
|
|
126
|
+
correlation convert to ``pandas.DataFrame`` first.
|
|
127
|
+
|
|
128
|
+
The result is a plain ``pandas.DataFrame``: it summarises columns, so
|
|
129
|
+
it carries no row weights.
|
|
130
|
+
|
|
131
|
+
:param method: Only "pearson" is supported.
|
|
132
|
+
:param min_periods: Minimum usable row pairs per cell, not the sum of
|
|
133
|
+
frequency weights. Cells with fewer are NaN.
|
|
134
|
+
:param numeric_only: Use only numeric columns. Otherwise every column
|
|
135
|
+
is converted to float, raising the error pandas raises.
|
|
136
|
+
:returns: Correlation matrix indexed by column in both directions.
|
|
137
|
+
"""
|
|
138
|
+
if method != "pearson":
|
|
139
|
+
raise ValueError("weighted correlation only supports method='pearson'")
|
|
140
|
+
return self._weighted_pairwise(
|
|
141
|
+
"corr", min_periods=min_periods, ddof=1, numeric_only=numeric_only
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
def _weighted_pairwise(
|
|
145
|
+
self,
|
|
146
|
+
statistic: str,
|
|
147
|
+
*,
|
|
148
|
+
min_periods: Optional[int],
|
|
149
|
+
ddof: int,
|
|
150
|
+
numeric_only: bool,
|
|
151
|
+
) -> pd.DataFrame:
|
|
152
|
+
"""Fill a symmetric column matrix one usable pair at a time."""
|
|
153
|
+
min_periods, ddof = _pair_options(min_periods, ddof)
|
|
154
|
+
data = self._get_numeric_data() if numeric_only else self
|
|
155
|
+
frame = pd.DataFrame(data, copy=False)
|
|
156
|
+
# The conversion pandas uses, so non-numeric columns raise its error.
|
|
157
|
+
values = frame.to_numpy(dtype=float, na_value=np.nan)
|
|
158
|
+
weights = np.asarray(self.weights, dtype=float)
|
|
159
|
+
_validate_frequency_weights(weights)
|
|
160
|
+
columns = frame.columns
|
|
161
|
+
matrix = np.full((len(columns), len(columns)), np.nan)
|
|
162
|
+
for i in range(len(columns)):
|
|
163
|
+
for j in range(i, len(columns)):
|
|
164
|
+
pair = _usable_pair(
|
|
165
|
+
values[:, i], values[:, j], weights, min_periods, ddof, True
|
|
166
|
+
)
|
|
167
|
+
if pair is None:
|
|
168
|
+
continue
|
|
169
|
+
x, y, pair_weights, denominator = pair
|
|
170
|
+
if statistic == "cov":
|
|
171
|
+
cell = _weighted_covariance(x, y, pair_weights, denominator)
|
|
172
|
+
else:
|
|
173
|
+
cell = _weighted_correlation(x, y, pair_weights)
|
|
174
|
+
matrix[i, j] = matrix[j, i] = cell
|
|
175
|
+
result = pd.DataFrame(matrix, index=columns, columns=columns)
|
|
176
|
+
# Column summaries have no observation weights, even if labels match.
|
|
177
|
+
return pd.DataFrame.__finalize__(result, self, method=statistic)
|
|
87
178
|
|
|
88
179
|
def __setstate__(self, state) -> None:
|
|
89
180
|
"""Restore a pickled MicroDataFrame.
|
microdf/microseries.py
CHANGED
|
@@ -51,6 +51,84 @@ def _weighted_centered_vector(
|
|
|
51
51
|
return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)
|
|
52
52
|
|
|
53
53
|
|
|
54
|
+
def _pair_options(min_periods: Optional[int], ddof: int) -> tuple[int, int]:
|
|
55
|
+
"""Validate the row minimum and degrees of freedom for paired moments."""
|
|
56
|
+
if not isinstance(ddof, (int, np.integer)):
|
|
57
|
+
raise TypeError("ddof must be an integer")
|
|
58
|
+
if min_periods is None:
|
|
59
|
+
min_periods = 1
|
|
60
|
+
if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
|
|
61
|
+
raise ValueError("min_periods must be a nonnegative integer")
|
|
62
|
+
return int(min_periods), int(ddof)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _validate_frequency_weights(weights: np.ndarray) -> None:
|
|
66
|
+
"""Reject weights that cannot be frequencies."""
|
|
67
|
+
if not np.isfinite(weights).all() or (weights < 0).any():
|
|
68
|
+
raise ValueError("frequency weights must be finite and nonnegative")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _usable_pair(
|
|
72
|
+
x: np.ndarray,
|
|
73
|
+
y: np.ndarray,
|
|
74
|
+
weights: np.ndarray,
|
|
75
|
+
min_periods: int,
|
|
76
|
+
ddof: int,
|
|
77
|
+
skipna: bool,
|
|
78
|
+
) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
|
|
79
|
+
"""Keep the present, positively weighted rows of an aligned pair."""
|
|
80
|
+
# Zero frequency means the row is absent, including for skipna=False.
|
|
81
|
+
positive = weights > 0
|
|
82
|
+
x, y, weights = x[positive], y[positive], weights[positive]
|
|
83
|
+
missing = np.isnan(x) | np.isnan(y)
|
|
84
|
+
if not skipna and missing.any():
|
|
85
|
+
return None
|
|
86
|
+
x, y, weights = x[~missing], y[~missing], weights[~missing]
|
|
87
|
+
total_weight = weights.sum()
|
|
88
|
+
if not np.isfinite(total_weight):
|
|
89
|
+
raise ValueError("the sum of frequency weights must be finite")
|
|
90
|
+
if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
|
|
91
|
+
return None
|
|
92
|
+
return x, y, weights, float(total_weight - ddof)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _weighted_covariance(
|
|
96
|
+
x: np.ndarray, y: np.ndarray, weights: np.ndarray, denominator: float
|
|
97
|
+
) -> float:
|
|
98
|
+
"""Frequency-weighted covariance of a usable pair."""
|
|
99
|
+
if not np.isfinite(x).all() or not np.isfinite(y).all():
|
|
100
|
+
return np.nan
|
|
101
|
+
x, x_exponent = _weighted_centered_vector(x, weights)
|
|
102
|
+
y, y_exponent = _weighted_centered_vector(y, weights)
|
|
103
|
+
# Combine exponents only after dividing out sum(weights) - ddof.
|
|
104
|
+
# Neither the original squared scale nor raw weighted sum need fit.
|
|
105
|
+
denominator, denominator_exponent = np.frexp(denominator)
|
|
106
|
+
return float(
|
|
107
|
+
np.ldexp(
|
|
108
|
+
np.sum(x * y) / denominator,
|
|
109
|
+
x_exponent + y_exponent - int(denominator_exponent),
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _weighted_correlation(x: np.ndarray, y: np.ndarray, weights: np.ndarray) -> float:
|
|
115
|
+
"""Frequency-weighted Pearson correlation of a usable pair."""
|
|
116
|
+
if not np.isfinite(x).all() or not np.isfinite(y).all():
|
|
117
|
+
return np.nan
|
|
118
|
+
# A weighted mean can round away from identical decimal inputs.
|
|
119
|
+
# Check the retained observations exactly before subtracting it.
|
|
120
|
+
if (x == x[0]).all() or (y == y[0]).all():
|
|
121
|
+
return np.nan
|
|
122
|
+
x, _ = _weighted_centered_vector(x, weights)
|
|
123
|
+
y, _ = _weighted_centered_vector(y, weights)
|
|
124
|
+
x_ss = np.sum(x * x)
|
|
125
|
+
y_ss = np.sum(y * y)
|
|
126
|
+
if x_ss == 0 or y_ss == 0:
|
|
127
|
+
return np.nan
|
|
128
|
+
result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
|
|
129
|
+
return float(np.clip(result, -1.0, 1.0))
|
|
130
|
+
|
|
131
|
+
|
|
54
132
|
def _weighted_top_share(
|
|
55
133
|
values: np.ndarray, weights: np.ndarray, top_x_pct: float
|
|
56
134
|
) -> float:
|
|
@@ -456,12 +534,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
456
534
|
"""Align usable paired observations with their left weights."""
|
|
457
535
|
if not isinstance(other, pd.Series):
|
|
458
536
|
raise TypeError("other must be a pandas Series or MicroSeries")
|
|
459
|
-
|
|
460
|
-
raise TypeError("ddof must be an integer")
|
|
461
|
-
if min_periods is None:
|
|
462
|
-
min_periods = 1
|
|
463
|
-
if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
|
|
464
|
-
raise ValueError("min_periods must be a nonnegative integer")
|
|
537
|
+
min_periods, ddof = _pair_options(min_periods, ddof)
|
|
465
538
|
if len(self) == 0 or len(other) == 0:
|
|
466
539
|
return None
|
|
467
540
|
|
|
@@ -477,27 +550,8 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
477
550
|
)
|
|
478
551
|
y = right.to_numpy(dtype=float, na_value=np.nan)
|
|
479
552
|
weights = np.asarray(self.weights, dtype=float)[positions]
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
# Zero frequency means the row is absent, including for skipna=False.
|
|
484
|
-
positive = weights > 0
|
|
485
|
-
x, y, weights = x[positive], y[positive], weights[positive]
|
|
486
|
-
missing = np.isnan(x) | np.isnan(y)
|
|
487
|
-
if not skipna and missing.any():
|
|
488
|
-
return None
|
|
489
|
-
x, y, weights = x[~missing], y[~missing], weights[~missing]
|
|
490
|
-
total_weight = weights.sum()
|
|
491
|
-
if not np.isfinite(total_weight):
|
|
492
|
-
raise ValueError("the sum of frequency weights must be finite")
|
|
493
|
-
if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
|
|
494
|
-
return None
|
|
495
|
-
return (
|
|
496
|
-
x,
|
|
497
|
-
y,
|
|
498
|
-
weights,
|
|
499
|
-
float(total_weight - ddof),
|
|
500
|
-
)
|
|
553
|
+
_validate_frequency_weights(weights)
|
|
554
|
+
return _usable_pair(x, y, weights, min_periods, ddof, skipna)
|
|
501
555
|
|
|
502
556
|
def cov(
|
|
503
557
|
self,
|
|
@@ -531,19 +585,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
531
585
|
if pair is None:
|
|
532
586
|
return np.nan
|
|
533
587
|
x, y, weights, denominator = pair
|
|
534
|
-
|
|
535
|
-
return np.nan
|
|
536
|
-
x, x_exponent = _weighted_centered_vector(x, weights)
|
|
537
|
-
y, y_exponent = _weighted_centered_vector(y, weights)
|
|
538
|
-
# Combine exponents only after dividing out sum(weights) - ddof.
|
|
539
|
-
# Neither the original squared scale nor raw weighted sum need fit.
|
|
540
|
-
denominator, denominator_exponent = np.frexp(denominator)
|
|
541
|
-
return float(
|
|
542
|
-
np.ldexp(
|
|
543
|
-
np.sum(x * y) / denominator,
|
|
544
|
-
x_exponent + y_exponent - int(denominator_exponent),
|
|
545
|
-
)
|
|
546
|
-
)
|
|
588
|
+
return _weighted_covariance(x, y, weights, denominator)
|
|
547
589
|
|
|
548
590
|
def corr(
|
|
549
591
|
self,
|
|
@@ -578,20 +620,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
|
|
|
578
620
|
if pair is None:
|
|
579
621
|
return np.nan
|
|
580
622
|
x, y, weights, _ = pair
|
|
581
|
-
|
|
582
|
-
return np.nan
|
|
583
|
-
# A weighted mean can round away from identical decimal inputs.
|
|
584
|
-
# Check the retained observations exactly before subtracting it.
|
|
585
|
-
if (x == x[0]).all() or (y == y[0]).all():
|
|
586
|
-
return np.nan
|
|
587
|
-
x, _ = _weighted_centered_vector(x, weights)
|
|
588
|
-
y, _ = _weighted_centered_vector(y, weights)
|
|
589
|
-
x_ss = np.sum(x * x)
|
|
590
|
-
y_ss = np.sum(y * y)
|
|
591
|
-
if x_ss == 0 or y_ss == 0:
|
|
592
|
-
return np.nan
|
|
593
|
-
result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
|
|
594
|
-
return float(np.clip(result, -1.0, 1.0))
|
|
623
|
+
return _weighted_correlation(x, y, weights)
|
|
595
624
|
|
|
596
625
|
def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
|
|
597
626
|
"""Calculates weighted quantiles of the MicroSeries.
|
|
@@ -355,31 +355,51 @@ def test_series_to_dataframe_weights_follow_aligned_rows_and_explicit_override(m
|
|
|
355
355
|
mdf.MicroDataFrame(data, index=[3, 99])
|
|
356
356
|
|
|
357
357
|
|
|
358
|
+
def replicated(data, weights, index=None):
|
|
359
|
+
"""Frequency-weight oracle: the plain frame with each row repeated."""
|
|
360
|
+
plain = pd.DataFrame(data, index=index)
|
|
361
|
+
return plain.iloc[np.repeat(np.arange(len(plain)), weights)]
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def pairwise(oracle, method, *args, **kwargs):
|
|
365
|
+
"""Apply a pandas matrix summary one complete pair at a time.
|
|
366
|
+
|
|
367
|
+
pandas honours ``ddof`` only through ``np.cov`` on complete data; once any
|
|
368
|
+
value is missing it falls back to a kernel that ignores it. Dropping the
|
|
369
|
+
missing rows per pair keeps the oracle on the ``np.cov`` path.
|
|
370
|
+
"""
|
|
371
|
+
columns = oracle.columns
|
|
372
|
+
out = pd.DataFrame(np.nan, index=columns, columns=columns, dtype=float)
|
|
373
|
+
for i, left in enumerate(columns):
|
|
374
|
+
for right in columns[i:]:
|
|
375
|
+
pair = oracle[[left, right]] if left != right else oracle[[left]]
|
|
376
|
+
cell = getattr(pair.dropna(), method)(*args, **kwargs).iloc[0, -1]
|
|
377
|
+
out.loc[left, right] = out.loc[right, left] = cell
|
|
378
|
+
return out
|
|
379
|
+
|
|
380
|
+
|
|
358
381
|
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
359
382
|
@pytest.mark.parametrize("coincident_labels", [False, True])
|
|
360
|
-
def
|
|
383
|
+
def test_dataframe_matrix_summaries_are_plain_and_frequency_weighted(
|
|
384
|
+
method, coincident_labels
|
|
385
|
+
):
|
|
361
386
|
if coincident_labels:
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
# Sample covariance divides centered cross-products by n - 1.
|
|
368
|
-
covariance = [[50.0, 20.0], [20.0, 8.0]]
|
|
387
|
+
data = {"x": [10.0, 20.0], "y": [4.0, 8.0]}
|
|
388
|
+
weights = [2, 3]
|
|
389
|
+
frame = mdf.MicroDataFrame(data, index=["x", "y"], weights=weights)
|
|
390
|
+
# Weighted means x = 16, y = 6.4; sum(w) - 1 = 4 in the denominator.
|
|
391
|
+
covariance = [[30.0, 12.0], [12.0, 4.8]]
|
|
369
392
|
correlation = [[1.0, 1.0], [1.0, 1.0]]
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
393
|
+
expected = pd.DataFrame(
|
|
394
|
+
covariance if method == "cov" else correlation,
|
|
395
|
+
index=frame.columns,
|
|
396
|
+
columns=frame.columns,
|
|
374
397
|
)
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
index=frame.columns,
|
|
381
|
-
columns=frame.columns,
|
|
382
|
-
)
|
|
398
|
+
else:
|
|
399
|
+
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
|
|
400
|
+
weights = [2, 3, 5]
|
|
401
|
+
frame = mdf.MicroDataFrame(data, weights=weights)
|
|
402
|
+
expected = getattr(replicated(data, weights), method)()
|
|
383
403
|
|
|
384
404
|
result = getattr(frame, method)()
|
|
385
405
|
|
|
@@ -395,17 +415,15 @@ def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_
|
|
|
395
415
|
[
|
|
396
416
|
("cov", (), {}),
|
|
397
417
|
("cov", (2, 0), {}),
|
|
398
|
-
("cov", (), {"min_periods":
|
|
418
|
+
("cov", (), {"min_periods": 2, "ddof": 2}),
|
|
399
419
|
("cov", (), {"min_periods": 2, "ddof": 0, "numeric_only": True}),
|
|
400
420
|
("corr", (), {}),
|
|
401
421
|
("corr", ("pearson", 2, True), {}),
|
|
402
|
-
("corr", (), {"
|
|
403
|
-
("corr", (), {"min_periods": 4}),
|
|
404
|
-
("corr", (), {"method": lambda x, y: np.dot(x, y), "min_periods": 2}),
|
|
422
|
+
("corr", (), {"min_periods": 2}),
|
|
405
423
|
],
|
|
406
424
|
)
|
|
407
425
|
@pytest.mark.parametrize("missing", [False, True])
|
|
408
|
-
def
|
|
426
|
+
def test_dataframe_matrix_summaries_follow_pandas_arguments(
|
|
409
427
|
method, args, kwargs, missing
|
|
410
428
|
):
|
|
411
429
|
data = {
|
|
@@ -413,8 +431,11 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments(
|
|
|
413
431
|
"y": [4.0, 8.0, np.nan if missing else 6.0, 9.0],
|
|
414
432
|
"flag": [True, False, True, True],
|
|
415
433
|
}
|
|
416
|
-
|
|
417
|
-
|
|
434
|
+
weights = [2, 3, 5, 7]
|
|
435
|
+
# Duplicate labels: weights follow row position, never labels.
|
|
436
|
+
frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=weights)
|
|
437
|
+
oracle = replicated(data, weights, index=frame.index)
|
|
438
|
+
expected = pairwise(oracle, method, *args, **kwargs)
|
|
418
439
|
|
|
419
440
|
result = getattr(frame, method)(*args, **kwargs)
|
|
420
441
|
|
|
@@ -423,12 +444,38 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments(
|
|
|
423
444
|
pd.testing.assert_series_equal(result.sum(), expected.sum())
|
|
424
445
|
|
|
425
446
|
|
|
447
|
+
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
448
|
+
def test_dataframe_matrix_summaries_min_periods_counts_usable_rows(method):
|
|
449
|
+
data = {"x": [10.0, 20.0, 30.0, 40.0], "y": [4.0, 8.0, np.nan, 9.0]}
|
|
450
|
+
frame = mdf.MicroDataFrame(data, weights=[20, 30, 50, 70])
|
|
451
|
+
# x and y share three usable rows, whatever their weight total.
|
|
452
|
+
assert np.isfinite(getattr(frame, method)(min_periods=3).loc["x", "y"])
|
|
453
|
+
result = getattr(frame, method)(min_periods=4)
|
|
454
|
+
assert np.isnan(result.loc["x", "y"]) and np.isnan(result.loc["y", "x"])
|
|
455
|
+
assert np.isfinite(result.loc["x", "x"])
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
@pytest.mark.parametrize(
|
|
459
|
+
"method", ["spearman", "kendall", lambda x, y: float(np.dot(x, y))]
|
|
460
|
+
)
|
|
461
|
+
def test_dataframe_correlation_rejects_other_methods(method):
|
|
462
|
+
frame = mdf.MicroDataFrame(
|
|
463
|
+
{"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}, weights=[2, 3, 5]
|
|
464
|
+
)
|
|
465
|
+
with pytest.raises(ValueError, match="pearson") as frame_error:
|
|
466
|
+
frame.corr(method=method)
|
|
467
|
+
with pytest.raises(ValueError) as series_error:
|
|
468
|
+
frame.x.corr(frame.y, method=method)
|
|
469
|
+
assert str(frame_error.value) == str(series_error.value)
|
|
470
|
+
|
|
471
|
+
|
|
426
472
|
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
427
473
|
def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method):
|
|
428
474
|
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0], "label": ["a", "b", "c"]}
|
|
429
|
-
|
|
475
|
+
weights = [2, 3, 5]
|
|
476
|
+
frame = mdf.MicroDataFrame(data, weights=weights)
|
|
430
477
|
plain = pd.DataFrame(data)
|
|
431
|
-
expected = getattr(
|
|
478
|
+
expected = getattr(replicated(data, weights), method)(numeric_only=True)
|
|
432
479
|
|
|
433
480
|
result = getattr(frame, method)(numeric_only=True)
|
|
434
481
|
|
|
@@ -442,37 +489,25 @@ def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method):
|
|
|
442
489
|
assert str(microdf_error.value) == str(pandas_error.value)
|
|
443
490
|
|
|
444
491
|
|
|
445
|
-
def test_dataframe_correlation_preserves_optional_kendall_support():
|
|
446
|
-
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
|
|
447
|
-
frame = mdf.MicroDataFrame(data, weights=[2, 3, 5])
|
|
448
|
-
try:
|
|
449
|
-
expected = pd.DataFrame(data).corr(method="kendall")
|
|
450
|
-
except ImportError as pandas_error:
|
|
451
|
-
# Kendall requires scipy; delegation preserves pandas' dependency error.
|
|
452
|
-
with pytest.raises(type(pandas_error)) as microdf_error:
|
|
453
|
-
frame.corr(method="kendall")
|
|
454
|
-
assert str(microdf_error.value) == str(pandas_error)
|
|
455
|
-
else:
|
|
456
|
-
result = frame.corr(method="kendall")
|
|
457
|
-
assert type(result) is pd.DataFrame
|
|
458
|
-
pd.testing.assert_frame_equal(result, expected)
|
|
459
|
-
|
|
460
|
-
|
|
461
492
|
@pytest.mark.parametrize("method", ["cov", "corr"])
|
|
462
493
|
def test_dataframe_matrix_summaries_preserve_pandas_metadata(method):
|
|
463
494
|
data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
expected = getattr(plain, method)()
|
|
495
|
+
weights = [2, 3, 5]
|
|
496
|
+
frame = mdf.MicroDataFrame(data, weights=weights)
|
|
497
|
+
frame.attrs = {"survey": {"year": 2026}}
|
|
498
|
+
frame.flags.allows_duplicate_labels = False
|
|
499
|
+
frame.columns.name = "measure"
|
|
500
|
+
expected = getattr(replicated(data, weights), method)()
|
|
471
501
|
|
|
472
502
|
result = getattr(frame, method)()
|
|
473
503
|
|
|
474
504
|
assert type(result) is pd.DataFrame
|
|
475
|
-
pd.testing.assert_frame_equal(
|
|
476
|
-
|
|
505
|
+
pd.testing.assert_frame_equal(
|
|
506
|
+
result, expected, check_flags=False, check_names=False
|
|
507
|
+
)
|
|
508
|
+
assert result.attrs == {"survey": {"year": 2026}}
|
|
509
|
+
assert result.flags.allows_duplicate_labels is False
|
|
510
|
+
assert result.index.name == "measure" and result.columns.name == "measure"
|
|
511
|
+
# attrs are copied, not shared, with the source frame.
|
|
477
512
|
result.attrs["survey"]["year"] = 2025
|
|
478
513
|
assert frame.attrs["survey"]["year"] == 2026
|
|
@@ -341,3 +341,108 @@ def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof(
|
|
|
341
341
|
else:
|
|
342
342
|
expected = exact_weighted_moments(x, y, weights, ddof=ddof)
|
|
343
343
|
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def replicated_frame(data, weights):
|
|
347
|
+
"""Frequency-weight oracle for a frame: repeat each row by its weight."""
|
|
348
|
+
plain = pd.DataFrame(data)
|
|
349
|
+
return plain.iloc[np.repeat(np.arange(len(plain)), weights)]
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
353
|
+
def test_dataframe_cov_corr_match_replicated_frequency_sample(ddof):
|
|
354
|
+
data = {"x": [1.0, 4.0, 8.0, 2.0], "y": [5.0, 2.0, 9.0, 7.0], "z": [3, 3, 1, 6]}
|
|
355
|
+
weights = [1, 3, 2, 4]
|
|
356
|
+
frame = mdf.MicroDataFrame(data, weights=weights)
|
|
357
|
+
oracle = replicated_frame(data, weights)
|
|
358
|
+
pd.testing.assert_frame_equal(frame.cov(ddof=ddof), oracle.cov(ddof=ddof))
|
|
359
|
+
pd.testing.assert_frame_equal(frame.corr(), oracle.corr())
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def test_dataframe_cov_corr_cells_equal_microseries_results():
|
|
363
|
+
frame = mdf.MicroDataFrame(
|
|
364
|
+
{
|
|
365
|
+
"a": [1.0, 4.0, 8.0, 3.0, 6.0],
|
|
366
|
+
"b": [5.0, 2.0, 9.0, np.nan, 1.0],
|
|
367
|
+
"c": [2.0, 2.0, 7.0, 1.0, 1e9],
|
|
368
|
+
},
|
|
369
|
+
weights=[1, 3, 2, 0, 0.5],
|
|
370
|
+
)
|
|
371
|
+
cov, corr = frame.cov(), frame.corr()
|
|
372
|
+
for left in frame.columns:
|
|
373
|
+
for right in frame.columns:
|
|
374
|
+
assert cov.loc[left, right] == frame[left].cov(frame[right])
|
|
375
|
+
assert corr.loc[left, right] == frame[left].corr(frame[right])
|
|
376
|
+
assert cov.loc[left, right] == cov.loc[right, left]
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def test_dataframe_cov_corr_use_pairwise_complete_rows():
|
|
380
|
+
frame = mdf.MicroDataFrame(
|
|
381
|
+
{
|
|
382
|
+
"x": [1.0, np.nan, 5.0, 9.0, 20.0],
|
|
383
|
+
"y": [2.0, 4.0, np.nan, 8.0, 30.0],
|
|
384
|
+
"z": [1.0, 1.0, 2.0, 3.0, 5.0],
|
|
385
|
+
},
|
|
386
|
+
weights=[1, 9, 2, 3, 7],
|
|
387
|
+
)
|
|
388
|
+
cov, corr = frame.cov(), frame.corr()
|
|
389
|
+
# x-y keeps rows 0, 3, 4; x-z rows 0, 2, 3, 4; y-z rows 0, 1, 3, 4.
|
|
390
|
+
cells = {
|
|
391
|
+
("x", "y"): replicated_moments([1, 9, 20], [2, 8, 30], [1, 3, 7]),
|
|
392
|
+
("x", "z"): replicated_moments([1, 5, 9, 20], [1, 2, 3, 5], [1, 2, 3, 7]),
|
|
393
|
+
("y", "z"): replicated_moments([2, 4, 8, 30], [1, 1, 3, 5], [1, 9, 3, 7]),
|
|
394
|
+
}
|
|
395
|
+
for (left, right), (expected_cov, expected_corr) in cells.items():
|
|
396
|
+
assert cov.loc[left, right] == pytest.approx(expected_cov)
|
|
397
|
+
assert corr.loc[left, right] == pytest.approx(expected_corr)
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def test_dataframe_cov_corr_omit_zero_weight_rows():
|
|
401
|
+
frame = mdf.MicroDataFrame(
|
|
402
|
+
{"x": [1.0, 4.0, 8.0, 1e6], "y": [5.0, 2.0, 9.0, -1e6]}, weights=[1, 3, 2, 0]
|
|
403
|
+
)
|
|
404
|
+
expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2])
|
|
405
|
+
assert frame.cov().loc["x", "y"] == pytest.approx(expected_cov)
|
|
406
|
+
assert frame.corr().loc["x", "y"] == pytest.approx(expected_corr)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
@pytest.mark.parametrize("weights", [[1, -1, 2], [1, np.inf, 2], [1, np.nan, 2]])
|
|
410
|
+
def test_dataframe_cov_corr_reject_invalid_frequency_weights(weights):
|
|
411
|
+
frame = mdf.MicroDataFrame(
|
|
412
|
+
{"x": [1.0, 2.0, 3.0], "y": [3.0, 1.0, 2.0]}, weights=weights
|
|
413
|
+
)
|
|
414
|
+
with pytest.raises(ValueError, match="finite and nonnegative"):
|
|
415
|
+
frame.cov()
|
|
416
|
+
with pytest.raises(ValueError, match="finite and nonnegative"):
|
|
417
|
+
frame.corr()
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def test_dataframe_cov_corr_diagonal_and_constant_columns():
|
|
421
|
+
frame = mdf.MicroDataFrame(
|
|
422
|
+
{"x": [1.0, 4.0, 8.0], "c": [2.0, 2.0, 2.0]}, weights=[1, 3, 2]
|
|
423
|
+
)
|
|
424
|
+
cov, corr = frame.cov(), frame.corr()
|
|
425
|
+
assert cov.loc["x", "x"] == pytest.approx(frame.x.var())
|
|
426
|
+
assert cov.loc["c", "c"] == 0 and cov.loc["x", "c"] == 0
|
|
427
|
+
assert corr.loc["x", "x"] == 1.0
|
|
428
|
+
assert np.isnan(corr.loc["c", "c"]) and np.isnan(corr.loc["x", "c"])
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def test_dataframe_cov_corr_insufficient_weight_total_and_empty_frame():
|
|
432
|
+
frame = mdf.MicroDataFrame({"x": [1.0, 4.0], "y": [5.0, 2.0]}, weights=[1, 1])
|
|
433
|
+
assert np.isnan(frame.cov(ddof=2)).all().all()
|
|
434
|
+
assert np.isfinite(frame.cov(ddof=1)).all().all()
|
|
435
|
+
empty = mdf.MicroDataFrame({"x": [], "y": []}, weights=[])
|
|
436
|
+
for result in (empty.cov(), empty.corr()):
|
|
437
|
+
assert list(result.columns) == ["x", "y"]
|
|
438
|
+
assert np.isnan(result).all().all()
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def test_dataframe_cov_corr_validate_options_like_microseries():
|
|
442
|
+
frame = mdf.MicroDataFrame({"x": [1.0, 2.0], "y": [2.0, 1.0]}, weights=[1, 1])
|
|
443
|
+
with pytest.raises(TypeError, match="ddof"):
|
|
444
|
+
frame.cov(ddof=1.5)
|
|
445
|
+
with pytest.raises(ValueError, match="min_periods"):
|
|
446
|
+
frame.cov(min_periods=-1)
|
|
447
|
+
with pytest.raises(ValueError, match="min_periods"):
|
|
448
|
+
frame.corr(min_periods=-1)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
microdf/__init__.py,sha256=G4m4UDiGePngG1i0DdGYhsTAXqyDELR6Mbq72OXOMG0,790
|
|
2
2
|
microdf/_weights.py,sha256=uBOcTZGmVJlzi27sSAALSTb4V9j56tXuaBfJB83x7j0,10303
|
|
3
|
-
microdf/microdataframe.py,sha256=
|
|
4
|
-
microdf/microseries.py,sha256=
|
|
3
|
+
microdf/microdataframe.py,sha256=2HFXbS4gAmOQKX6eYcI5vwafCqv9aBzl3lDLODU9SXg,43436
|
|
4
|
+
microdf/microseries.py,sha256=ZjSiUg88zOWAslUl2QpJWK0BcglbG-l7bU6_preOqNs,50039
|
|
5
5
|
microdf/replication.py,sha256=3iZ6xG24ucKofVwmXxyd7ME6rZP6iJVnN9JpWcPxg9s,7697
|
|
6
6
|
microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
|
|
7
7
|
microdf/tests/test_aggregation_errors.py,sha256=9jJDiEyxMb2z1Zmj-o8AHDNp8LOputbEkAMefifnWaE,2013
|
|
@@ -16,10 +16,10 @@ microdf/tests/test_serialization.py,sha256=a7pHL2hNiG5iJjRtfx3C1BCmgOZouAekiOwUx
|
|
|
16
16
|
microdf/tests/test_sum_axes.py,sha256=N05ocwI5lLv2OgoaovRIqFIae-70356kZemRRet0ac8,8521
|
|
17
17
|
microdf/tests/test_ufunc_weight_dispatch.py,sha256=abQ5I4wFRM3poxO76nbR_1dQymHkaQKLlTaI5nVI70c,25664
|
|
18
18
|
microdf/tests/test_version_metadata.py,sha256=M1EabzHLKZZw3Djd6Zu2UuMQtDLV6rZ1zDrOU7W_jf0,227
|
|
19
|
-
microdf/tests/test_weight_propagation.py,sha256=
|
|
20
|
-
microdf/tests/test_weighted_cov_corr.py,sha256=
|
|
21
|
-
microdf_python-1.5.
|
|
22
|
-
microdf_python-1.5.
|
|
23
|
-
microdf_python-1.5.
|
|
24
|
-
microdf_python-1.5.
|
|
25
|
-
microdf_python-1.5.
|
|
19
|
+
microdf/tests/test_weight_propagation.py,sha256=YCHZRdeWyWEwnHdpoY1jRBgFLQcBE1nLs61LrcT1wao,20359
|
|
20
|
+
microdf/tests/test_weighted_cov_corr.py,sha256=Ag4YJDy0eH3r_DhOaiX44kEZ0GHuQOoQtBVAOAkBXmw,18291
|
|
21
|
+
microdf_python-1.5.8.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
|
|
22
|
+
microdf_python-1.5.8.dist-info/METADATA,sha256=cG5neCZ-WXOvFPEZNt1saWP2WPD4URKohDdmwn_1t-M,3535
|
|
23
|
+
microdf_python-1.5.8.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
24
|
+
microdf_python-1.5.8.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
|
|
25
|
+
microdf_python-1.5.8.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|