microdf-python 1.3.9__py3-none-any.whl → 1.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- microdf/microdataframe.py +48 -1
- microdf/microseries.py +201 -31
- microdf/tests/test_microseries_dataframe.py +0 -22
- microdf/tests/test_sum_axes.py +216 -0
- microdf/tests/test_weighted_cov_corr.py +343 -0
- {microdf_python-1.3.9.dist-info → microdf_python-1.4.0.dist-info}/METADATA +1 -1
- {microdf_python-1.3.9.dist-info → microdf_python-1.4.0.dist-info}/RECORD +10 -8
- {microdf_python-1.3.9.dist-info → microdf_python-1.4.0.dist-info}/WHEEL +0 -0
- {microdf_python-1.3.9.dist-info → microdf_python-1.4.0.dist-info}/licenses/LICENSE +0 -0
- {microdf_python-1.3.9.dist-info → microdf_python-1.4.0.dist-info}/top_level.txt +0 -0
microdf/microdataframe.py
CHANGED
|
@@ -161,13 +161,60 @@ class MicroDataFrame(pd.DataFrame):
|
|
|
161
161
|
def override_df_functions(self) -> None:
|
|
162
162
|
"""Override DataFrame functions to work with weighted operations."""
|
|
163
163
|
for name in MicroSeries.FUNCTIONS:
|
|
164
|
-
if name
|
|
164
|
+
if name == "sum":
|
|
165
|
+
# Sum has its own axis-aware signature and result types.
|
|
166
|
+
continue
|
|
167
|
+
elif name in MicroSeries.SCALAR_FUNCTIONS:
|
|
165
168
|
setattr(self, name, self._create_scalar_function(name))
|
|
166
169
|
elif name in MicroSeries.VECTOR_FUNCTIONS:
|
|
167
170
|
setattr(self, name, self._create_vector_function(name))
|
|
168
171
|
elif name in MicroSeries.AGNOSTIC_FUNCTIONS:
|
|
169
172
|
setattr(self, name, self._create_agnostic_function(name))
|
|
170
173
|
|
|
174
|
+
def sum(
|
|
175
|
+
self,
|
|
176
|
+
axis: Optional[Union[int, str]] = 0,
|
|
177
|
+
skipna: bool = True,
|
|
178
|
+
numeric_only: bool = False,
|
|
179
|
+
min_count: int = 0,
|
|
180
|
+
**kwargs,
|
|
181
|
+
) -> Union[pd.Series, MicroSeries, float]:
|
|
182
|
+
"""Sum numeric columns, weighting reductions across observations.
|
|
183
|
+
|
|
184
|
+
Column sums (axis=0 or 'index') apply observation weights and return a
|
|
185
|
+
plain Series. Row sums (axis=1 or 'columns') do not multiply row values
|
|
186
|
+
by weights; they return a MicroSeries with an independent copy of the
|
|
187
|
+
original weights for subsequent weighted aggregation.
|
|
188
|
+
|
|
189
|
+
Non-numeric columns are excluded, matching other MicroDataFrame
|
|
190
|
+
aggregations. skipna and min_count follow pandas sum semantics.
|
|
191
|
+
Explicit axis=None follows the installed pandas version: column sums in
|
|
192
|
+
pandas 2, and a weighted total over both axes in pandas 3.
|
|
193
|
+
"""
|
|
194
|
+
axis_number = None if axis is None else self._get_axis_number(axis)
|
|
195
|
+
values = pd.DataFrame(self)
|
|
196
|
+
numeric_columns = [
|
|
197
|
+
pd.api.types.is_numeric_dtype(dtype) for dtype in values.dtypes
|
|
198
|
+
]
|
|
199
|
+
values = values.iloc[:, numeric_columns]
|
|
200
|
+
if axis_number != 1 and self.weights is not None:
|
|
201
|
+
values = values.mul(self.weights, axis=0)
|
|
202
|
+
result = values.sum(
|
|
203
|
+
axis=axis,
|
|
204
|
+
skipna=skipna,
|
|
205
|
+
numeric_only=numeric_only,
|
|
206
|
+
min_count=min_count,
|
|
207
|
+
**kwargs,
|
|
208
|
+
)
|
|
209
|
+
if axis_number == 1:
|
|
210
|
+
weights = (
|
|
211
|
+
self.weights.copy()
|
|
212
|
+
if self.weights is not None
|
|
213
|
+
else pd.Series(1.0, index=self.index)
|
|
214
|
+
)
|
|
215
|
+
return MicroSeries(result, weights=weights)
|
|
216
|
+
return result
|
|
217
|
+
|
|
171
218
|
def _create_scalar_function(self, name: str) -> Callable:
|
|
172
219
|
"""Create a scalar function that returns a Series of results.
|
|
173
220
|
|
microdf/microseries.py
CHANGED
|
@@ -9,6 +9,41 @@ import pandas as pd
|
|
|
9
9
|
logger = logging.getLogger(__name__)
|
|
10
10
|
|
|
11
11
|
|
|
12
|
+
def _weighted_centered_vector(
|
|
13
|
+
values: np.ndarray, weights: np.ndarray
|
|
14
|
+
) -> tuple[np.ndarray, int]:
|
|
15
|
+
"""Return scaled sqrt-weighted deviations and their power-of-two
|
|
16
|
+
exponent."""
|
|
17
|
+
# Center relative to a maximum-weight observation: shifting by a low-weight
|
|
18
|
+
# extreme could erase differences among the influential observations.
|
|
19
|
+
# A relative mean also preserves nearby values at a large common offset.
|
|
20
|
+
reference = values[np.argmax(weights)]
|
|
21
|
+
with np.errstate(over="ignore"):
|
|
22
|
+
shifted = values - reference
|
|
23
|
+
exponent = 0
|
|
24
|
+
if np.isinf(shifted).any():
|
|
25
|
+
# Opposite finite extremes can overflow their difference. Halving is
|
|
26
|
+
# exact for those values; restore that factor in the final exponent.
|
|
27
|
+
shifted = values / 2 - reference / 2
|
|
28
|
+
exponent = 1
|
|
29
|
+
magnitude = np.max(np.abs(shifted))
|
|
30
|
+
if magnitude == 0:
|
|
31
|
+
return shifted, 0
|
|
32
|
+
_, shift = np.frexp(magnitude)
|
|
33
|
+
shifted = np.ldexp(shifted, -shift)
|
|
34
|
+
# Raise tiny mean weights by an exact common power of two so products
|
|
35
|
+
# with the scaled deviations do not underflow. Never scale down: that
|
|
36
|
+
# could discard small weights when frequencies span a wide range.
|
|
37
|
+
_, mean_weight_exponent = np.frexp(np.max(weights))
|
|
38
|
+
mean_weights = np.ldexp(weights, -min(int(mean_weight_exponent), 0))
|
|
39
|
+
shifted -= np.average(shifted, weights=mean_weights)
|
|
40
|
+
# Weight each vector before taking products, then scale again so squared
|
|
41
|
+
# deviations never accumulate raw frequencies at the original value scale.
|
|
42
|
+
shifted *= np.sqrt(weights)
|
|
43
|
+
_, weight_shift = np.frexp(np.max(np.abs(shifted)))
|
|
44
|
+
return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)
|
|
45
|
+
|
|
46
|
+
|
|
12
47
|
def _weighted_top_share(
|
|
13
48
|
values: np.ndarray, weights: np.ndarray, top_x_pct: float
|
|
14
49
|
) -> float:
|
|
@@ -187,16 +222,38 @@ class MicroSeries(pd.Series):
|
|
|
187
222
|
:returns: A Series multiplying the MicroSeries by its weight.
|
|
188
223
|
:rtype: pd.Series
|
|
189
224
|
"""
|
|
190
|
-
return self.multiply(self.weights)
|
|
225
|
+
return pd.Series(self, copy=False).multiply(self.weights)
|
|
191
226
|
|
|
192
227
|
@scalar_function
|
|
193
|
-
def sum(
|
|
228
|
+
def sum(
|
|
229
|
+
self,
|
|
230
|
+
axis: Optional[Union[int, str]] = 0,
|
|
231
|
+
skipna: bool = True,
|
|
232
|
+
numeric_only: bool = False,
|
|
233
|
+
min_count: int = 0,
|
|
234
|
+
**kwargs,
|
|
235
|
+
) -> float:
|
|
194
236
|
"""Calculates the weighted sum of the MicroSeries.
|
|
195
237
|
|
|
238
|
+
axis may be 0, 'index' or None, as for pandas Series.sum. skipna,
|
|
239
|
+
numeric_only and min_count are applied to the weighted values;
|
|
240
|
+
min_count counts valid observations, not the sum of their weights.
|
|
241
|
+
|
|
196
242
|
:returns: The weighted sum.
|
|
197
243
|
:rtype: float
|
|
198
244
|
"""
|
|
199
|
-
|
|
245
|
+
# Keep the intermediate unweighted so subclass constructors cannot
|
|
246
|
+
# apply observation weights a second time during the final reduction.
|
|
247
|
+
values = pd.Series(self)
|
|
248
|
+
if not self.empty:
|
|
249
|
+
values = values.multiply(self.weights)
|
|
250
|
+
return values.sum(
|
|
251
|
+
axis=axis,
|
|
252
|
+
skipna=skipna,
|
|
253
|
+
numeric_only=numeric_only,
|
|
254
|
+
min_count=min_count,
|
|
255
|
+
**kwargs,
|
|
256
|
+
)
|
|
200
257
|
|
|
201
258
|
@scalar_function
|
|
202
259
|
def count(self, skipna: bool = True) -> float:
|
|
@@ -293,39 +350,152 @@ class MicroSeries(pd.Series):
|
|
|
293
350
|
v = self._weighted_variance(ddof=ddof, skipna=skipna)
|
|
294
351
|
return float(np.sqrt(v)) if np.isfinite(v) else v
|
|
295
352
|
|
|
296
|
-
def
|
|
297
|
-
|
|
353
|
+
def _weighted_pair(
|
|
354
|
+
self,
|
|
355
|
+
other: pd.Series,
|
|
356
|
+
min_periods: Optional[int],
|
|
357
|
+
ddof: int,
|
|
358
|
+
skipna: bool,
|
|
359
|
+
) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
|
|
360
|
+
"""Align usable paired observations with their left weights."""
|
|
361
|
+
if not isinstance(other, pd.Series):
|
|
362
|
+
raise TypeError("other must be a pandas Series or MicroSeries")
|
|
363
|
+
if not isinstance(ddof, (int, np.integer)):
|
|
364
|
+
raise TypeError("ddof must be an integer")
|
|
365
|
+
if min_periods is None:
|
|
366
|
+
min_periods = 1
|
|
367
|
+
if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
|
|
368
|
+
raise ValueError("min_periods must be a nonnegative integer")
|
|
369
|
+
if len(self) == 0 or len(other) == 0:
|
|
370
|
+
return None
|
|
371
|
+
|
|
372
|
+
# Align left row positions so values and weights undergo exactly the
|
|
373
|
+
# same join, including pandas' expansion of duplicate index labels.
|
|
374
|
+
positions = pd.Series(np.arange(len(self)), index=self.index)
|
|
375
|
+
positions, right = positions.align(pd.Series(other), join="inner")
|
|
376
|
+
positions = positions.to_numpy(dtype=int)
|
|
377
|
+
x = (
|
|
378
|
+
pd.Series(self._values)
|
|
379
|
+
.iloc[positions]
|
|
380
|
+
.to_numpy(dtype=float, na_value=np.nan)
|
|
381
|
+
)
|
|
382
|
+
y = right.to_numpy(dtype=float, na_value=np.nan)
|
|
383
|
+
weights = np.asarray(self.weights, dtype=float)[positions]
|
|
384
|
+
if not np.isfinite(weights).all() or (weights < 0).any():
|
|
385
|
+
raise ValueError("frequency weights must be finite and nonnegative")
|
|
386
|
+
|
|
387
|
+
# Zero frequency means the row is absent, including for skipna=False.
|
|
388
|
+
positive = weights > 0
|
|
389
|
+
x, y, weights = x[positive], y[positive], weights[positive]
|
|
390
|
+
missing = np.isnan(x) | np.isnan(y)
|
|
391
|
+
if not skipna and missing.any():
|
|
392
|
+
return None
|
|
393
|
+
x, y, weights = x[~missing], y[~missing], weights[~missing]
|
|
394
|
+
total_weight = weights.sum()
|
|
395
|
+
if not np.isfinite(total_weight):
|
|
396
|
+
raise ValueError("the sum of frequency weights must be finite")
|
|
397
|
+
if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
|
|
398
|
+
return None
|
|
399
|
+
return (
|
|
400
|
+
x,
|
|
401
|
+
y,
|
|
402
|
+
weights,
|
|
403
|
+
float(total_weight - ddof),
|
|
404
|
+
)
|
|
298
405
|
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
406
|
+
def cov(
|
|
407
|
+
self,
|
|
408
|
+
other: pd.Series,
|
|
409
|
+
min_periods: Optional[int] = None,
|
|
410
|
+
ddof: int = 1,
|
|
411
|
+
*,
|
|
412
|
+
skipna: bool = True,
|
|
413
|
+
) -> float:
|
|
414
|
+
"""Calculate frequency-weighted covariance with another Series.
|
|
415
|
+
|
|
416
|
+
Observations align by index as in pandas, including its duplicate-
|
|
417
|
+
label join behavior. Only this Series' weights are used; weights on
|
|
418
|
+
another MicroSeries are ignored. Each aligned left weight must be
|
|
419
|
+
finite and nonnegative. Zero-weight rows are omitted.
|
|
420
|
+
|
|
421
|
+
Uses ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) - ddof)``.
|
|
422
|
+
Integer weights therefore match covariance on the replicated sample.
|
|
423
|
+
Missing values are removed pairwise before computing both means.
|
|
424
|
+
|
|
425
|
+
:param other: A pandas Series or MicroSeries to align by index.
|
|
426
|
+
:param min_periods: Minimum usable aligned row pairs, not the sum of
|
|
427
|
+
frequency weights. Defaults to 1.
|
|
428
|
+
:param ddof: Degrees of freedom subtracted from the weight total.
|
|
429
|
+
:param skipna: Drop pairs with a missing value. If False, any missing
|
|
430
|
+
value in a positive-weight aligned pair produces NaN.
|
|
431
|
+
:returns: Weighted covariance, or NaN for an empty or insufficient
|
|
432
|
+
sample (including a weight total no greater than ddof).
|
|
303
433
|
"""
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
434
|
+
pair = self._weighted_pair(other, min_periods, ddof, skipna)
|
|
435
|
+
if pair is None:
|
|
436
|
+
return np.nan
|
|
437
|
+
x, y, weights, denominator = pair
|
|
438
|
+
if not np.isfinite(x).all() or not np.isfinite(y).all():
|
|
439
|
+
return np.nan
|
|
440
|
+
x, x_exponent = _weighted_centered_vector(x, weights)
|
|
441
|
+
y, y_exponent = _weighted_centered_vector(y, weights)
|
|
442
|
+
# Combine exponents only after dividing out sum(weights) - ddof.
|
|
443
|
+
# Neither the original squared scale nor raw weighted sum need fit.
|
|
444
|
+
denominator, denominator_exponent = np.frexp(denominator)
|
|
445
|
+
return float(
|
|
446
|
+
np.ldexp(
|
|
447
|
+
np.sum(x * y) / denominator,
|
|
448
|
+
x_exponent + y_exponent - int(denominator_exponent),
|
|
449
|
+
)
|
|
311
450
|
)
|
|
312
|
-
return super().cov(other, *args, **kwargs)
|
|
313
451
|
|
|
314
|
-
def corr(
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
452
|
+
def corr(
|
|
453
|
+
self,
|
|
454
|
+
other: pd.Series,
|
|
455
|
+
method: str = "pearson",
|
|
456
|
+
min_periods: Optional[int] = None,
|
|
457
|
+
*,
|
|
458
|
+
ddof: int = 1,
|
|
459
|
+
skipna: bool = True,
|
|
460
|
+
) -> float:
|
|
461
|
+
"""Calculate frequency-weighted Pearson correlation.
|
|
462
|
+
|
|
463
|
+
Uses the same aligned pairs and left Series weights for covariance and
|
|
464
|
+
both variances. Weights on another MicroSeries are ignored. Weights
|
|
465
|
+
must be finite and nonnegative; zero-weight rows are omitted. Other
|
|
466
|
+
correlation methods, including callables, are unsupported.
|
|
467
|
+
|
|
468
|
+
:param other: A pandas Series or MicroSeries to align by index.
|
|
469
|
+
:param method: Only "pearson" is supported.
|
|
470
|
+
:param min_periods: Minimum usable aligned row pairs, not frequency
|
|
471
|
+
weight total. Defaults to 1.
|
|
472
|
+
:param ddof: Degrees of freedom for all three moments. It cancels from
|
|
473
|
+
the correlation but the weight total must exceed it.
|
|
474
|
+
:param skipna: Drop pairs with a missing value. If False, any missing
|
|
475
|
+
value in a positive-weight aligned pair produces NaN.
|
|
476
|
+
:returns: Weighted correlation, or NaN for an empty, insufficient, or
|
|
477
|
+
constant sample.
|
|
320
478
|
"""
|
|
321
|
-
|
|
322
|
-
"
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
)
|
|
328
|
-
|
|
479
|
+
if method != "pearson":
|
|
480
|
+
raise ValueError("weighted correlation only supports method='pearson'")
|
|
481
|
+
pair = self._weighted_pair(other, min_periods, ddof, skipna)
|
|
482
|
+
if pair is None:
|
|
483
|
+
return np.nan
|
|
484
|
+
x, y, weights, _ = pair
|
|
485
|
+
if not np.isfinite(x).all() or not np.isfinite(y).all():
|
|
486
|
+
return np.nan
|
|
487
|
+
# A weighted mean can round away from identical decimal inputs.
|
|
488
|
+
# Check the retained observations exactly before subtracting it.
|
|
489
|
+
if (x == x[0]).all() or (y == y[0]).all():
|
|
490
|
+
return np.nan
|
|
491
|
+
x, _ = _weighted_centered_vector(x, weights)
|
|
492
|
+
y, _ = _weighted_centered_vector(y, weights)
|
|
493
|
+
x_ss = np.sum(x * x)
|
|
494
|
+
y_ss = np.sum(y * y)
|
|
495
|
+
if x_ss == 0 or y_ss == 0:
|
|
496
|
+
return np.nan
|
|
497
|
+
result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
|
|
498
|
+
return float(np.clip(result, -1.0, 1.0))
|
|
329
499
|
|
|
330
500
|
def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
|
|
331
501
|
"""Calculates weighted quantiles of the MicroSeries.
|
|
@@ -744,28 +744,6 @@ def test_std_var_are_weighted() -> None:
|
|
|
744
744
|
)
|
|
745
745
|
|
|
746
746
|
|
|
747
|
-
def test_cov_corr_warn_when_fallthrough() -> None:
|
|
748
|
-
"""Regression: cov/corr silently returned unweighted pandas values.
|
|
749
|
-
|
|
750
|
-
They still fall through to pandas (a weighted impl is a separate issue) but
|
|
751
|
-
now emit a UserWarning so callers aren't misled.
|
|
752
|
-
"""
|
|
753
|
-
s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
|
|
754
|
-
s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1])
|
|
755
|
-
|
|
756
|
-
with warnings.catch_warnings(record=True) as w:
|
|
757
|
-
warnings.simplefilter("always")
|
|
758
|
-
_ = s1.cov(s2)
|
|
759
|
-
msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
|
|
760
|
-
assert any("unweighted" in m.lower() for m in msgs)
|
|
761
|
-
|
|
762
|
-
with warnings.catch_warnings(record=True) as w:
|
|
763
|
-
warnings.simplefilter("always")
|
|
764
|
-
_ = s1.corr(s2)
|
|
765
|
-
msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
|
|
766
|
-
assert any("unweighted" in m.lower() for m in msgs)
|
|
767
|
-
|
|
768
|
-
|
|
769
747
|
def test_count_skips_nan_by_default() -> None:
|
|
770
748
|
"""Regression: ``count()`` included NaN-row weight, contrary to pandas.
|
|
771
749
|
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
import warnings
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
import microdf as mdf
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def test_sum_axis_1() -> None:
|
|
11
|
+
# Test basic row-wise sum
|
|
12
|
+
df = mdf.MicroDataFrame(
|
|
13
|
+
{"A": [1, 2, 3], "B": [4, 5, 6], "C": [7, 8, 9]},
|
|
14
|
+
weights=[0.5, 1.0, 2.0],
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
# Row-wise sum (axis=1) should not use weights
|
|
18
|
+
row_sums = df.sum(axis=1)
|
|
19
|
+
expected = pd.Series([12, 15, 18], index=df.index) # 1+4+7, 2+5+8, 3+6+9
|
|
20
|
+
pd.testing.assert_series_equal(pd.Series(row_sums), expected)
|
|
21
|
+
|
|
22
|
+
# Column-wise sum (axis=0) should use weights
|
|
23
|
+
col_sums = df.sum(axis=0)
|
|
24
|
+
expected_weighted = pd.Series(
|
|
25
|
+
{
|
|
26
|
+
"A": 1 * 0.5 + 2 * 1.0 + 3 * 2.0, # 8.5
|
|
27
|
+
"B": 4 * 0.5 + 5 * 1.0 + 6 * 2.0, # 19.0
|
|
28
|
+
"C": 7 * 0.5 + 8 * 1.0 + 9 * 2.0, # 29.5
|
|
29
|
+
}
|
|
30
|
+
)
|
|
31
|
+
pd.testing.assert_series_equal(col_sums, expected_weighted)
|
|
32
|
+
|
|
33
|
+
# Test with mixed types (non-numeric columns should be ignored)
|
|
34
|
+
df_mixed = mdf.MicroDataFrame(
|
|
35
|
+
{"A": [1, 2, 3], "B": [4, 5, 6], "text": ["a", "b", "c"]},
|
|
36
|
+
weights=[1, 1, 1],
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
row_sums_mixed = df_mixed.sum(axis=1)
|
|
40
|
+
expected_mixed = pd.Series([5, 7, 9], index=df_mixed.index) # Only A+B
|
|
41
|
+
pd.testing.assert_series_equal(pd.Series(row_sums_mixed), expected_mixed)
|
|
42
|
+
|
|
43
|
+
# Test with axis='columns' (string form)
|
|
44
|
+
row_sums_str = df.sum(axis="columns")
|
|
45
|
+
pd.testing.assert_series_equal(pd.Series(row_sums_str), expected)
|
|
46
|
+
|
|
47
|
+
# Test with additional parameters
|
|
48
|
+
df_with_nan = mdf.MicroDataFrame(
|
|
49
|
+
{"A": [1, np.nan, 3], "B": [4, 5, 6], "C": [7, 8, np.nan]},
|
|
50
|
+
weights=[1, 1, 1],
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
# skipna=True (default)
|
|
54
|
+
row_sums_skipna = df_with_nan.sum(axis=1)
|
|
55
|
+
expected_skipna = pd.Series([12.0, 13.0, 9.0]) # NaN values skipped
|
|
56
|
+
pd.testing.assert_series_equal(pd.Series(row_sums_skipna), expected_skipna)
|
|
57
|
+
|
|
58
|
+
# skipna=False
|
|
59
|
+
row_sums_no_skipna = df_with_nan.sum(axis=1, skipna=False)
|
|
60
|
+
expected_no_skipna = pd.Series([12.0, np.nan, np.nan]) # NaN propagates
|
|
61
|
+
pd.testing.assert_series_equal(pd.Series(row_sums_no_skipna), expected_no_skipna)
|
|
62
|
+
|
|
63
|
+
# Test min_count parameter
|
|
64
|
+
row_sums_min_count = df_with_nan.sum(axis=1, min_count=3)
|
|
65
|
+
expected_min_count = pd.Series(
|
|
66
|
+
[12.0, np.nan, np.nan]
|
|
67
|
+
) # Row 1 and 2 have < 3 non-NA values
|
|
68
|
+
pd.testing.assert_series_equal(pd.Series(row_sums_min_count), expected_min_count)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@pytest.mark.parametrize("axis", [0, "index", 1, "columns"])
|
|
72
|
+
@pytest.mark.parametrize("positional", [False, True])
|
|
73
|
+
def test_sum_binds_positional_and_keyword_axes(axis, positional):
|
|
74
|
+
frame = mdf.MicroDataFrame(
|
|
75
|
+
{"a": [1, 2, 3], "b": [4, 5, 6]}, index=[7, 8, 9], weights=[1, 2, 3]
|
|
76
|
+
)
|
|
77
|
+
result = frame.sum(axis) if positional else frame.sum(axis=axis)
|
|
78
|
+
if axis in (0, "index"):
|
|
79
|
+
assert type(result) is pd.Series
|
|
80
|
+
pd.testing.assert_series_equal(result, pd.Series({"a": 14.0, "b": 32.0}))
|
|
81
|
+
else:
|
|
82
|
+
assert isinstance(result, mdf.MicroSeries)
|
|
83
|
+
pd.testing.assert_series_equal(
|
|
84
|
+
pd.Series(result), pd.Series([5, 7, 9], index=frame.index)
|
|
85
|
+
)
|
|
86
|
+
pd.testing.assert_series_equal(result.weights, frame.weights)
|
|
87
|
+
# Row values are not weighted yet; subsequent aggregation is weighted.
|
|
88
|
+
assert result.sum() == 5 * 1 + 7 * 2 + 9 * 3
|
|
89
|
+
result.weights.iloc[0] = 100
|
|
90
|
+
assert frame.weights.iloc[0] == 1
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@pytest.mark.parametrize("skipna,min_count", [(True, 0), (False, 0), (True, 3)])
|
|
94
|
+
@pytest.mark.parametrize("axis", [0, "index", None])
|
|
95
|
+
def test_weighted_column_sum_options(axis, skipna, min_count):
|
|
96
|
+
raw = pd.DataFrame({"a": [1.0, np.nan, 3.0], "b": [4.0, 5.0, 6.0]})
|
|
97
|
+
weights = pd.Series([1.0, 2.0, 3.0])
|
|
98
|
+
frame = mdf.MicroDataFrame(raw, weights=weights)
|
|
99
|
+
# Native sum defines version-specific axis=None and missing-value behavior.
|
|
100
|
+
# The independently weighted entries are [1, NaN, 9] and [4, 10, 18].
|
|
101
|
+
expected_data = pd.DataFrame({"a": [1.0, np.nan, 9.0], "b": [4.0, 10.0, 18.0]})
|
|
102
|
+
with warnings.catch_warnings():
|
|
103
|
+
warnings.simplefilter("ignore", FutureWarning)
|
|
104
|
+
expected = expected_data.sum(axis=axis, skipna=skipna, min_count=min_count)
|
|
105
|
+
actual = frame.sum(axis=axis, skipna=skipna, min_count=min_count)
|
|
106
|
+
if isinstance(expected, pd.Series):
|
|
107
|
+
assert type(actual) is pd.Series
|
|
108
|
+
pd.testing.assert_series_equal(actual, expected)
|
|
109
|
+
else:
|
|
110
|
+
np.testing.assert_allclose(actual, expected, equal_nan=True)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@pytest.mark.parametrize("axis", [None, 0, "index"])
|
|
114
|
+
@pytest.mark.parametrize("skipna,min_count", [(True, 0), (False, 0), (True, 3)])
|
|
115
|
+
def test_microseries_sum_options(axis, skipna, min_count):
|
|
116
|
+
series = mdf.MicroSeries([1.0, np.nan, 3.0], index=[7, 8, 9], weights=[1, 2, 3])
|
|
117
|
+
expected = pd.Series([1.0, np.nan, 9.0]).sum(
|
|
118
|
+
axis=axis, skipna=skipna, min_count=min_count
|
|
119
|
+
)
|
|
120
|
+
np.testing.assert_allclose(
|
|
121
|
+
series.sum(axis, skipna=skipna, min_count=min_count), expected, equal_nan=True
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@pytest.mark.parametrize("min_count,expected", [(0, 0.0), (1, np.nan)])
|
|
126
|
+
def test_empty_numeric_row_sum_identity(min_count, expected):
|
|
127
|
+
frame = mdf.MicroDataFrame({"text": ["a", "b"]}, index=[7, 8], weights=[2, 3])
|
|
128
|
+
actual = frame.sum(axis=1, min_count=min_count)
|
|
129
|
+
assert isinstance(actual, mdf.MicroSeries)
|
|
130
|
+
pd.testing.assert_series_equal(
|
|
131
|
+
pd.Series(actual), pd.Series([expected, expected], index=frame.index)
|
|
132
|
+
)
|
|
133
|
+
pd.testing.assert_series_equal(actual.weights, frame.weights)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def test_sum_rejects_invalid_arguments():
|
|
137
|
+
frame = mdf.MicroDataFrame({"a": [1, 2]}, weights=[1, 2])
|
|
138
|
+
with pytest.raises(TypeError):
|
|
139
|
+
frame.sum(1, axis=0)
|
|
140
|
+
with pytest.raises(TypeError):
|
|
141
|
+
frame.sum(bogus=True)
|
|
142
|
+
with pytest.raises(ValueError):
|
|
143
|
+
frame.sum(axis=2)
|
|
144
|
+
with pytest.raises(ValueError):
|
|
145
|
+
frame["a"].sum(axis=1)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def test_sum_handles_boolean_and_nullable_numeric_columns():
|
|
149
|
+
raw = pd.DataFrame(
|
|
150
|
+
{
|
|
151
|
+
"count": pd.Series([1, None, 3], dtype="Int64"),
|
|
152
|
+
"flag": pd.Series([True, False, True], dtype="boolean"),
|
|
153
|
+
"text": ["a", "b", "c"],
|
|
154
|
+
}
|
|
155
|
+
)
|
|
156
|
+
frame = mdf.MicroDataFrame(raw, weights=[1, 2, 3])
|
|
157
|
+
expected = raw[["count", "flag"]].sum(axis=1)
|
|
158
|
+
pd.testing.assert_series_equal(pd.Series(frame.sum(1)), expected)
|
|
159
|
+
totals = frame.sum(numeric_only=True)
|
|
160
|
+
assert list(totals.index) == ["count", "flag"]
|
|
161
|
+
assert totals["count"] == 10
|
|
162
|
+
assert totals["flag"] == 4
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_sum_preserves_other_scalar_positional_arguments():
|
|
166
|
+
frame = mdf.MicroDataFrame({"a": [-1.0, 2.0, 3.0]}, weights=[1, 2, 3])
|
|
167
|
+
for method, argument in [
|
|
168
|
+
("gini", "shift"),
|
|
169
|
+
("top_x_pct_share", 0.25),
|
|
170
|
+
("mean", False),
|
|
171
|
+
("var", 0),
|
|
172
|
+
]:
|
|
173
|
+
actual = getattr(frame, method)(argument)["a"]
|
|
174
|
+
expected = getattr(frame["a"], method)(argument)
|
|
175
|
+
assert actual == expected
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
@pytest.mark.parametrize("min_count,expected", [(0, 0.0), (1, np.nan)])
|
|
179
|
+
def test_sum_of_empty_inputs(min_count, expected):
|
|
180
|
+
frame = mdf.MicroDataFrame(pd.DataFrame({"a": pd.Series([], dtype=float)}))
|
|
181
|
+
row_sums = frame.sum(axis=1, min_count=min_count)
|
|
182
|
+
assert isinstance(row_sums, mdf.MicroSeries)
|
|
183
|
+
assert row_sums.empty
|
|
184
|
+
pd.testing.assert_series_equal(row_sums.weights, pd.Series([], dtype=float))
|
|
185
|
+
pd.testing.assert_series_equal(
|
|
186
|
+
frame.sum(min_count=min_count), pd.Series({"a": expected})
|
|
187
|
+
)
|
|
188
|
+
series = mdf.MicroSeries([], dtype=float)
|
|
189
|
+
np.testing.assert_allclose(
|
|
190
|
+
series.sum(min_count=min_count), expected, equal_nan=True
|
|
191
|
+
)
|
|
192
|
+
with pytest.raises(ValueError):
|
|
193
|
+
series.sum(axis=1)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
@pytest.mark.parametrize("axis", [0, 1])
|
|
197
|
+
@pytest.mark.parametrize("mixed_dtypes", [False, True])
|
|
198
|
+
def test_sum_preserves_duplicate_numeric_column_labels(axis, mixed_dtypes):
|
|
199
|
+
if mixed_dtypes:
|
|
200
|
+
raw = pd.DataFrame([[1.0, "x", 4.0], [2.0, "y", 5.0]], columns=["a", "a", "a"])
|
|
201
|
+
else:
|
|
202
|
+
raw = pd.DataFrame([[1.0, 4.0], [2.0, 5.0]], columns=["a", "a"])
|
|
203
|
+
frame = mdf.MicroDataFrame(raw, weights=[2, 3])
|
|
204
|
+
|
|
205
|
+
result = frame.sum(axis)
|
|
206
|
+
|
|
207
|
+
if axis == 0:
|
|
208
|
+
assert type(result) is pd.Series
|
|
209
|
+
expected = pd.Series([8.0, 23.0], index=["a", "a"])
|
|
210
|
+
pd.testing.assert_series_equal(result, expected)
|
|
211
|
+
else:
|
|
212
|
+
assert isinstance(result, mdf.MicroSeries)
|
|
213
|
+
pd.testing.assert_series_equal(pd.Series(result), pd.Series([5.0, 7.0]))
|
|
214
|
+
pd.testing.assert_series_equal(result.weights, frame.weights)
|
|
215
|
+
assert result.sum() == 31.0
|
|
216
|
+
pd.testing.assert_frame_equal(pd.DataFrame(frame), raw)
|
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
import warnings
|
|
2
|
+
from decimal import Decimal, localcontext
|
|
3
|
+
from fractions import Fraction
|
|
4
|
+
from itertools import permutations
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
import microdf as mdf
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def replicated_moments(x, y, weights, ddof=1):
|
|
14
|
+
"""Independent frequency-weight oracle: expand to an ordinary sample."""
|
|
15
|
+
repeated_x = np.repeat(np.asarray(x, dtype=float), weights)
|
|
16
|
+
repeated_y = np.repeat(np.asarray(y, dtype=float), weights)
|
|
17
|
+
return (
|
|
18
|
+
np.cov(repeated_x, repeated_y, ddof=ddof)[0, 1],
|
|
19
|
+
np.corrcoef(repeated_x, repeated_y)[0, 1],
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
24
|
+
def test_cov_corr_match_replicated_frequency_sample(ddof):
|
|
25
|
+
x, y, weights = [1, 4, 8], [5, 2, 9], [1, 3, 2]
|
|
26
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
27
|
+
right = pd.Series(y)
|
|
28
|
+
expected_cov, expected_corr = replicated_moments(x, y, weights, ddof)
|
|
29
|
+
with warnings.catch_warnings(record=True) as caught:
|
|
30
|
+
warnings.simplefilter("always")
|
|
31
|
+
assert left.cov(right, ddof=ddof) == pytest.approx(expected_cov)
|
|
32
|
+
assert left.corr(right, ddof=ddof) == pytest.approx(expected_corr)
|
|
33
|
+
assert not any("unweighted" in str(item.message).lower() for item in caught)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_cov_corr_align_indices_and_use_only_left_weights():
|
|
37
|
+
left = mdf.MicroSeries([1, 4, 8], index=["a", "b", "c"], weights=[1, 3, 2])
|
|
38
|
+
right = mdf.MicroSeries(
|
|
39
|
+
[9, 5, 2, 100], index=["c", "a", "b", "d"], weights=[99, 1, 1, 9]
|
|
40
|
+
)
|
|
41
|
+
expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2])
|
|
42
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
43
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
44
|
+
assert right.weights.tolist() == [99, 1, 1, 9]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def test_cov_corr_use_same_pairwise_nonmissing_sample():
|
|
48
|
+
left = mdf.MicroSeries(
|
|
49
|
+
[1, np.nan, 5, 9, 20], index=list("abcde"), weights=[1, 9, 2, 3, 7]
|
|
50
|
+
)
|
|
51
|
+
right = pd.Series([2, 4, np.nan, 8, 30], index=list("abcdf"))
|
|
52
|
+
expected_cov, expected_corr = replicated_moments([1, 9], [2, 8], [1, 3])
|
|
53
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
54
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
55
|
+
assert np.isnan(left.cov(right, skipna=False))
|
|
56
|
+
assert np.isnan(left.corr(right, skipna=False))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@pytest.mark.parametrize("same_index", [True, False])
|
|
60
|
+
def test_cov_corr_follow_pandas_duplicate_index_alignment(same_index):
|
|
61
|
+
left = mdf.MicroSeries([1, 4, 7], index=["a", "a", "b"], weights=[1, 3, 2])
|
|
62
|
+
if same_index:
|
|
63
|
+
right = pd.Series([2, 3, 8], index=["a", "a", "b"])
|
|
64
|
+
x, y, weights = [1, 4, 7], [2, 3, 8], [1, 3, 2]
|
|
65
|
+
else:
|
|
66
|
+
right = pd.Series([2, 5, 8], index=["a", "b", "b"])
|
|
67
|
+
# The shared a/b labels join, repeating the left row weight per pair.
|
|
68
|
+
x, y, weights = [1, 4, 7, 7], [2, 2, 5, 8], [1, 3, 2, 2]
|
|
69
|
+
expected_cov, expected_corr = replicated_moments(x, y, weights)
|
|
70
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
71
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_zero_weight_rows_do_not_enter_pairwise_sample():
|
|
75
|
+
left = mdf.MicroSeries([1, np.nan, 5, 1000], weights=[2, 0, 1, 0])
|
|
76
|
+
right = pd.Series([3, np.nan, 9, -1000])
|
|
77
|
+
expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [2, 1])
|
|
78
|
+
assert left.cov(right, skipna=False) == pytest.approx(expected_cov)
|
|
79
|
+
assert left.corr(right, skipna=False) == pytest.approx(expected_corr)
|
|
80
|
+
assert np.isnan(left.cov(right, min_periods=3))
|
|
81
|
+
assert np.isnan(left.corr(right, min_periods=3))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_min_periods_counts_usable_rows_separately_from_frequency_weight():
|
|
85
|
+
left = mdf.MicroSeries([1, 5], weights=[10, 20])
|
|
86
|
+
right = pd.Series([3, 9])
|
|
87
|
+
expected_cov, expected_corr = replicated_moments([1, 5], [3, 9], [10, 20], ddof=0)
|
|
88
|
+
# Existing pandas positional arguments retain their order.
|
|
89
|
+
assert left.cov(right, 2, 0) == pytest.approx(expected_cov)
|
|
90
|
+
assert left.corr(right, "pearson", 2, ddof=0) == pytest.approx(expected_corr)
|
|
91
|
+
assert np.isnan(left.cov(right, min_periods=3))
|
|
92
|
+
assert np.isnan(left.corr(right, min_periods=3))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@pytest.mark.parametrize(
|
|
96
|
+
"values,other,weights",
|
|
97
|
+
[([], [], []), ([np.nan], [1], [3]), ([1], [2], [0]), ([1], [2], [1])],
|
|
98
|
+
)
|
|
99
|
+
def test_cov_corr_return_nan_when_sample_is_insufficient(values, other, weights):
|
|
100
|
+
left = mdf.MicroSeries(values, weights=weights, dtype=float)
|
|
101
|
+
right = pd.Series(other, dtype=float)
|
|
102
|
+
assert np.isnan(left.cov(right))
|
|
103
|
+
assert np.isnan(left.corr(right))
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_cov_corr_no_index_overlap():
|
|
107
|
+
left = mdf.MicroSeries([1, 2], index=["a", "b"], weights=[1, 2])
|
|
108
|
+
right = pd.Series([3, 4], index=["c", "d"])
|
|
109
|
+
assert np.isnan(left.cov(right))
|
|
110
|
+
assert np.isnan(left.corr(right))
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_cov_corr_constant_and_frequency_singleton():
|
|
114
|
+
left = mdf.MicroSeries([4, 4], weights=[2, 3])
|
|
115
|
+
right = pd.Series([1, 5])
|
|
116
|
+
assert left.cov(right) == 0
|
|
117
|
+
assert np.isnan(left.corr(right))
|
|
118
|
+
singleton = mdf.MicroSeries([4], weights=[3])
|
|
119
|
+
assert singleton.cov(pd.Series([2])) == 0
|
|
120
|
+
assert np.isnan(singleton.corr(pd.Series([2])))
|
|
121
|
+
assert np.isnan(singleton.cov(pd.Series([2]), ddof=3))
|
|
122
|
+
assert np.isnan(singleton.corr(pd.Series([2]), ddof=3))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def test_cov_corr_nullable_numeric_data():
|
|
126
|
+
left = mdf.MicroSeries(pd.Series([1, pd.NA, 4], dtype="Int64"), weights=[2, 9, 3])
|
|
127
|
+
right = pd.Series([3, 8, 7], dtype="Float64")
|
|
128
|
+
expected_cov, expected_corr = replicated_moments([1, 4], [3, 7], [2, 3])
|
|
129
|
+
assert left.cov(right) == pytest.approx(expected_cov)
|
|
130
|
+
assert left.corr(right) == pytest.approx(expected_corr)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@pytest.mark.parametrize("method", ["spearman", "kendall", lambda x, y: 1.0])
|
|
134
|
+
def test_non_pearson_methods_are_explicitly_unsupported(method):
|
|
135
|
+
left = mdf.MicroSeries([1, 2, 3], weights=[1, 2, 3])
|
|
136
|
+
with pytest.raises(ValueError, match="pearson"):
|
|
137
|
+
left.corr(pd.Series([4, 2, 5]), method=method)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@pytest.mark.parametrize("weights", [[1, -1], [1, np.nan], [1, np.inf]])
|
|
141
|
+
def test_cov_corr_reject_invalid_frequency_weights(weights):
|
|
142
|
+
left = mdf.MicroSeries([1, 2], weights=weights)
|
|
143
|
+
for method in (left.cov, left.corr):
|
|
144
|
+
with pytest.raises(ValueError, match="weights"):
|
|
145
|
+
method(pd.Series([2, 4]))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def test_binary_statistics_not_in_dataframe_aggregation_factories():
|
|
149
|
+
assert "cov" not in mdf.MicroSeries.FUNCTIONS
|
|
150
|
+
assert "corr" not in mdf.MicroSeries.FUNCTIONS
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@pytest.mark.parametrize(
|
|
154
|
+
"x,y",
|
|
155
|
+
[([0.1, 0.1], [0, 1]), ([0, 1], [0.1, 0.1]), ([0.1, 0.1], [0.1, 0.1])],
|
|
156
|
+
)
|
|
157
|
+
@pytest.mark.parametrize("with_filtered_rows", [False, True])
|
|
158
|
+
def test_corr_exact_decimal_constants_return_nan(x, y, with_filtered_rows):
|
|
159
|
+
# Unequal weights can round the mean away from the identical 0.1 values.
|
|
160
|
+
# Constant detection must inspect usable observations before centering.
|
|
161
|
+
weights = [1, 2]
|
|
162
|
+
if with_filtered_rows:
|
|
163
|
+
x = x + [9, np.nan]
|
|
164
|
+
y = y + [7, 4]
|
|
165
|
+
weights = weights + [0, 3]
|
|
166
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
167
|
+
assert np.isnan(left.corr(pd.Series(y)))
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_corr_does_not_treat_nearby_distinct_values_as_constant():
|
|
171
|
+
x = [0.1, np.nextafter(0.1, np.inf)]
|
|
172
|
+
left = mdf.MicroSeries(x, weights=[1, 2])
|
|
173
|
+
assert np.isfinite(left.corr(pd.Series([0, 1])))
|
|
174
|
+
assert np.isfinite(mdf.MicroSeries([0, 1], weights=[1, 2]).corr(pd.Series(x)))
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def exact_weighted_moments(x, y, weights, ddof=1):
|
|
178
|
+
"""Compute moments of the actual input floats with exact rational
|
|
179
|
+
arithmetic."""
|
|
180
|
+
x, y, weights = [
|
|
181
|
+
[Fraction(float(value)) for value in values] for values in (x, y, weights)
|
|
182
|
+
]
|
|
183
|
+
total = sum(weights)
|
|
184
|
+
xmean = sum(w * value for w, value in zip(weights, x)) / total
|
|
185
|
+
ymean = sum(w * value for w, value in zip(weights, y)) / total
|
|
186
|
+
xy = sum(w * (a - xmean) * (b - ymean) for a, b, w in zip(x, y, weights))
|
|
187
|
+
xx = sum(w * (value - xmean) ** 2 for value, w in zip(x, weights))
|
|
188
|
+
yy = sum(w * (value - ymean) ** 2 for value, w in zip(y, weights))
|
|
189
|
+
with localcontext() as context:
|
|
190
|
+
context.prec = 100
|
|
191
|
+
product = xx * yy
|
|
192
|
+
correlation = (Decimal(xy.numerator) / Decimal(xy.denominator)) / (
|
|
193
|
+
Decimal(product.numerator) / Decimal(product.denominator)
|
|
194
|
+
).sqrt()
|
|
195
|
+
return float(xy / (total - ddof)), float(correlation)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
199
|
+
@pytest.mark.parametrize("swap", [False, True])
|
|
200
|
+
def test_cov_corr_preserve_small_differences_at_large_offsets(ddof, swap):
|
|
201
|
+
# Two distinct points are perfectly linear even two float steps apart.
|
|
202
|
+
x = np.array([1e12 - 2**-13, 1e12 + 2**-13])
|
|
203
|
+
y = np.array([0.0, 1.0])
|
|
204
|
+
if swap:
|
|
205
|
+
x, y = y, x
|
|
206
|
+
weights = [1, 2]
|
|
207
|
+
expected = exact_weighted_moments(x, y, weights, ddof)
|
|
208
|
+
assert expected[1] == 1.0
|
|
209
|
+
for shifted_x, shifted_y in [(x, y), (x - x[0], y - y[0])]:
|
|
210
|
+
left = mdf.MicroSeries(shifted_x, weights=weights)
|
|
211
|
+
right = pd.Series(shifted_y)
|
|
212
|
+
np.testing.assert_allclose(
|
|
213
|
+
[left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)],
|
|
214
|
+
expected,
|
|
215
|
+
rtol=2e-15,
|
|
216
|
+
atol=0,
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@pytest.mark.parametrize("frequency", [1, 1_000_000])
|
|
221
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
222
|
+
def test_cov_corr_large_finite_values_do_not_overflow_raw_frequencies(frequency, ddof):
|
|
223
|
+
x = np.array([1.0, 2.0, 3.0]) * 1e153
|
|
224
|
+
weights = [frequency] * 3
|
|
225
|
+
expected = exact_weighted_moments(x, x, weights, ddof)
|
|
226
|
+
assert np.isfinite(expected).all()
|
|
227
|
+
assert expected[1] == 1.0
|
|
228
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
229
|
+
with np.errstate(over="raise", invalid="raise"):
|
|
230
|
+
actual = [left.cov(pd.Series(x), ddof=ddof), left.corr(pd.Series(x), ddof=ddof)]
|
|
231
|
+
np.testing.assert_allclose(actual, expected, rtol=2e-15, atol=0)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
@pytest.mark.parametrize("frequency", [0.5, 1, 1_000_000])
|
|
235
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
236
|
+
@pytest.mark.parametrize("scales", [(1.0, 1.0), (1e153, -1e153), (1e200, 1e-200)])
|
|
237
|
+
def test_cov_corr_preserve_frequency_correction_across_value_scales(
|
|
238
|
+
frequency, ddof, scales
|
|
239
|
+
):
|
|
240
|
+
x = np.array([1.0, 4.0, 8.0]) * scales[0]
|
|
241
|
+
y = np.array([5.0, 2.0, 9.0]) * scales[1]
|
|
242
|
+
weights = np.array([1, 3, 2]) * frequency
|
|
243
|
+
expected = exact_weighted_moments(x, y, weights, ddof)
|
|
244
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
245
|
+
with np.errstate(over="raise", invalid="raise"):
|
|
246
|
+
actual = [left.cov(pd.Series(y), ddof=ddof), left.corr(pd.Series(y), ddof=ddof)]
|
|
247
|
+
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
@pytest.mark.parametrize("huge", [1e16, 1e20])
|
|
251
|
+
@pytest.mark.parametrize("order", list(permutations(range(3))))
|
|
252
|
+
def test_cov_corr_low_weight_extreme_does_not_make_result_depend_on_row_order(
|
|
253
|
+
huge, order
|
|
254
|
+
):
|
|
255
|
+
# The large observation contributes to covariance, but using it as the
|
|
256
|
+
# centering origin must not erase the difference between 1 and 2.
|
|
257
|
+
x = np.array([huge, 1.0, 2.0])
|
|
258
|
+
y = np.array([0.0, 1.0, 2.0])
|
|
259
|
+
weights = np.array([1 / huge, 1.0, 1.0])
|
|
260
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
261
|
+
order = list(order)
|
|
262
|
+
left = mdf.MicroSeries(x[order], weights=weights[order])
|
|
263
|
+
right = pd.Series(y[order])
|
|
264
|
+
np.testing.assert_allclose(
|
|
265
|
+
[left.cov(right, ddof=0), left.corr(right, ddof=0)],
|
|
266
|
+
expected,
|
|
267
|
+
rtol=3e-15,
|
|
268
|
+
atol=0,
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def test_cov_corr_smallest_common_positive_weight_cancels_from_population_moments():
|
|
273
|
+
# The common positive weight cancels: xy = 1, xx = yy = 2 for
|
|
274
|
+
# centered observations [-1, 0, 1] and [-1, 1, 0].
|
|
275
|
+
x, y = [1, 2, 3], [3, 5, 4]
|
|
276
|
+
weights = [np.nextafter(0.0, 1.0)] * 3
|
|
277
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
278
|
+
assert expected == (1 / 3, 0.5)
|
|
279
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
280
|
+
right = pd.Series(y)
|
|
281
|
+
np.testing.assert_allclose(
|
|
282
|
+
[left.cov(right, ddof=0), left.corr(right, ddof=0)],
|
|
283
|
+
expected,
|
|
284
|
+
rtol=3e-15,
|
|
285
|
+
atol=0,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
@pytest.mark.parametrize(
|
|
290
|
+
"frequency", [np.nextafter(0.0, 1.0), np.finfo(float).tiny, 1.0]
|
|
291
|
+
)
|
|
292
|
+
@pytest.mark.parametrize("multipliers", [[1, 2, 3], [1, 1, 2]])
|
|
293
|
+
@pytest.mark.parametrize("order", list(permutations(range(3))))
|
|
294
|
+
def test_cov_corr_unequal_tiny_and_normal_weights_match_exact_moments(
|
|
295
|
+
frequency, multipliers, order
|
|
296
|
+
):
|
|
297
|
+
x, y = np.array([1.0, 2.0, 3.0]), np.array([3.0, 5.0, 4.0])
|
|
298
|
+
weights = np.array(multipliers) * frequency
|
|
299
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
300
|
+
order = list(order)
|
|
301
|
+
left = mdf.MicroSeries(x[order], weights=weights[order])
|
|
302
|
+
right = pd.Series(y[order])
|
|
303
|
+
np.testing.assert_allclose(
|
|
304
|
+
[left.cov(right, ddof=0), left.corr(right, ddof=0)],
|
|
305
|
+
expected,
|
|
306
|
+
rtol=3e-15,
|
|
307
|
+
atol=0,
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
@pytest.mark.parametrize(
|
|
312
|
+
"x,y",
|
|
313
|
+
[
|
|
314
|
+
(np.array([1, 2, 3]) * 1e153, np.array([3, 5, 4]) * -1e153),
|
|
315
|
+
(np.array([1, 2, 3]) * 1e200, np.array([3, 5, 4]) * 1e-200),
|
|
316
|
+
([1e12 - 2**-13, 1e12, 1e12 + 2**-13], [3, 5, 4]),
|
|
317
|
+
],
|
|
318
|
+
)
|
|
319
|
+
def test_cov_corr_subnormal_weights_preserve_extreme_value_scales(x, y):
|
|
320
|
+
weights = np.array([1, 2, 3]) * np.nextafter(0.0, 1.0)
|
|
321
|
+
expected = exact_weighted_moments(x, y, weights, ddof=0)
|
|
322
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
323
|
+
right = pd.Series(y)
|
|
324
|
+
with np.errstate(over="raise", invalid="raise"):
|
|
325
|
+
actual = [left.cov(right, ddof=0), left.corr(right, ddof=0)]
|
|
326
|
+
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
@pytest.mark.parametrize("frequency", [np.nextafter(0.0, 1.0), 0.5])
|
|
330
|
+
@pytest.mark.parametrize("ddof", [0, 1, 2])
|
|
331
|
+
def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof(
|
|
332
|
+
frequency, ddof
|
|
333
|
+
):
|
|
334
|
+
x, y = [1, 2, 3], [3, 5, 4]
|
|
335
|
+
weights = np.array([1, 2, 3]) * frequency
|
|
336
|
+
left = mdf.MicroSeries(x, weights=weights)
|
|
337
|
+
right = pd.Series(y)
|
|
338
|
+
actual = [left.cov(right, ddof=ddof), left.corr(right, ddof=ddof)]
|
|
339
|
+
if weights.sum() <= ddof:
|
|
340
|
+
assert np.isnan(actual).all()
|
|
341
|
+
else:
|
|
342
|
+
expected = exact_weighted_moments(x, y, weights, ddof=ddof)
|
|
343
|
+
np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0)
|
|
@@ -1,17 +1,19 @@
|
|
|
1
1
|
microdf/__init__.py,sha256=sddmTcTZFSb1fjZkoLUR5TK7r5PEld4v2F92xBI5Sxs,641
|
|
2
|
-
microdf/microdataframe.py,sha256=
|
|
3
|
-
microdf/microseries.py,sha256=
|
|
2
|
+
microdf/microdataframe.py,sha256=yqLUC43WBDT8Z-AQ7OfjKUoizehcUGBHgt7wg_BVlFI,44416
|
|
3
|
+
microdf/microseries.py,sha256=Y2k80dUOy8XxDZ6xN-YiUX04NXajILaLEc2isMahBCA,45004
|
|
4
4
|
microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
|
|
5
5
|
microdf/tests/test_aggregation_errors.py,sha256=9jJDiEyxMb2z1Zmj-o8AHDNp8LOputbEkAMefifnWaE,2013
|
|
6
6
|
microdf/tests/test_dataframe_weight_storage.py,sha256=ngIsWa_QcBnnaLpAyIZhTgMxjv_7cK8Nbf7f4n29R4Q,1975
|
|
7
|
-
microdf/tests/test_microseries_dataframe.py,sha256=
|
|
7
|
+
microdf/tests/test_microseries_dataframe.py,sha256=vL0fg_NydU6a5myVVtXyHMr8eQOB_yyAkZYwLmoEnOA,30075
|
|
8
8
|
microdf/tests/test_nullify_weights_index.py,sha256=kZgzMaZEa_PXbsor2S4E-6VRid3C3rcC9ufk0qa7mgY,341
|
|
9
9
|
microdf/tests/test_pandas3_compatibility.py,sha256=A34Ni_WQ303sSNv-sqv5CGAQp54zj-ZSGAPEBHZslNI,8573
|
|
10
10
|
microdf/tests/test_quantile_missing_values.py,sha256=lfntDvV2q7KH_CPVrXFJSQFoaGlc_OkRGhKwpxtDQtY,5327
|
|
11
11
|
microdf/tests/test_serialization.py,sha256=a7pHL2hNiG5iJjRtfx3C1BCmgOZouAekiOwUxouAPfo,5083
|
|
12
|
+
microdf/tests/test_sum_axes.py,sha256=N05ocwI5lLv2OgoaovRIqFIae-70356kZemRRet0ac8,8521
|
|
12
13
|
microdf/tests/test_version_metadata.py,sha256=M1EabzHLKZZw3Djd6Zu2UuMQtDLV6rZ1zDrOU7W_jf0,227
|
|
13
|
-
|
|
14
|
-
microdf_python-1.
|
|
15
|
-
microdf_python-1.
|
|
16
|
-
microdf_python-1.
|
|
17
|
-
microdf_python-1.
|
|
14
|
+
microdf/tests/test_weighted_cov_corr.py,sha256=LTnFhMWnb28f7L_9kOhPiV5UlLMAUbC9OlLIaZ1XgLs,13954
|
|
15
|
+
microdf_python-1.4.0.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
|
|
16
|
+
microdf_python-1.4.0.dist-info/METADATA,sha256=NjHgMRpJzMIc2G3IOjOOaY3YaLc-gBMStki-4LuY9d4,2305
|
|
17
|
+
microdf_python-1.4.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
18
|
+
microdf_python-1.4.0.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
|
|
19
|
+
microdf_python-1.4.0.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|