microdf-python 1.2.2__py3-none-any.whl → 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- microdf/microdataframe.py +103 -55
- microdf/microseries.py +257 -83
- microdf/tests/test_microseries_dataframe.py +373 -21
- microdf/tests/test_pandas3_compatibility.py +27 -33
- {microdf_python-1.2.2.dist-info → microdf_python-1.3.0.dist-info}/METADATA +2 -6
- microdf_python-1.3.0.dist-info/RECORD +11 -0
- {microdf_python-1.2.2.dist-info → microdf_python-1.3.0.dist-info}/WHEEL +1 -1
- microdf_python-1.2.2.dist-info/RECORD +0 -11
- {microdf_python-1.2.2.dist-info → microdf_python-1.3.0.dist-info}/licenses/LICENSE +0 -0
- {microdf_python-1.2.2.dist-info → microdf_python-1.3.0.dist-info}/top_level.txt +0 -0
|
@@ -2,6 +2,7 @@ import warnings
|
|
|
2
2
|
|
|
3
3
|
import numpy as np
|
|
4
4
|
import pandas as pd
|
|
5
|
+
import pytest
|
|
5
6
|
|
|
6
7
|
import microdf as mdf
|
|
7
8
|
from microdf.microdataframe import MicroDataFrame
|
|
@@ -261,9 +262,7 @@ def test_decile_rank() -> None:
|
|
|
261
262
|
|
|
262
263
|
|
|
263
264
|
def test_copy_equals() -> None:
|
|
264
|
-
d = mdf.MicroDataFrame(
|
|
265
|
-
{"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]
|
|
266
|
-
)
|
|
265
|
+
d = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8])
|
|
267
266
|
d_copy = d.copy()
|
|
268
267
|
d_copy_diff_weights = d_copy.copy()
|
|
269
268
|
d_copy_diff_weights.weights *= 2
|
|
@@ -275,9 +274,7 @@ def test_copy_equals() -> None:
|
|
|
275
274
|
|
|
276
275
|
|
|
277
276
|
def test_subset() -> None:
|
|
278
|
-
df = mdf.MicroDataFrame(
|
|
279
|
-
{"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]
|
|
280
|
-
)
|
|
277
|
+
df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8])
|
|
281
278
|
df_no_z = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4]}, weights=[7, 8])
|
|
282
279
|
assert df[["x", "y"]].equals(df_no_z)
|
|
283
280
|
df_no_z_diff_weights = df_no_z.copy()
|
|
@@ -353,9 +350,7 @@ def test_reset_index_inplace() -> None:
|
|
|
353
350
|
# Test 4: Multi-level index
|
|
354
351
|
arrays = [["bar", "bar", "baz", "baz"], ["one", "two", "one", "two"]]
|
|
355
352
|
multi_index = pd.MultiIndex.from_arrays(arrays, names=["first", "second"])
|
|
356
|
-
df_multi = pd.DataFrame(
|
|
357
|
-
{"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=multi_index
|
|
358
|
-
)
|
|
353
|
+
df_multi = pd.DataFrame({"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=multi_index)
|
|
359
354
|
mdf_multi = MicroDataFrame(df_multi, weights=weights)
|
|
360
355
|
result = mdf_multi.reset_index(level="first")
|
|
361
356
|
assert "first" in result.columns
|
|
@@ -373,9 +368,7 @@ def test_reset_index_inplace() -> None:
|
|
|
373
368
|
def test_loc_preserves_weights() -> None:
|
|
374
369
|
"""Test that .loc[] returns MicroDataFrame with proper weights (issue
|
|
375
370
|
#265)."""
|
|
376
|
-
df = mdf.MicroDataFrame(
|
|
377
|
-
{"one": [1, 1, 1, 1, 1]}, weights=[10, 20, 30, 40, 50]
|
|
378
|
-
)
|
|
371
|
+
df = mdf.MicroDataFrame({"one": [1, 1, 1, 1, 1]}, weights=[10, 20, 30, 40, 50])
|
|
379
372
|
|
|
380
373
|
# Filter all rows (should get same weights)
|
|
381
374
|
filtered = df.loc[df.one == 1]
|
|
@@ -383,9 +376,7 @@ def test_loc_preserves_weights() -> None:
|
|
|
383
376
|
assert filtered.one.sum() == 150.0 # Weighted sum
|
|
384
377
|
|
|
385
378
|
# Partial filter
|
|
386
|
-
df2 = mdf.MicroDataFrame(
|
|
387
|
-
{"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]
|
|
388
|
-
)
|
|
379
|
+
df2 = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50])
|
|
389
380
|
subset = df2.loc[df2.x > 2]
|
|
390
381
|
assert isinstance(subset, MicroDataFrame)
|
|
391
382
|
assert subset.x.sum() == 500.0 # 3*30 + 4*40 + 5*50 = 500
|
|
@@ -394,9 +385,7 @@ def test_loc_preserves_weights() -> None:
|
|
|
394
385
|
|
|
395
386
|
def test_iloc_preserves_weights() -> None:
|
|
396
387
|
"""Test that .iloc[] returns MicroDataFrame with proper weights."""
|
|
397
|
-
df = mdf.MicroDataFrame(
|
|
398
|
-
{"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]
|
|
399
|
-
)
|
|
388
|
+
df = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50])
|
|
400
389
|
|
|
401
390
|
# Select rows by position
|
|
402
391
|
subset = df.iloc[2:5]
|
|
@@ -407,9 +396,7 @@ def test_iloc_preserves_weights() -> None:
|
|
|
407
396
|
|
|
408
397
|
def test_groupby_column_selection() -> None:
|
|
409
398
|
"""Test that groupby column selection preserves weights (issue #193)."""
|
|
410
|
-
d = mdf.MicroDataFrame(
|
|
411
|
-
dict(g=["a", "a", "b"], y=[1, 2, 3]), weights=[4, 5, 6]
|
|
412
|
-
)
|
|
399
|
+
d = mdf.MicroDataFrame(dict(g=["a", "a", "b"], y=[1, 2, 3]), weights=[4, 5, 6])
|
|
413
400
|
|
|
414
401
|
# Test single column string selection
|
|
415
402
|
result_str = d.groupby("g")["y"].sum()
|
|
@@ -459,6 +446,43 @@ def test_mean_no_warning() -> None:
|
|
|
459
446
|
assert len(user_warnings) == 0
|
|
460
447
|
|
|
461
448
|
|
|
449
|
+
def test_sum_with_non_default_index() -> None:
|
|
450
|
+
"""Weighted sum must not silently return 0 with a non-default index.
|
|
451
|
+
|
|
452
|
+
Regression test for the bug where ``set_weights`` stored the weights
|
|
453
|
+
Series with a default ``RangeIndex`` regardless of ``self.index``.
|
|
454
|
+
Element-wise ops like ``self.multiply(self.weights)`` then aligned on
|
|
455
|
+
label, producing all-NaN and a silent ``0.0`` from ``.sum()`` while
|
|
456
|
+
``.mean()`` (which uses a positional ndarray) stayed correct.
|
|
457
|
+
"""
|
|
458
|
+
# MicroSeries with custom integer index.
|
|
459
|
+
s = mdf.MicroSeries([1, 2, 3], index=[100, 200, 300], weights=[10, 20, 30])
|
|
460
|
+
assert s.sum() == 140.0
|
|
461
|
+
assert s.weights.index.tolist() == [100, 200, 300]
|
|
462
|
+
|
|
463
|
+
# MicroDataFrame with custom integer index.
|
|
464
|
+
df = mdf.MicroDataFrame(
|
|
465
|
+
{"x": [1, 2, 3]}, index=[100, 200, 300], weights=[10, 20, 30]
|
|
466
|
+
)
|
|
467
|
+
assert df.x.sum() == 140.0
|
|
468
|
+
assert df.weights.index.tolist() == [100, 200, 300]
|
|
469
|
+
|
|
470
|
+
# MicroDataFrame with string index + set_weights via column name.
|
|
471
|
+
df2 = mdf.MicroDataFrame(
|
|
472
|
+
{"x": [1, 2, 3], "w": [10, 20, 30]},
|
|
473
|
+
index=["a", "b", "c"],
|
|
474
|
+
)
|
|
475
|
+
df2.set_weights("w")
|
|
476
|
+
assert df2.x.sum() == 140.0
|
|
477
|
+
assert df2.weights.index.tolist() == ["a", "b", "c"]
|
|
478
|
+
|
|
479
|
+
# Passing a Series with its own index should position-align, not
|
|
480
|
+
# label-align, so sum does not depend on accidental index alignment.
|
|
481
|
+
df3 = mdf.MicroDataFrame({"x": [1, 2, 3]}, index=[100, 200, 300])
|
|
482
|
+
df3.set_weights(pd.Series([10, 20, 30], index=[0, 1, 2]))
|
|
483
|
+
assert df3.x.sum() == 140.0
|
|
484
|
+
|
|
485
|
+
|
|
462
486
|
def test_repr_no_warning() -> None:
|
|
463
487
|
"""Internal .values usage in __repr__ should NOT emit a warning."""
|
|
464
488
|
ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6])
|
|
@@ -467,3 +491,331 @@ def test_repr_no_warning() -> None:
|
|
|
467
491
|
_ = repr(ms)
|
|
468
492
|
user_warnings = [x for x in w if issubclass(x.category, UserWarning)]
|
|
469
493
|
assert len(user_warnings) == 0
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def test_drop_inplace_aligns_weights() -> None:
|
|
497
|
+
"""Regression: ``drop(inplace=True)`` must keep weights in sync.
|
|
498
|
+
|
|
499
|
+
Previously, ``weights_backup = self.weights.copy()`` was taken *before*
|
|
500
|
+
the drop, then reassigned back afterwards — so the weights vector
|
|
501
|
+
kept its original length and any subsequent weighted op either
|
|
502
|
+
raised a length-mismatch ValueError or silently returned 0.
|
|
503
|
+
"""
|
|
504
|
+
# Row drop inplace.
|
|
505
|
+
df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40])
|
|
506
|
+
df.drop(index=[0, 1], inplace=True)
|
|
507
|
+
assert len(df) == len(df.weights) == 2
|
|
508
|
+
assert df.x.sum() == 3 * 30 + 4 * 40 # 250
|
|
509
|
+
|
|
510
|
+
# Row drop non-inplace.
|
|
511
|
+
df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40])
|
|
512
|
+
df2 = df.drop(index=[0, 1])
|
|
513
|
+
assert len(df2) == len(df2.weights) == 2
|
|
514
|
+
assert df2.x.sum() == 250
|
|
515
|
+
# Original untouched.
|
|
516
|
+
assert len(df) == 4
|
|
517
|
+
assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30 + 4 * 40
|
|
518
|
+
|
|
519
|
+
# Column drop (weights length unchanged).
|
|
520
|
+
df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [4, 5, 6]}, weights=[10, 20, 30])
|
|
521
|
+
df.drop(columns=["y"], inplace=True)
|
|
522
|
+
assert list(df.columns) == ["x"]
|
|
523
|
+
assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30
|
|
524
|
+
|
|
525
|
+
# String index row drop.
|
|
526
|
+
df = mdf.MicroDataFrame(
|
|
527
|
+
{"x": [1, 2, 3, 4]},
|
|
528
|
+
index=["a", "b", "c", "d"],
|
|
529
|
+
weights=[10, 20, 30, 40],
|
|
530
|
+
)
|
|
531
|
+
df.drop(index=["a", "b"], inplace=True)
|
|
532
|
+
assert df.x.sum() == 250
|
|
533
|
+
|
|
534
|
+
# labels= with default axis=0.
|
|
535
|
+
df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40])
|
|
536
|
+
df.drop(labels=[0, 1], inplace=True)
|
|
537
|
+
assert df.x.sum() == 250
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def test_merge_preserves_weights_per_surviving_row() -> None:
|
|
541
|
+
"""Regression: merge must propagate weights onto the merged rows.
|
|
542
|
+
|
|
543
|
+
Previously the implementation passed ``self.weights`` straight to the
|
|
544
|
+
MicroDataFrame constructor, so any merge that changed row count
|
|
545
|
+
(inner filtering, left-with-missing, many-to-many, outer) raised
|
|
546
|
+
``ValueError: Length of weights (N) does not match length of
|
|
547
|
+
DataFrame (M)``.
|
|
548
|
+
"""
|
|
549
|
+
# Inner join filters rows.
|
|
550
|
+
left = mdf.MicroDataFrame(
|
|
551
|
+
{"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4]
|
|
552
|
+
)
|
|
553
|
+
right = pd.DataFrame({"k": [2, 4], "w": [20, 40]})
|
|
554
|
+
res = left.merge(right, on="k")
|
|
555
|
+
assert len(res) == 2
|
|
556
|
+
# k=2 carries weight 2; k=4 carries weight 4.
|
|
557
|
+
np.testing.assert_array_equal(sorted(res.weights.values), [2.0, 4.0])
|
|
558
|
+
assert res.v.sum() == 2 * 20 + 4 * 40
|
|
559
|
+
|
|
560
|
+
# Left join with missing from right.
|
|
561
|
+
left = mdf.MicroDataFrame(
|
|
562
|
+
{"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4]
|
|
563
|
+
)
|
|
564
|
+
right = pd.DataFrame({"k": [2, 4], "w": [100, 200]})
|
|
565
|
+
res = left.merge(right, on="k", how="left")
|
|
566
|
+
assert len(res) == 4
|
|
567
|
+
assert res.v.sum() == 300 # 1*10 + 2*20 + 3*30 + 4*40
|
|
568
|
+
|
|
569
|
+
# Many-to-many duplicates left rows; the same weight should ride
|
|
570
|
+
# along on each duplicate.
|
|
571
|
+
left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[5, 7])
|
|
572
|
+
right = pd.DataFrame({"k": [1, 1, 2], "w": [100, 200, 300]})
|
|
573
|
+
res = left.merge(right, on="k")
|
|
574
|
+
assert len(res) == 3
|
|
575
|
+
# v=10 weighted 5 appears twice, v=20 weighted 7 appears once.
|
|
576
|
+
assert res.v.sum() == 10 * 5 + 10 * 5 + 20 * 7
|
|
577
|
+
|
|
578
|
+
# Outer join: right-only rows have no left weight. We default to 0
|
|
579
|
+
# so they don't silently poison downstream aggregations.
|
|
580
|
+
left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[1, 2])
|
|
581
|
+
right = pd.DataFrame({"k": [2, 3], "w": [20, 30]})
|
|
582
|
+
res = left.merge(right, on="k", how="outer")
|
|
583
|
+
# k=1 -> weight 1, k=2 -> weight 2, k=3 -> weight 0 (right-only).
|
|
584
|
+
assert sorted(res.weights.values) == [0.0, 1.0, 2.0]
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
def test_groupby_does_not_leak_tmp_weights_column() -> None:
|
|
588
|
+
"""Regression: groupby used to mutate self by adding __tmp_weights.
|
|
589
|
+
|
|
590
|
+
Previously, ``MicroDataFrame.groupby`` set ``self["__tmp_weights"]``
|
|
591
|
+
and never cleaned it up, so ``df.columns`` afterwards included the
|
|
592
|
+
weight column and any later ``df.sum()`` or iteration over columns
|
|
593
|
+
picked it up as data.
|
|
594
|
+
"""
|
|
595
|
+
df = mdf.MicroDataFrame({"g": ["a", "a", "b"], "v": [1, 2, 3]}, weights=[1, 2, 3])
|
|
596
|
+
original_cols = list(df.columns)
|
|
597
|
+
_ = df.groupby("g").sum()
|
|
598
|
+
assert list(df.columns) == original_cols
|
|
599
|
+
assert "__tmp_weights" not in df.columns
|
|
600
|
+
|
|
601
|
+
# Groupby by a list of columns should also not leak.
|
|
602
|
+
df2 = mdf.MicroDataFrame(
|
|
603
|
+
{"g1": ["a", "a", "b"], "g2": [1, 1, 2], "v": [1, 2, 3]},
|
|
604
|
+
weights=[1, 2, 3],
|
|
605
|
+
)
|
|
606
|
+
orig2 = list(df2.columns)
|
|
607
|
+
_ = df2.groupby(["g1", "g2"]).v.sum()
|
|
608
|
+
assert list(df2.columns) == orig2
|
|
609
|
+
|
|
610
|
+
# Weighted aggregation is still correct after the fix.
|
|
611
|
+
result = df.groupby("g").v.sum()
|
|
612
|
+
assert result["a"] == 1 * 1 + 2 * 2
|
|
613
|
+
assert result["b"] == 3 * 3
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def test_quantile_skips_zero_weight_rows() -> None:
|
|
617
|
+
"""Regression: quantile(0) shouldn't pick a zero-weight element.
|
|
618
|
+
|
|
619
|
+
Previously, ``np.searchsorted(cumsum_norm, 0, side='left')`` returned
|
|
620
|
+
0 even when that first sorted element had zero weight, so
|
|
621
|
+
``MicroSeries([10, 20, 30], weights=[0, 1, 1]).quantile(0)`` returned
|
|
622
|
+
10 instead of 20. The fix drops zero-weight rows before computing
|
|
623
|
+
the CDF.
|
|
624
|
+
"""
|
|
625
|
+
s = mdf.MicroSeries([10, 20, 30], weights=[0, 1, 1])
|
|
626
|
+
assert s.quantile(0.0) == 20
|
|
627
|
+
assert s.quantile(0.5) == 20
|
|
628
|
+
assert s.quantile(1.0) == 30
|
|
629
|
+
|
|
630
|
+
# Internal plateau of zero weight.
|
|
631
|
+
s = mdf.MicroSeries([10, 20, 30, 40], weights=[1, 0, 1, 1])
|
|
632
|
+
assert s.quantile(0.0) == 10
|
|
633
|
+
# Post-filter values [10, 30, 40] with equal weights -> cum=[.33,.67,1].
|
|
634
|
+
# 0.4 -> smallest cum >= 0.4 is index 1 -> value 30.
|
|
635
|
+
assert s.quantile(0.4) == 30
|
|
636
|
+
# The zero-weight value (20) should never be selected.
|
|
637
|
+
for q in np.linspace(0, 1, 21):
|
|
638
|
+
assert s.quantile(q) != 20
|
|
639
|
+
|
|
640
|
+
# All zero weights -> NaN (defined behaviour).
|
|
641
|
+
s = mdf.MicroSeries([10, 20, 30], weights=[0, 0, 0])
|
|
642
|
+
assert np.isnan(s.quantile(0.5))
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
def test_top_x_pct_share_handles_ties_and_edges() -> None:
|
|
646
|
+
"""Regression: top_x_pct_share double-counted threshold ties.
|
|
647
|
+
|
|
648
|
+
Old implementation: ``self[self >= threshold].sum() / self.sum()``.
|
|
649
|
+
With constant values every call returned 1.0 regardless of the
|
|
650
|
+
requested top percent; ``top_x_pct_share(0)`` returned the share of
|
|
651
|
+
the max bucket instead of 0.
|
|
652
|
+
"""
|
|
653
|
+
# Constant values: the top p% should hold exactly p% of the total.
|
|
654
|
+
for p in [0.0, 0.01, 0.1, 0.5, 1.0]:
|
|
655
|
+
got = mdf.MicroSeries([5] * 10, weights=[1] * 10).top_x_pct_share(p)
|
|
656
|
+
assert np.isclose(got, p), f"top={p}, got {got}"
|
|
657
|
+
|
|
658
|
+
# Non-constant, equal weights.
|
|
659
|
+
s = mdf.MicroSeries(list(range(1, 11)), weights=[1] * 10)
|
|
660
|
+
# Sum 1..10 = 55. Top 10% = top 1 row = 10 -> 10/55.
|
|
661
|
+
assert np.isclose(s.top_x_pct_share(0.1), 10 / 55)
|
|
662
|
+
# Top 50% = rows 6..10 -> 40/55.
|
|
663
|
+
assert np.isclose(s.top_x_pct_share(0.5), 40 / 55)
|
|
664
|
+
# Top 0% = 0, top 100% = 1.
|
|
665
|
+
assert s.top_x_pct_share(0.0) == 0.0
|
|
666
|
+
assert s.top_x_pct_share(1.0) == 1.0
|
|
667
|
+
|
|
668
|
+
# Bottom share complements the top share.
|
|
669
|
+
assert np.isclose(s.bottom_x_pct_share(0.1), 1 - s.top_x_pct_share(0.9))
|
|
670
|
+
|
|
671
|
+
# Ties with unequal totals.
|
|
672
|
+
s_ties = mdf.MicroSeries([1, 1, 10, 10], weights=[1, 1, 1, 1])
|
|
673
|
+
# Top 50% = the two 10s -> 20/22.
|
|
674
|
+
assert np.isclose(s_ties.top_x_pct_share(0.5), 20 / 22)
|
|
675
|
+
|
|
676
|
+
# Downstream helpers still work.
|
|
677
|
+
assert np.isclose(s_ties.top_10_pct_share(), s_ties.top_x_pct_share(0.1))
|
|
678
|
+
assert np.isclose(s_ties.top_50_pct_share(), s_ties.top_x_pct_share(0.5))
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def test_gini_negatives_option_applied() -> None:
|
|
682
|
+
"""Regression: gini(negatives=...) was silently ignored.
|
|
683
|
+
|
|
684
|
+
Both branches of the old implementation sorted ``self`` directly
|
|
685
|
+
rather than the local ``x`` that was mutated by the ``negatives``
|
|
686
|
+
option, so ``negatives='zero'`` and ``negatives='shift'`` did
|
|
687
|
+
nothing.
|
|
688
|
+
"""
|
|
689
|
+
s = mdf.MicroSeries([-5, 0, 10], weights=[1, 1, 1])
|
|
690
|
+
|
|
691
|
+
# Leaving negatives in place now warns.
|
|
692
|
+
with warnings.catch_warnings(record=True) as w:
|
|
693
|
+
warnings.simplefilter("always")
|
|
694
|
+
_ = s.gini()
|
|
695
|
+
user_warnings = [x for x in w if issubclass(x.category, UserWarning)]
|
|
696
|
+
assert len(user_warnings) == 1
|
|
697
|
+
assert "negative" in str(user_warnings[0].message).lower()
|
|
698
|
+
|
|
699
|
+
# 'zero' clamps negatives. Values become [0, 0, 10] with equal
|
|
700
|
+
# weights; closed-form gini = 2/3.
|
|
701
|
+
assert np.isclose(s.gini(negatives="zero"), 2 / 3)
|
|
702
|
+
|
|
703
|
+
# 'shift' adds |min|. Values become [0, 5, 15]; Gini in [0, 1].
|
|
704
|
+
shifted = s.gini(negatives="shift")
|
|
705
|
+
assert 0 <= shifted <= 1
|
|
706
|
+
|
|
707
|
+
# All-zero short-circuits to 0 instead of nan/RuntimeWarning.
|
|
708
|
+
assert mdf.MicroSeries([0, 0, 0], weights=[1, 2, 3]).gini() == 0.0
|
|
709
|
+
|
|
710
|
+
# Invalid negatives arg raises.
|
|
711
|
+
with pytest.raises(ValueError):
|
|
712
|
+
mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]).gini(negatives="bogus")
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
def test_std_var_are_weighted() -> None:
|
|
716
|
+
"""Regression: std/var used to silently fall through to pandas.
|
|
717
|
+
|
|
718
|
+
The old implementation had no override, so a MicroSeries with very
|
|
719
|
+
uneven weights returned the unweighted 1.0. Now std and var treat
|
|
720
|
+
the weights as frequency counts, matching numpy on the replicated
|
|
721
|
+
sample.
|
|
722
|
+
"""
|
|
723
|
+
s = mdf.MicroSeries([1, 2, 3], weights=[100, 1, 1])
|
|
724
|
+
# Unweighted would be 1.0. Weighted std pulls toward the heavy row.
|
|
725
|
+
assert s.std() < 1.0
|
|
726
|
+
assert s.var() < 1.0
|
|
727
|
+
|
|
728
|
+
# Integer-replication equivalence.
|
|
729
|
+
s = mdf.MicroSeries([1, 2, 3], weights=[2, 3, 1])
|
|
730
|
+
rep = np.array([1, 1, 2, 2, 2, 3])
|
|
731
|
+
assert np.isclose(s.std(), np.std(rep, ddof=1))
|
|
732
|
+
assert np.isclose(s.var(), np.var(rep, ddof=1))
|
|
733
|
+
assert np.isclose(s.var(ddof=0), np.var(rep, ddof=0))
|
|
734
|
+
|
|
735
|
+
# NaN handling.
|
|
736
|
+
s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[2, 3, 1])
|
|
737
|
+
assert not np.isnan(s.std())
|
|
738
|
+
assert np.isnan(s.std(skipna=False))
|
|
739
|
+
|
|
740
|
+
# DataFrame dispatch: df.std() / df.var() now return weighted stats.
|
|
741
|
+
df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [10, 20, 30]}, weights=[2, 3, 1])
|
|
742
|
+
np.testing.assert_allclose(
|
|
743
|
+
df.std().values,
|
|
744
|
+
[
|
|
745
|
+
np.std(rep, ddof=1),
|
|
746
|
+
np.std(np.array([10, 10, 20, 20, 20, 30]), ddof=1),
|
|
747
|
+
],
|
|
748
|
+
)
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
def test_cov_corr_warn_when_fallthrough() -> None:
|
|
752
|
+
"""Regression: cov/corr silently returned unweighted pandas values.
|
|
753
|
+
|
|
754
|
+
They still fall through to pandas (a weighted impl is a separate
|
|
755
|
+
issue) but now emit a UserWarning so callers aren't misled.
|
|
756
|
+
"""
|
|
757
|
+
s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
|
|
758
|
+
s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1])
|
|
759
|
+
|
|
760
|
+
with warnings.catch_warnings(record=True) as w:
|
|
761
|
+
warnings.simplefilter("always")
|
|
762
|
+
_ = s1.cov(s2)
|
|
763
|
+
msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
|
|
764
|
+
assert any("unweighted" in m.lower() for m in msgs)
|
|
765
|
+
|
|
766
|
+
with warnings.catch_warnings(record=True) as w:
|
|
767
|
+
warnings.simplefilter("always")
|
|
768
|
+
_ = s1.corr(s2)
|
|
769
|
+
msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
|
|
770
|
+
assert any("unweighted" in m.lower() for m in msgs)
|
|
771
|
+
|
|
772
|
+
|
|
773
|
+
def test_count_skips_nan_by_default() -> None:
|
|
774
|
+
"""Regression: ``count()`` included NaN-row weight, contrary to pandas.
|
|
775
|
+
|
|
776
|
+
Pandas ``Series.count`` skips NaN; MicroSeries returned the full
|
|
777
|
+
weight sum regardless. The fix matches pandas semantics and adds a
|
|
778
|
+
``skipna`` kwarg so callers can opt out.
|
|
779
|
+
"""
|
|
780
|
+
s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[10, 20, 30])
|
|
781
|
+
assert s.count() == 40.0
|
|
782
|
+
assert s.count(skipna=True) == 40.0
|
|
783
|
+
assert s.count(skipna=False) == 60.0
|
|
784
|
+
|
|
785
|
+
# No NaN: skipna is a no-op.
|
|
786
|
+
assert mdf.MicroSeries([1, 2, 3], weights=[2, 3, 4]).count() == 9.0
|
|
787
|
+
|
|
788
|
+
# All NaN: count skips everything.
|
|
789
|
+
all_nan = mdf.MicroSeries([np.nan] * 3, weights=[1, 2, 3])
|
|
790
|
+
assert all_nan.count() == 0.0
|
|
791
|
+
assert all_nan.count(skipna=False) == 6.0
|
|
792
|
+
|
|
793
|
+
|
|
794
|
+
def test_rank_ties_share_bucket() -> None:
|
|
795
|
+
"""Regression: rank used to assign ties to different ranks/buckets.
|
|
796
|
+
|
|
797
|
+
Previously ``rank`` returned the running cumulative weight in sort
|
|
798
|
+
order, so every row — tied or not — got a distinct value. As a
|
|
799
|
+
result ``MicroSeries([5]*5, weights=[1]*5).decile_rank()`` returned
|
|
800
|
+
``[2, 4, 6, 8, 10]`` rather than all 10. With max-rank semantics,
|
|
801
|
+
tied values share the cumulative weight at the end of their tie
|
|
802
|
+
group, so bucketing is stable under ties.
|
|
803
|
+
"""
|
|
804
|
+
# All tied: every element lands in the top decile.
|
|
805
|
+
s = mdf.MicroSeries([5] * 5, weights=[1] * 5)
|
|
806
|
+
np.testing.assert_array_equal(s.rank().values, [5, 5, 5, 5, 5])
|
|
807
|
+
np.testing.assert_array_equal(s.decile_rank().values, [10] * 5)
|
|
808
|
+
np.testing.assert_array_equal(s.quintile_rank().values, [5] * 5)
|
|
809
|
+
|
|
810
|
+
# Partial ties.
|
|
811
|
+
s = mdf.MicroSeries([1, 2, 2, 3], weights=[1, 1, 1, 1])
|
|
812
|
+
np.testing.assert_array_equal(s.rank().values, [1, 3, 3, 4])
|
|
813
|
+
|
|
814
|
+
# pct=True normalizes to (0, 1] and still shares ranks on ties.
|
|
815
|
+
s = mdf.MicroSeries([5] * 4, weights=[1] * 4)
|
|
816
|
+
np.testing.assert_allclose(s.rank(pct=True).values, [1.0, 1.0, 1.0, 1.0])
|
|
817
|
+
|
|
818
|
+
# Non-ties still match the old cumulative-weight behaviour, so the
|
|
819
|
+
# existing ``test_rank`` expectations hold.
|
|
820
|
+
s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6])
|
|
821
|
+
np.testing.assert_array_equal(s.rank().values, [4, 9, 15])
|
|
@@ -38,23 +38,23 @@ class TestMicroSeriesSubclassPreservation:
|
|
|
38
38
|
|
|
39
39
|
# Addition
|
|
40
40
|
result = ms + 1
|
|
41
|
-
assert isinstance(
|
|
42
|
-
result
|
|
43
|
-
)
|
|
41
|
+
assert isinstance(result, MicroSeries), (
|
|
42
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
43
|
+
)
|
|
44
44
|
assert hasattr(result, "weights")
|
|
45
45
|
assert hasattr(result, "set_weights")
|
|
46
46
|
|
|
47
47
|
# Multiplication
|
|
48
48
|
result = ms * 2
|
|
49
|
-
assert isinstance(
|
|
50
|
-
result
|
|
51
|
-
)
|
|
49
|
+
assert isinstance(result, MicroSeries), (
|
|
50
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
51
|
+
)
|
|
52
52
|
|
|
53
53
|
# Division
|
|
54
54
|
result = ms / 2
|
|
55
|
-
assert isinstance(
|
|
56
|
-
result
|
|
57
|
-
)
|
|
55
|
+
assert isinstance(result, MicroSeries), (
|
|
56
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
57
|
+
)
|
|
58
58
|
|
|
59
59
|
def test_microseries_preserved_after_comparison(self):
|
|
60
60
|
"""Comparison operations should return MicroSeries, not plain
|
|
@@ -63,35 +63,33 @@ class TestMicroSeriesSubclassPreservation:
|
|
|
63
63
|
|
|
64
64
|
# Greater than
|
|
65
65
|
result = ms > 1
|
|
66
|
-
assert isinstance(
|
|
67
|
-
result
|
|
68
|
-
)
|
|
66
|
+
assert isinstance(result, MicroSeries), (
|
|
67
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
68
|
+
)
|
|
69
69
|
assert hasattr(result, "weights")
|
|
70
70
|
|
|
71
71
|
# Less than
|
|
72
72
|
result = ms < 3
|
|
73
|
-
assert isinstance(
|
|
74
|
-
result
|
|
75
|
-
)
|
|
73
|
+
assert isinstance(result, MicroSeries), (
|
|
74
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
75
|
+
)
|
|
76
76
|
|
|
77
77
|
def test_microseries_preserved_after_indexing(self):
|
|
78
78
|
"""Indexing operations should return MicroSeries, not plain Series."""
|
|
79
|
-
ms = MicroSeries(
|
|
80
|
-
[1, 2, 3, 4, 5], weights=np.array([1.0, 2.0, 3.0, 4.0, 5.0])
|
|
81
|
-
)
|
|
79
|
+
ms = MicroSeries([1, 2, 3, 4, 5], weights=np.array([1.0, 2.0, 3.0, 4.0, 5.0]))
|
|
82
80
|
|
|
83
81
|
# Boolean indexing
|
|
84
82
|
result = ms[ms > 2]
|
|
85
|
-
assert isinstance(
|
|
86
|
-
result
|
|
87
|
-
)
|
|
83
|
+
assert isinstance(result, MicroSeries), (
|
|
84
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
85
|
+
)
|
|
88
86
|
assert hasattr(result, "weights")
|
|
89
87
|
|
|
90
88
|
# Slice indexing
|
|
91
89
|
result = ms[1:3]
|
|
92
|
-
assert isinstance(
|
|
93
|
-
result
|
|
94
|
-
)
|
|
90
|
+
assert isinstance(result, MicroSeries), (
|
|
91
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
92
|
+
)
|
|
95
93
|
|
|
96
94
|
|
|
97
95
|
class TestMicroDataFrameSubclassPreservation:
|
|
@@ -105,9 +103,7 @@ class TestMicroDataFrameSubclassPreservation:
|
|
|
105
103
|
|
|
106
104
|
# Column access
|
|
107
105
|
col = mdf["a"]
|
|
108
|
-
assert isinstance(
|
|
109
|
-
col, MicroSeries
|
|
110
|
-
), f"Got {type(col)} instead of MicroSeries"
|
|
106
|
+
assert isinstance(col, MicroSeries), f"Got {type(col)} instead of MicroSeries"
|
|
111
107
|
assert hasattr(col, "weights")
|
|
112
108
|
assert hasattr(col, "set_weights")
|
|
113
109
|
|
|
@@ -120,9 +116,9 @@ class TestMicroDataFrameSubclassPreservation:
|
|
|
120
116
|
|
|
121
117
|
# Column operations
|
|
122
118
|
result = mdf["a"] + mdf["b"]
|
|
123
|
-
assert isinstance(
|
|
124
|
-
result
|
|
125
|
-
)
|
|
119
|
+
assert isinstance(result, MicroSeries), (
|
|
120
|
+
f"Got {type(result)} instead of MicroSeries"
|
|
121
|
+
)
|
|
126
122
|
assert hasattr(result, "weights")
|
|
127
123
|
|
|
128
124
|
|
|
@@ -186,9 +182,7 @@ class TestCopyOnWriteCompatibility:
|
|
|
186
182
|
|
|
187
183
|
def test_microdataframe_copy_independent(self):
|
|
188
184
|
"""Copying a MicroDataFrame should create an independent copy."""
|
|
189
|
-
mdf = MicroDataFrame(
|
|
190
|
-
{"a": [1, 2, 3]}, weights=np.array([1.0, 2.0, 3.0])
|
|
191
|
-
)
|
|
185
|
+
mdf = MicroDataFrame({"a": [1, 2, 3]}, weights=np.array([1.0, 2.0, 3.0]))
|
|
192
186
|
mdf_copy = mdf.copy()
|
|
193
187
|
|
|
194
188
|
# Modify original
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: microdf-python
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Weighted pandas DataFrames and Series for survey microdata
|
|
5
5
|
Author-email: Max Ghenis <max@policyengine.org>
|
|
6
6
|
License: MIT
|
|
@@ -11,12 +11,8 @@ Requires-Dist: numpy
|
|
|
11
11
|
Requires-Dist: pandas
|
|
12
12
|
Provides-Extra: dev
|
|
13
13
|
Requires-Dist: codecov; extra == "dev"
|
|
14
|
-
Requires-Dist:
|
|
15
|
-
Requires-Dist: flake8-pyproject; extra == "dev"
|
|
16
|
-
Requires-Dist: black; extra == "dev"
|
|
14
|
+
Requires-Dist: ruff>=0.9.0; extra == "dev"
|
|
17
15
|
Requires-Dist: docformatter; extra == "dev"
|
|
18
|
-
Requires-Dist: isort; extra == "dev"
|
|
19
|
-
Requires-Dist: linecheck; extra == "dev"
|
|
20
16
|
Requires-Dist: pytest; extra == "dev"
|
|
21
17
|
Requires-Dist: pytest-cov; extra == "dev"
|
|
22
18
|
Requires-Dist: setuptools; extra == "dev"
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
microdf/__init__.py,sha256=ldRvhE8t4cD7qlxQ5yLndMnWEUYpDw4m-jCDV_psXkY,319
|
|
2
|
+
microdf/microdataframe.py,sha256=8r3Ajq71I3QDP_odegM2BS3RIg1Es1ATdWYkxiIcY4c,37265
|
|
3
|
+
microdf/microseries.py,sha256=0jx-CJpWnRHe_vcNg1914Pkct9C5JcfDisFdu0sltR4,33392
|
|
4
|
+
microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
|
|
5
|
+
microdf/tests/test_microseries_dataframe.py,sha256=h_GeIj9o_vgAauQhl7Nue0DPGXINq4HJRN2KceuQ-mw,30985
|
|
6
|
+
microdf/tests/test_pandas3_compatibility.py,sha256=A34Ni_WQ303sSNv-sqv5CGAQp54zj-ZSGAPEBHZslNI,8573
|
|
7
|
+
microdf_python-1.3.0.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
|
|
8
|
+
microdf_python-1.3.0.dist-info/METADATA,sha256=SIb647k9Jm3iMPVGgyAWWQE7xeZW1tI8OFFHrnOmFfc,2311
|
|
9
|
+
microdf_python-1.3.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
10
|
+
microdf_python-1.3.0.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
|
|
11
|
+
microdf_python-1.3.0.dist-info/RECORD,,
|
|
@@ -1,11 +0,0 @@
|
|
|
1
|
-
microdf/__init__.py,sha256=ldRvhE8t4cD7qlxQ5yLndMnWEUYpDw4m-jCDV_psXkY,319
|
|
2
|
-
microdf/microdataframe.py,sha256=2lZU3FAtVCgKiVD-iD_3xu3NMEEl-7Tbn-T-aOR_ihc,33920
|
|
3
|
-
microdf/microseries.py,sha256=2-UmvJJkxycvqtIW6NfQr1HXFs-LTRctvQr6bkVNYdI,25061
|
|
4
|
-
microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
|
|
5
|
-
microdf/tests/test_microseries_dataframe.py,sha256=YidEKYJwwPyCf_ZiAXoc8kydHj4XkZO-0jXQ-olouho,16284
|
|
6
|
-
microdf/tests/test_pandas3_compatibility.py,sha256=p4SZoW59REA5GV84HH9InfCdKQ1nEBGmatnRK84GiGY,8623
|
|
7
|
-
microdf_python-1.2.2.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
|
|
8
|
-
microdf_python-1.2.2.dist-info/METADATA,sha256=o-2nZ_NkEuvOF-BgVG-3oCRr_3X5YdXg15utoUPCfdM,2469
|
|
9
|
-
microdf_python-1.2.2.dist-info/WHEEL,sha256=YCfwYGOYMi5Jhw2fU4yNgwErybb2IX5PEwBKV4ZbdBo,91
|
|
10
|
-
microdf_python-1.2.2.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
|
|
11
|
-
microdf_python-1.2.2.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|