microdf-python 1.2.2__py3-none-any.whl → 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,6 +2,7 @@ import warnings
2
2
 
3
3
  import numpy as np
4
4
  import pandas as pd
5
+ import pytest
5
6
 
6
7
  import microdf as mdf
7
8
  from microdf.microdataframe import MicroDataFrame
@@ -261,9 +262,7 @@ def test_decile_rank() -> None:
261
262
 
262
263
 
263
264
  def test_copy_equals() -> None:
264
- d = mdf.MicroDataFrame(
265
- {"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]
266
- )
265
+ d = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8])
267
266
  d_copy = d.copy()
268
267
  d_copy_diff_weights = d_copy.copy()
269
268
  d_copy_diff_weights.weights *= 2
@@ -275,9 +274,7 @@ def test_copy_equals() -> None:
275
274
 
276
275
 
277
276
  def test_subset() -> None:
278
- df = mdf.MicroDataFrame(
279
- {"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]
280
- )
277
+ df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8])
281
278
  df_no_z = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4]}, weights=[7, 8])
282
279
  assert df[["x", "y"]].equals(df_no_z)
283
280
  df_no_z_diff_weights = df_no_z.copy()
@@ -353,9 +350,7 @@ def test_reset_index_inplace() -> None:
353
350
  # Test 4: Multi-level index
354
351
  arrays = [["bar", "bar", "baz", "baz"], ["one", "two", "one", "two"]]
355
352
  multi_index = pd.MultiIndex.from_arrays(arrays, names=["first", "second"])
356
- df_multi = pd.DataFrame(
357
- {"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=multi_index
358
- )
353
+ df_multi = pd.DataFrame({"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=multi_index)
359
354
  mdf_multi = MicroDataFrame(df_multi, weights=weights)
360
355
  result = mdf_multi.reset_index(level="first")
361
356
  assert "first" in result.columns
@@ -373,9 +368,7 @@ def test_reset_index_inplace() -> None:
373
368
  def test_loc_preserves_weights() -> None:
374
369
  """Test that .loc[] returns MicroDataFrame with proper weights (issue
375
370
  #265)."""
376
- df = mdf.MicroDataFrame(
377
- {"one": [1, 1, 1, 1, 1]}, weights=[10, 20, 30, 40, 50]
378
- )
371
+ df = mdf.MicroDataFrame({"one": [1, 1, 1, 1, 1]}, weights=[10, 20, 30, 40, 50])
379
372
 
380
373
  # Filter all rows (should get same weights)
381
374
  filtered = df.loc[df.one == 1]
@@ -383,9 +376,7 @@ def test_loc_preserves_weights() -> None:
383
376
  assert filtered.one.sum() == 150.0 # Weighted sum
384
377
 
385
378
  # Partial filter
386
- df2 = mdf.MicroDataFrame(
387
- {"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]
388
- )
379
+ df2 = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50])
389
380
  subset = df2.loc[df2.x > 2]
390
381
  assert isinstance(subset, MicroDataFrame)
391
382
  assert subset.x.sum() == 500.0 # 3*30 + 4*40 + 5*50 = 500
@@ -394,9 +385,7 @@ def test_loc_preserves_weights() -> None:
394
385
 
395
386
  def test_iloc_preserves_weights() -> None:
396
387
  """Test that .iloc[] returns MicroDataFrame with proper weights."""
397
- df = mdf.MicroDataFrame(
398
- {"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]
399
- )
388
+ df = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50])
400
389
 
401
390
  # Select rows by position
402
391
  subset = df.iloc[2:5]
@@ -407,9 +396,7 @@ def test_iloc_preserves_weights() -> None:
407
396
 
408
397
  def test_groupby_column_selection() -> None:
409
398
  """Test that groupby column selection preserves weights (issue #193)."""
410
- d = mdf.MicroDataFrame(
411
- dict(g=["a", "a", "b"], y=[1, 2, 3]), weights=[4, 5, 6]
412
- )
399
+ d = mdf.MicroDataFrame(dict(g=["a", "a", "b"], y=[1, 2, 3]), weights=[4, 5, 6])
413
400
 
414
401
  # Test single column string selection
415
402
  result_str = d.groupby("g")["y"].sum()
@@ -459,6 +446,43 @@ def test_mean_no_warning() -> None:
459
446
  assert len(user_warnings) == 0
460
447
 
461
448
 
449
+ def test_sum_with_non_default_index() -> None:
450
+ """Weighted sum must not silently return 0 with a non-default index.
451
+
452
+ Regression test for the bug where ``set_weights`` stored the weights
453
+ Series with a default ``RangeIndex`` regardless of ``self.index``.
454
+ Element-wise ops like ``self.multiply(self.weights)`` then aligned on
455
+ label, producing all-NaN and a silent ``0.0`` from ``.sum()`` while
456
+ ``.mean()`` (which uses a positional ndarray) stayed correct.
457
+ """
458
+ # MicroSeries with custom integer index.
459
+ s = mdf.MicroSeries([1, 2, 3], index=[100, 200, 300], weights=[10, 20, 30])
460
+ assert s.sum() == 140.0
461
+ assert s.weights.index.tolist() == [100, 200, 300]
462
+
463
+ # MicroDataFrame with custom integer index.
464
+ df = mdf.MicroDataFrame(
465
+ {"x": [1, 2, 3]}, index=[100, 200, 300], weights=[10, 20, 30]
466
+ )
467
+ assert df.x.sum() == 140.0
468
+ assert df.weights.index.tolist() == [100, 200, 300]
469
+
470
+ # MicroDataFrame with string index + set_weights via column name.
471
+ df2 = mdf.MicroDataFrame(
472
+ {"x": [1, 2, 3], "w": [10, 20, 30]},
473
+ index=["a", "b", "c"],
474
+ )
475
+ df2.set_weights("w")
476
+ assert df2.x.sum() == 140.0
477
+ assert df2.weights.index.tolist() == ["a", "b", "c"]
478
+
479
+ # Passing a Series with its own index should position-align, not
480
+ # label-align, so sum does not depend on accidental index alignment.
481
+ df3 = mdf.MicroDataFrame({"x": [1, 2, 3]}, index=[100, 200, 300])
482
+ df3.set_weights(pd.Series([10, 20, 30], index=[0, 1, 2]))
483
+ assert df3.x.sum() == 140.0
484
+
485
+
462
486
  def test_repr_no_warning() -> None:
463
487
  """Internal .values usage in __repr__ should NOT emit a warning."""
464
488
  ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6])
@@ -467,3 +491,331 @@ def test_repr_no_warning() -> None:
467
491
  _ = repr(ms)
468
492
  user_warnings = [x for x in w if issubclass(x.category, UserWarning)]
469
493
  assert len(user_warnings) == 0
494
+
495
+
496
+ def test_drop_inplace_aligns_weights() -> None:
497
+ """Regression: ``drop(inplace=True)`` must keep weights in sync.
498
+
499
+ Previously, ``weights_backup = self.weights.copy()`` was taken *before*
500
+ the drop, then reassigned back afterwards — so the weights vector
501
+ kept its original length and any subsequent weighted op either
502
+ raised a length-mismatch ValueError or silently returned 0.
503
+ """
504
+ # Row drop inplace.
505
+ df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40])
506
+ df.drop(index=[0, 1], inplace=True)
507
+ assert len(df) == len(df.weights) == 2
508
+ assert df.x.sum() == 3 * 30 + 4 * 40 # 250
509
+
510
+ # Row drop non-inplace.
511
+ df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40])
512
+ df2 = df.drop(index=[0, 1])
513
+ assert len(df2) == len(df2.weights) == 2
514
+ assert df2.x.sum() == 250
515
+ # Original untouched.
516
+ assert len(df) == 4
517
+ assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30 + 4 * 40
518
+
519
+ # Column drop (weights length unchanged).
520
+ df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [4, 5, 6]}, weights=[10, 20, 30])
521
+ df.drop(columns=["y"], inplace=True)
522
+ assert list(df.columns) == ["x"]
523
+ assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30
524
+
525
+ # String index row drop.
526
+ df = mdf.MicroDataFrame(
527
+ {"x": [1, 2, 3, 4]},
528
+ index=["a", "b", "c", "d"],
529
+ weights=[10, 20, 30, 40],
530
+ )
531
+ df.drop(index=["a", "b"], inplace=True)
532
+ assert df.x.sum() == 250
533
+
534
+ # labels= with default axis=0.
535
+ df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40])
536
+ df.drop(labels=[0, 1], inplace=True)
537
+ assert df.x.sum() == 250
538
+
539
+
540
+ def test_merge_preserves_weights_per_surviving_row() -> None:
541
+ """Regression: merge must propagate weights onto the merged rows.
542
+
543
+ Previously the implementation passed ``self.weights`` straight to the
544
+ MicroDataFrame constructor, so any merge that changed row count
545
+ (inner filtering, left-with-missing, many-to-many, outer) raised
546
+ ``ValueError: Length of weights (N) does not match length of
547
+ DataFrame (M)``.
548
+ """
549
+ # Inner join filters rows.
550
+ left = mdf.MicroDataFrame(
551
+ {"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4]
552
+ )
553
+ right = pd.DataFrame({"k": [2, 4], "w": [20, 40]})
554
+ res = left.merge(right, on="k")
555
+ assert len(res) == 2
556
+ # k=2 carries weight 2; k=4 carries weight 4.
557
+ np.testing.assert_array_equal(sorted(res.weights.values), [2.0, 4.0])
558
+ assert res.v.sum() == 2 * 20 + 4 * 40
559
+
560
+ # Left join with missing from right.
561
+ left = mdf.MicroDataFrame(
562
+ {"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4]
563
+ )
564
+ right = pd.DataFrame({"k": [2, 4], "w": [100, 200]})
565
+ res = left.merge(right, on="k", how="left")
566
+ assert len(res) == 4
567
+ assert res.v.sum() == 300 # 1*10 + 2*20 + 3*30 + 4*40
568
+
569
+ # Many-to-many duplicates left rows; the same weight should ride
570
+ # along on each duplicate.
571
+ left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[5, 7])
572
+ right = pd.DataFrame({"k": [1, 1, 2], "w": [100, 200, 300]})
573
+ res = left.merge(right, on="k")
574
+ assert len(res) == 3
575
+ # v=10 weighted 5 appears twice, v=20 weighted 7 appears once.
576
+ assert res.v.sum() == 10 * 5 + 10 * 5 + 20 * 7
577
+
578
+ # Outer join: right-only rows have no left weight. We default to 0
579
+ # so they don't silently poison downstream aggregations.
580
+ left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[1, 2])
581
+ right = pd.DataFrame({"k": [2, 3], "w": [20, 30]})
582
+ res = left.merge(right, on="k", how="outer")
583
+ # k=1 -> weight 1, k=2 -> weight 2, k=3 -> weight 0 (right-only).
584
+ assert sorted(res.weights.values) == [0.0, 1.0, 2.0]
585
+
586
+
587
+ def test_groupby_does_not_leak_tmp_weights_column() -> None:
588
+ """Regression: groupby used to mutate self by adding __tmp_weights.
589
+
590
+ Previously, ``MicroDataFrame.groupby`` set ``self["__tmp_weights"]``
591
+ and never cleaned it up, so ``df.columns`` afterwards included the
592
+ weight column and any later ``df.sum()`` or iteration over columns
593
+ picked it up as data.
594
+ """
595
+ df = mdf.MicroDataFrame({"g": ["a", "a", "b"], "v": [1, 2, 3]}, weights=[1, 2, 3])
596
+ original_cols = list(df.columns)
597
+ _ = df.groupby("g").sum()
598
+ assert list(df.columns) == original_cols
599
+ assert "__tmp_weights" not in df.columns
600
+
601
+ # Groupby by a list of columns should also not leak.
602
+ df2 = mdf.MicroDataFrame(
603
+ {"g1": ["a", "a", "b"], "g2": [1, 1, 2], "v": [1, 2, 3]},
604
+ weights=[1, 2, 3],
605
+ )
606
+ orig2 = list(df2.columns)
607
+ _ = df2.groupby(["g1", "g2"]).v.sum()
608
+ assert list(df2.columns) == orig2
609
+
610
+ # Weighted aggregation is still correct after the fix.
611
+ result = df.groupby("g").v.sum()
612
+ assert result["a"] == 1 * 1 + 2 * 2
613
+ assert result["b"] == 3 * 3
614
+
615
+
616
+ def test_quantile_skips_zero_weight_rows() -> None:
617
+ """Regression: quantile(0) shouldn't pick a zero-weight element.
618
+
619
+ Previously, ``np.searchsorted(cumsum_norm, 0, side='left')`` returned
620
+ 0 even when that first sorted element had zero weight, so
621
+ ``MicroSeries([10, 20, 30], weights=[0, 1, 1]).quantile(0)`` returned
622
+ 10 instead of 20. The fix drops zero-weight rows before computing
623
+ the CDF.
624
+ """
625
+ s = mdf.MicroSeries([10, 20, 30], weights=[0, 1, 1])
626
+ assert s.quantile(0.0) == 20
627
+ assert s.quantile(0.5) == 20
628
+ assert s.quantile(1.0) == 30
629
+
630
+ # Internal plateau of zero weight.
631
+ s = mdf.MicroSeries([10, 20, 30, 40], weights=[1, 0, 1, 1])
632
+ assert s.quantile(0.0) == 10
633
+ # Post-filter values [10, 30, 40] with equal weights -> cum=[.33,.67,1].
634
+ # 0.4 -> smallest cum >= 0.4 is index 1 -> value 30.
635
+ assert s.quantile(0.4) == 30
636
+ # The zero-weight value (20) should never be selected.
637
+ for q in np.linspace(0, 1, 21):
638
+ assert s.quantile(q) != 20
639
+
640
+ # All zero weights -> NaN (defined behaviour).
641
+ s = mdf.MicroSeries([10, 20, 30], weights=[0, 0, 0])
642
+ assert np.isnan(s.quantile(0.5))
643
+
644
+
645
+ def test_top_x_pct_share_handles_ties_and_edges() -> None:
646
+ """Regression: top_x_pct_share double-counted threshold ties.
647
+
648
+ Old implementation: ``self[self >= threshold].sum() / self.sum()``.
649
+ With constant values every call returned 1.0 regardless of the
650
+ requested top percent; ``top_x_pct_share(0)`` returned the share of
651
+ the max bucket instead of 0.
652
+ """
653
+ # Constant values: the top p% should hold exactly p% of the total.
654
+ for p in [0.0, 0.01, 0.1, 0.5, 1.0]:
655
+ got = mdf.MicroSeries([5] * 10, weights=[1] * 10).top_x_pct_share(p)
656
+ assert np.isclose(got, p), f"top={p}, got {got}"
657
+
658
+ # Non-constant, equal weights.
659
+ s = mdf.MicroSeries(list(range(1, 11)), weights=[1] * 10)
660
+ # Sum 1..10 = 55. Top 10% = top 1 row = 10 -> 10/55.
661
+ assert np.isclose(s.top_x_pct_share(0.1), 10 / 55)
662
+ # Top 50% = rows 6..10 -> 40/55.
663
+ assert np.isclose(s.top_x_pct_share(0.5), 40 / 55)
664
+ # Top 0% = 0, top 100% = 1.
665
+ assert s.top_x_pct_share(0.0) == 0.0
666
+ assert s.top_x_pct_share(1.0) == 1.0
667
+
668
+ # Bottom share complements the top share.
669
+ assert np.isclose(s.bottom_x_pct_share(0.1), 1 - s.top_x_pct_share(0.9))
670
+
671
+ # Ties with unequal totals.
672
+ s_ties = mdf.MicroSeries([1, 1, 10, 10], weights=[1, 1, 1, 1])
673
+ # Top 50% = the two 10s -> 20/22.
674
+ assert np.isclose(s_ties.top_x_pct_share(0.5), 20 / 22)
675
+
676
+ # Downstream helpers still work.
677
+ assert np.isclose(s_ties.top_10_pct_share(), s_ties.top_x_pct_share(0.1))
678
+ assert np.isclose(s_ties.top_50_pct_share(), s_ties.top_x_pct_share(0.5))
679
+
680
+
681
+ def test_gini_negatives_option_applied() -> None:
682
+ """Regression: gini(negatives=...) was silently ignored.
683
+
684
+ Both branches of the old implementation sorted ``self`` directly
685
+ rather than the local ``x`` that was mutated by the ``negatives``
686
+ option, so ``negatives='zero'`` and ``negatives='shift'`` did
687
+ nothing.
688
+ """
689
+ s = mdf.MicroSeries([-5, 0, 10], weights=[1, 1, 1])
690
+
691
+ # Leaving negatives in place now warns.
692
+ with warnings.catch_warnings(record=True) as w:
693
+ warnings.simplefilter("always")
694
+ _ = s.gini()
695
+ user_warnings = [x for x in w if issubclass(x.category, UserWarning)]
696
+ assert len(user_warnings) == 1
697
+ assert "negative" in str(user_warnings[0].message).lower()
698
+
699
+ # 'zero' clamps negatives. Values become [0, 0, 10] with equal
700
+ # weights; closed-form gini = 2/3.
701
+ assert np.isclose(s.gini(negatives="zero"), 2 / 3)
702
+
703
+ # 'shift' adds |min|. Values become [0, 5, 15]; Gini in [0, 1].
704
+ shifted = s.gini(negatives="shift")
705
+ assert 0 <= shifted <= 1
706
+
707
+ # All-zero short-circuits to 0 instead of nan/RuntimeWarning.
708
+ assert mdf.MicroSeries([0, 0, 0], weights=[1, 2, 3]).gini() == 0.0
709
+
710
+ # Invalid negatives arg raises.
711
+ with pytest.raises(ValueError):
712
+ mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]).gini(negatives="bogus")
713
+
714
+
715
+ def test_std_var_are_weighted() -> None:
716
+ """Regression: std/var used to silently fall through to pandas.
717
+
718
+ The old implementation had no override, so a MicroSeries with very
719
+ uneven weights returned the unweighted 1.0. Now std and var treat
720
+ the weights as frequency counts, matching numpy on the replicated
721
+ sample.
722
+ """
723
+ s = mdf.MicroSeries([1, 2, 3], weights=[100, 1, 1])
724
+ # Unweighted would be 1.0. Weighted std pulls toward the heavy row.
725
+ assert s.std() < 1.0
726
+ assert s.var() < 1.0
727
+
728
+ # Integer-replication equivalence.
729
+ s = mdf.MicroSeries([1, 2, 3], weights=[2, 3, 1])
730
+ rep = np.array([1, 1, 2, 2, 2, 3])
731
+ assert np.isclose(s.std(), np.std(rep, ddof=1))
732
+ assert np.isclose(s.var(), np.var(rep, ddof=1))
733
+ assert np.isclose(s.var(ddof=0), np.var(rep, ddof=0))
734
+
735
+ # NaN handling.
736
+ s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[2, 3, 1])
737
+ assert not np.isnan(s.std())
738
+ assert np.isnan(s.std(skipna=False))
739
+
740
+ # DataFrame dispatch: df.std() / df.var() now return weighted stats.
741
+ df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [10, 20, 30]}, weights=[2, 3, 1])
742
+ np.testing.assert_allclose(
743
+ df.std().values,
744
+ [
745
+ np.std(rep, ddof=1),
746
+ np.std(np.array([10, 10, 20, 20, 20, 30]), ddof=1),
747
+ ],
748
+ )
749
+
750
+
751
+ def test_cov_corr_warn_when_fallthrough() -> None:
752
+ """Regression: cov/corr silently returned unweighted pandas values.
753
+
754
+ They still fall through to pandas (a weighted impl is a separate
755
+ issue) but now emit a UserWarning so callers aren't misled.
756
+ """
757
+ s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
758
+ s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1])
759
+
760
+ with warnings.catch_warnings(record=True) as w:
761
+ warnings.simplefilter("always")
762
+ _ = s1.cov(s2)
763
+ msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
764
+ assert any("unweighted" in m.lower() for m in msgs)
765
+
766
+ with warnings.catch_warnings(record=True) as w:
767
+ warnings.simplefilter("always")
768
+ _ = s1.corr(s2)
769
+ msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)]
770
+ assert any("unweighted" in m.lower() for m in msgs)
771
+
772
+
773
+ def test_count_skips_nan_by_default() -> None:
774
+ """Regression: ``count()`` included NaN-row weight, contrary to pandas.
775
+
776
+ Pandas ``Series.count`` skips NaN; MicroSeries returned the full
777
+ weight sum regardless. The fix matches pandas semantics and adds a
778
+ ``skipna`` kwarg so callers can opt out.
779
+ """
780
+ s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[10, 20, 30])
781
+ assert s.count() == 40.0
782
+ assert s.count(skipna=True) == 40.0
783
+ assert s.count(skipna=False) == 60.0
784
+
785
+ # No NaN: skipna is a no-op.
786
+ assert mdf.MicroSeries([1, 2, 3], weights=[2, 3, 4]).count() == 9.0
787
+
788
+ # All NaN: count skips everything.
789
+ all_nan = mdf.MicroSeries([np.nan] * 3, weights=[1, 2, 3])
790
+ assert all_nan.count() == 0.0
791
+ assert all_nan.count(skipna=False) == 6.0
792
+
793
+
794
+ def test_rank_ties_share_bucket() -> None:
795
+ """Regression: rank used to assign ties to different ranks/buckets.
796
+
797
+ Previously ``rank`` returned the running cumulative weight in sort
798
+ order, so every row — tied or not — got a distinct value. As a
799
+ result ``MicroSeries([5]*5, weights=[1]*5).decile_rank()`` returned
800
+ ``[2, 4, 6, 8, 10]`` rather than all 10. With max-rank semantics,
801
+ tied values share the cumulative weight at the end of their tie
802
+ group, so bucketing is stable under ties.
803
+ """
804
+ # All tied: every element lands in the top decile.
805
+ s = mdf.MicroSeries([5] * 5, weights=[1] * 5)
806
+ np.testing.assert_array_equal(s.rank().values, [5, 5, 5, 5, 5])
807
+ np.testing.assert_array_equal(s.decile_rank().values, [10] * 5)
808
+ np.testing.assert_array_equal(s.quintile_rank().values, [5] * 5)
809
+
810
+ # Partial ties.
811
+ s = mdf.MicroSeries([1, 2, 2, 3], weights=[1, 1, 1, 1])
812
+ np.testing.assert_array_equal(s.rank().values, [1, 3, 3, 4])
813
+
814
+ # pct=True normalizes to (0, 1] and still shares ranks on ties.
815
+ s = mdf.MicroSeries([5] * 4, weights=[1] * 4)
816
+ np.testing.assert_allclose(s.rank(pct=True).values, [1.0, 1.0, 1.0, 1.0])
817
+
818
+ # Non-ties still match the old cumulative-weight behaviour, so the
819
+ # existing ``test_rank`` expectations hold.
820
+ s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6])
821
+ np.testing.assert_array_equal(s.rank().values, [4, 9, 15])
@@ -38,23 +38,23 @@ class TestMicroSeriesSubclassPreservation:
38
38
 
39
39
  # Addition
40
40
  result = ms + 1
41
- assert isinstance(
42
- result, MicroSeries
43
- ), f"Got {type(result)} instead of MicroSeries"
41
+ assert isinstance(result, MicroSeries), (
42
+ f"Got {type(result)} instead of MicroSeries"
43
+ )
44
44
  assert hasattr(result, "weights")
45
45
  assert hasattr(result, "set_weights")
46
46
 
47
47
  # Multiplication
48
48
  result = ms * 2
49
- assert isinstance(
50
- result, MicroSeries
51
- ), f"Got {type(result)} instead of MicroSeries"
49
+ assert isinstance(result, MicroSeries), (
50
+ f"Got {type(result)} instead of MicroSeries"
51
+ )
52
52
 
53
53
  # Division
54
54
  result = ms / 2
55
- assert isinstance(
56
- result, MicroSeries
57
- ), f"Got {type(result)} instead of MicroSeries"
55
+ assert isinstance(result, MicroSeries), (
56
+ f"Got {type(result)} instead of MicroSeries"
57
+ )
58
58
 
59
59
  def test_microseries_preserved_after_comparison(self):
60
60
  """Comparison operations should return MicroSeries, not plain
@@ -63,35 +63,33 @@ class TestMicroSeriesSubclassPreservation:
63
63
 
64
64
  # Greater than
65
65
  result = ms > 1
66
- assert isinstance(
67
- result, MicroSeries
68
- ), f"Got {type(result)} instead of MicroSeries"
66
+ assert isinstance(result, MicroSeries), (
67
+ f"Got {type(result)} instead of MicroSeries"
68
+ )
69
69
  assert hasattr(result, "weights")
70
70
 
71
71
  # Less than
72
72
  result = ms < 3
73
- assert isinstance(
74
- result, MicroSeries
75
- ), f"Got {type(result)} instead of MicroSeries"
73
+ assert isinstance(result, MicroSeries), (
74
+ f"Got {type(result)} instead of MicroSeries"
75
+ )
76
76
 
77
77
  def test_microseries_preserved_after_indexing(self):
78
78
  """Indexing operations should return MicroSeries, not plain Series."""
79
- ms = MicroSeries(
80
- [1, 2, 3, 4, 5], weights=np.array([1.0, 2.0, 3.0, 4.0, 5.0])
81
- )
79
+ ms = MicroSeries([1, 2, 3, 4, 5], weights=np.array([1.0, 2.0, 3.0, 4.0, 5.0]))
82
80
 
83
81
  # Boolean indexing
84
82
  result = ms[ms > 2]
85
- assert isinstance(
86
- result, MicroSeries
87
- ), f"Got {type(result)} instead of MicroSeries"
83
+ assert isinstance(result, MicroSeries), (
84
+ f"Got {type(result)} instead of MicroSeries"
85
+ )
88
86
  assert hasattr(result, "weights")
89
87
 
90
88
  # Slice indexing
91
89
  result = ms[1:3]
92
- assert isinstance(
93
- result, MicroSeries
94
- ), f"Got {type(result)} instead of MicroSeries"
90
+ assert isinstance(result, MicroSeries), (
91
+ f"Got {type(result)} instead of MicroSeries"
92
+ )
95
93
 
96
94
 
97
95
  class TestMicroDataFrameSubclassPreservation:
@@ -105,9 +103,7 @@ class TestMicroDataFrameSubclassPreservation:
105
103
 
106
104
  # Column access
107
105
  col = mdf["a"]
108
- assert isinstance(
109
- col, MicroSeries
110
- ), f"Got {type(col)} instead of MicroSeries"
106
+ assert isinstance(col, MicroSeries), f"Got {type(col)} instead of MicroSeries"
111
107
  assert hasattr(col, "weights")
112
108
  assert hasattr(col, "set_weights")
113
109
 
@@ -120,9 +116,9 @@ class TestMicroDataFrameSubclassPreservation:
120
116
 
121
117
  # Column operations
122
118
  result = mdf["a"] + mdf["b"]
123
- assert isinstance(
124
- result, MicroSeries
125
- ), f"Got {type(result)} instead of MicroSeries"
119
+ assert isinstance(result, MicroSeries), (
120
+ f"Got {type(result)} instead of MicroSeries"
121
+ )
126
122
  assert hasattr(result, "weights")
127
123
 
128
124
 
@@ -186,9 +182,7 @@ class TestCopyOnWriteCompatibility:
186
182
 
187
183
  def test_microdataframe_copy_independent(self):
188
184
  """Copying a MicroDataFrame should create an independent copy."""
189
- mdf = MicroDataFrame(
190
- {"a": [1, 2, 3]}, weights=np.array([1.0, 2.0, 3.0])
191
- )
185
+ mdf = MicroDataFrame({"a": [1, 2, 3]}, weights=np.array([1.0, 2.0, 3.0]))
192
186
  mdf_copy = mdf.copy()
193
187
 
194
188
  # Modify original
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.2.2
3
+ Version: 1.3.0
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -11,12 +11,8 @@ Requires-Dist: numpy
11
11
  Requires-Dist: pandas
12
12
  Provides-Extra: dev
13
13
  Requires-Dist: codecov; extra == "dev"
14
- Requires-Dist: flake8; extra == "dev"
15
- Requires-Dist: flake8-pyproject; extra == "dev"
16
- Requires-Dist: black; extra == "dev"
14
+ Requires-Dist: ruff>=0.9.0; extra == "dev"
17
15
  Requires-Dist: docformatter; extra == "dev"
18
- Requires-Dist: isort; extra == "dev"
19
- Requires-Dist: linecheck; extra == "dev"
20
16
  Requires-Dist: pytest; extra == "dev"
21
17
  Requires-Dist: pytest-cov; extra == "dev"
22
18
  Requires-Dist: setuptools; extra == "dev"
@@ -0,0 +1,11 @@
1
+ microdf/__init__.py,sha256=ldRvhE8t4cD7qlxQ5yLndMnWEUYpDw4m-jCDV_psXkY,319
2
+ microdf/microdataframe.py,sha256=8r3Ajq71I3QDP_odegM2BS3RIg1Es1ATdWYkxiIcY4c,37265
3
+ microdf/microseries.py,sha256=0jx-CJpWnRHe_vcNg1914Pkct9C5JcfDisFdu0sltR4,33392
4
+ microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
5
+ microdf/tests/test_microseries_dataframe.py,sha256=h_GeIj9o_vgAauQhl7Nue0DPGXINq4HJRN2KceuQ-mw,30985
6
+ microdf/tests/test_pandas3_compatibility.py,sha256=A34Ni_WQ303sSNv-sqv5CGAQp54zj-ZSGAPEBHZslNI,8573
7
+ microdf_python-1.3.0.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
8
+ microdf_python-1.3.0.dist-info/METADATA,sha256=SIb647k9Jm3iMPVGgyAWWQE7xeZW1tI8OFFHrnOmFfc,2311
9
+ microdf_python-1.3.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
10
+ microdf_python-1.3.0.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
11
+ microdf_python-1.3.0.dist-info/RECORD,,
@@ -1,5 +1,5 @@
1
1
  Wheel-Version: 1.0
2
- Generator: setuptools (82.0.0)
2
+ Generator: setuptools (82.0.1)
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any
5
5
 
@@ -1,11 +0,0 @@
1
- microdf/__init__.py,sha256=ldRvhE8t4cD7qlxQ5yLndMnWEUYpDw4m-jCDV_psXkY,319
2
- microdf/microdataframe.py,sha256=2lZU3FAtVCgKiVD-iD_3xu3NMEEl-7Tbn-T-aOR_ihc,33920
3
- microdf/microseries.py,sha256=2-UmvJJkxycvqtIW6NfQr1HXFs-LTRctvQr6bkVNYdI,25061
4
- microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
5
- microdf/tests/test_microseries_dataframe.py,sha256=YidEKYJwwPyCf_ZiAXoc8kydHj4XkZO-0jXQ-olouho,16284
6
- microdf/tests/test_pandas3_compatibility.py,sha256=p4SZoW59REA5GV84HH9InfCdKQ1nEBGmatnRK84GiGY,8623
7
- microdf_python-1.2.2.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
8
- microdf_python-1.2.2.dist-info/METADATA,sha256=o-2nZ_NkEuvOF-BgVG-3oCRr_3X5YdXg15utoUPCfdM,2469
9
- microdf_python-1.2.2.dist-info/WHEEL,sha256=YCfwYGOYMi5Jhw2fU4yNgwErybb2IX5PEwBKV4ZbdBo,91
10
- microdf_python-1.2.2.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
11
- microdf_python-1.2.2.dist-info/RECORD,,