microdf-python 1.2.2__py3-none-any.whl → 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
microdf/microseries.py CHANGED
@@ -9,6 +9,46 @@ import pandas as pd
9
9
  logger = logging.getLogger(__name__)
10
10
 
11
11
 
12
+ def _weighted_top_share(
13
+ values: np.ndarray, weights: np.ndarray, top_x_pct: float
14
+ ) -> float:
15
+ """Share of the sum held by the top ``top_x_pct`` of weight.
16
+
17
+ Sort by value ascending, cumulate the weight, pick the slice from
18
+ the top that covers exactly ``top_x_pct`` of total weight, and
19
+ distribute the tied-at-cutoff row proportionally so constant values
20
+ return exactly ``top_x_pct`` rather than 1.0.
21
+ """
22
+ if top_x_pct <= 0:
23
+ return 0.0
24
+ if top_x_pct >= 1:
25
+ return 1.0
26
+ total_weight = weights.sum()
27
+ total_sum = float((values * weights).sum())
28
+ if total_weight == 0 or total_sum == 0:
29
+ return np.nan
30
+ # Ascending sort; the "top" cutoff is the final ``top_x_pct`` of
31
+ # cumulative weight.
32
+ order = np.argsort(values, kind="mergesort")
33
+ v = values[order]
34
+ w = weights[order]
35
+ # Cumulative weight from the bottom up.
36
+ cum_w = np.cumsum(w)
37
+ target_bottom_weight = total_weight * (1.0 - top_x_pct)
38
+ # searchsorted(cum_w, target, side="right") gives the first index
39
+ # whose cumulative weight exceeds the bottom cutoff.
40
+ k = int(np.searchsorted(cum_w, target_bottom_weight, side="right"))
41
+ # Rows strictly above the cutoff contribute all of their weight.
42
+ if k >= len(v):
43
+ return 0.0
44
+ top_sum = float((v[k + 1 :] * w[k + 1 :]).sum())
45
+ # Row k straddles the cutoff; include the fraction of its weight
46
+ # that lies above the cutoff so ties don't double-count.
47
+ partial_weight = cum_w[k] - target_bottom_weight
48
+ top_sum += float(v[k] * partial_weight)
49
+ return top_sum / total_sum
50
+
51
+
12
52
  class MicroSeries(pd.Series):
13
53
  def __init__(self, *args, weights: np.array = None, **kwargs):
14
54
  """A Series-inheriting class for weighted microdata. Weights can be
@@ -64,23 +104,13 @@ class MicroSeries(pd.Series):
64
104
  )
65
105
  return super().to_numpy(*args, **kwargs)
66
106
 
67
- def weighted_function(fn: Callable) -> Callable:
68
- @wraps(fn)
69
- def safe_fn(*args, **kwargs):
70
- try:
71
- return fn(*args, **kwargs)
72
- except ZeroDivisionError:
73
- return np.NaN
74
-
75
- return safe_fn
76
-
77
- @weighted_function
78
107
  def scalar_function(fn: Callable) -> Callable:
108
+ """Decorator marking ``fn`` as returning a scalar (float)."""
79
109
  fn._rtype = float
80
110
  return fn
81
111
 
82
- @weighted_function
83
112
  def vector_function(fn: Callable) -> Callable:
113
+ """Decorator marking ``fn`` as returning a pandas Series."""
84
114
  fn._rtype = pd.Series
85
115
  return fn
86
116
 
@@ -97,7 +127,9 @@ class MicroSeries(pd.Series):
97
127
  if weights is None:
98
128
  if len(self) > 0:
99
129
  self.weights = pd.Series(
100
- np.ones_like(self._values), dtype=float
130
+ np.ones_like(self._values),
131
+ index=self.index,
132
+ dtype=float,
101
133
  )
102
134
  else:
103
135
  if len(weights) != len(self):
@@ -109,7 +141,14 @@ class MicroSeries(pd.Series):
109
141
  if preserve_old and self.weights is not None:
110
142
  self["old_weights"] = self.weights
111
143
 
112
- self.weights = pd.Series(weights, dtype=float)
144
+ # Align weights to self.index so element-wise operations such
145
+ # as self.multiply(self.weights) (used by .sum(), .weight())
146
+ # don't silently produce all-NaN when the caller uses a
147
+ # non-default index. If a pandas Series is passed in, strip
148
+ # its index first so we position-align rather than label-align.
149
+ if isinstance(weights, pd.Series):
150
+ weights = weights.values
151
+ self.weights = pd.Series(np.asarray(weights), index=self.index, dtype=float)
113
152
 
114
153
  def nullify_weights(self) -> None:
115
154
  """Set all weights to 1, effectively making the Series unweighted.
@@ -138,12 +177,22 @@ class MicroSeries(pd.Series):
138
177
  return self.multiply(self.weights).sum()
139
178
 
140
179
  @scalar_function
141
- def count(self) -> float:
180
+ def count(self, skipna: bool = True) -> float:
142
181
  """Calculates the weighted count of the MicroSeries.
143
182
 
183
+ By default skips NaN values (matching pandas ``Series.count``).
184
+
185
+ :param skipna: Exclude NaN values (default True). If False, the
186
+ weighted count of every row is returned.
187
+ :type skipna: bool
144
188
  :returns: The weighted count.
189
+ :rtype: float
145
190
  """
146
- return self.weights.sum()
191
+ weights = np.asarray(self.weights.values, dtype=float)
192
+ if not skipna:
193
+ return float(weights.sum())
194
+ mask = ~pd.isna(self._values)
195
+ return float(weights[mask].sum())
147
196
 
148
197
  @scalar_function
149
198
  def mean(self, skipna: bool = True) -> float:
@@ -173,6 +222,89 @@ class MicroSeries(pd.Series):
173
222
 
174
223
  return np.average(values, weights=weights)
175
224
 
225
+ def _weighted_variance(self, ddof: int = 1, skipna: bool = True) -> float:
226
+ """Frequency-weighted variance.
227
+
228
+ Uses ``sum(w * (x - wmean)**2) / (sum(w) - ddof)``. With
229
+ ``ddof=0`` this is the population variance; with ``ddof=1`` it
230
+ is Bessel-corrected assuming the weights are frequency counts —
231
+ matching ``np.var(..., ddof=ddof)`` on a replicated sample.
232
+ """
233
+ values = np.asarray(self._values, dtype=float)
234
+ weights = np.asarray(self.weights.values, dtype=float)
235
+ if skipna:
236
+ mask = ~np.isnan(values)
237
+ values = values[mask]
238
+ weights = weights[mask]
239
+ elif np.isnan(values).any():
240
+ return np.nan
241
+ total_w = weights.sum()
242
+ if total_w == 0 or total_w - ddof <= 0:
243
+ return np.nan
244
+ mean = np.average(values, weights=weights)
245
+ return float((weights * (values - mean) ** 2).sum() / (total_w - ddof))
246
+
247
+ @scalar_function
248
+ def var(self, ddof: int = 1, skipna: bool = True) -> float:
249
+ """Calculates the weighted variance of the MicroSeries.
250
+
251
+ Treats weights as frequency counts (``sum(w) - ddof`` in the
252
+ denominator) so that with integer weights the result matches
253
+ ``np.var`` on the replicated sample.
254
+
255
+ :param ddof: Delta degrees of freedom (default 1).
256
+ :param skipna: Exclude NaN values (default True).
257
+ :returns: The weighted variance.
258
+ :rtype: float
259
+ """
260
+ return self._weighted_variance(ddof=ddof, skipna=skipna)
261
+
262
+ @scalar_function
263
+ def std(self, ddof: int = 1, skipna: bool = True) -> float:
264
+ """Calculates the weighted standard deviation of the MicroSeries.
265
+
266
+ :param ddof: Delta degrees of freedom (default 1).
267
+ :param skipna: Exclude NaN values (default True).
268
+ :returns: The weighted standard deviation.
269
+ :rtype: float
270
+ """
271
+ v = self._weighted_variance(ddof=ddof, skipna=skipna)
272
+ return float(np.sqrt(v)) if np.isfinite(v) else v
273
+
274
+ def cov(self, other, *args, **kwargs):
275
+ """Pandas ``cov`` — **unweighted**.
276
+
277
+ MicroSeries does not yet compute weighted covariance. Emits a
278
+ ``UserWarning`` so callers aren't silently given an unweighted
279
+ number after ``.sum()`` and ``.mean()`` worked as expected. See
280
+ issue tracker for a weighted implementation.
281
+ """
282
+ warnings.warn(
283
+ "MicroSeries.cov() falls through to pandas and is "
284
+ "unweighted. Use MicroSeries.var()/std() for weighted "
285
+ "second moments, or compute covariance manually with the "
286
+ "weights.",
287
+ UserWarning,
288
+ stacklevel=2,
289
+ )
290
+ return super().cov(other, *args, **kwargs)
291
+
292
+ def corr(self, other, *args, **kwargs):
293
+ """Pandas ``corr`` — **unweighted**.
294
+
295
+ MicroSeries does not yet compute weighted correlation. Emits a
296
+ ``UserWarning`` so callers aren't silently given an unweighted
297
+ number. See issue tracker for a weighted implementation.
298
+ """
299
+ warnings.warn(
300
+ "MicroSeries.corr() falls through to pandas and is "
301
+ "unweighted. Compute correlation manually with the weights "
302
+ "if you need the survey-weighted value.",
303
+ UserWarning,
304
+ stacklevel=2,
305
+ )
306
+ return super().corr(other, *args, **kwargs)
307
+
176
308
  def quantile(self, q: np.array) -> pd.Series:
177
309
  """Calculates weighted quantiles of the MicroSeries.
178
310
 
@@ -189,9 +321,23 @@ class MicroSeries(pd.Series):
189
321
  values = np.array(self._values)
190
322
  quantiles = np.atleast_1d(q)
191
323
  sample_weight = np.array(self.weights)
192
- assert np.all(quantiles >= 0) and np.all(
193
- quantiles <= 1
194
- ), "quantiles should be in [0, 1]"
324
+ assert np.all(quantiles >= 0) and np.all(quantiles <= 1), (
325
+ "quantiles should be in [0, 1]"
326
+ )
327
+ # Drop zero-weight rows before sorting. Without this, q=0 (and
328
+ # internal plateaus of zero weight) picked a value with 0 weight
329
+ # that should have been skipped by the inverse CDF. E.g.
330
+ # MicroSeries([10, 20, 30], weights=[0, 1, 1]).quantile(0)
331
+ # returned 10 instead of 20.
332
+ nonzero = sample_weight > 0
333
+ if not nonzero.any():
334
+ return (
335
+ np.nan
336
+ if np.array(q).shape == ()
337
+ else pd.Series(np.full(len(quantiles), np.nan), index=quantiles)
338
+ )
339
+ values = values[nonzero]
340
+ sample_weight = sample_weight[nonzero]
195
341
  sorter = np.argsort(values)
196
342
  values = values[sorter]
197
343
  sample_weight = sample_weight[sorter]
@@ -199,11 +345,7 @@ class MicroSeries(pd.Series):
199
345
  cumsum_normalized = cumsum / cumsum[-1]
200
346
  result = np.array(
201
347
  [
202
- values[
203
- min(
204
- np.searchsorted(cumsum_normalized, qi), len(values) - 1
205
- )
206
- ]
348
+ values[min(np.searchsorted(cumsum_normalized, qi), len(values) - 1)]
207
349
  for qi in quantiles
208
350
  ]
209
351
  )
@@ -224,58 +366,83 @@ class MicroSeries(pd.Series):
224
366
  def gini(self, negatives: Optional[str] = None) -> float:
225
367
  """Calculates Gini index.
226
368
 
227
- :param negatives: An optional string indicating how to treat negative
228
- values of x:
369
+ :param negatives: An optional string indicating how to treat
370
+ negative values of x:
229
371
  'zero' replaces negative values with zeroes.
230
372
  'shift' subtracts the minimum value from all values of x,
231
373
  when this minimum is negative. That is, it adds the absolute
232
374
  minimum value.
233
375
  Defaults to None, which leaves negative values as they are.
234
- :type q: str
376
+ :type negatives: str
235
377
  :returns: Gini index.
236
378
  :rtype: float
237
379
  """
238
380
  x = np.array(self).astype("float")
381
+ w = np.asarray(self.weights.values, dtype=float)
239
382
  if negatives == "zero":
240
- x[x < 0] = 0
241
- if negatives == "shift" and np.amin(x) < 0:
242
- x -= np.amin(x)
243
- if (self.weights != np.ones(len(self))).any(): # Varying weights.
244
- sorted_indices = np.argsort(self)
245
- sorted_x = np.array(self[sorted_indices])
246
- sorted_w = np.array(self.weights[sorted_indices])
247
- cumw = np.cumsum(sorted_w)
248
- cumxw = np.cumsum(sorted_x * sorted_w)
249
- return np.sum(cumxw[1:] * cumw[:-1] - cumxw[:-1] * cumw[1:]) / (
250
- cumxw[-1] * cumw[-1]
383
+ x = np.where(x < 0, 0.0, x)
384
+ elif negatives == "shift" and len(x) > 0 and np.amin(x) < 0:
385
+ x = x - np.amin(x)
386
+ elif negatives is not None:
387
+ raise ValueError(
388
+ f"Unknown negatives option {negatives!r}; expected "
389
+ "'zero', 'shift', or None."
251
390
  )
252
- else:
253
- sorted_x = np.sort(self)
254
- n = len(x)
255
- cumxw = np.cumsum(sorted_x)
256
- # The above formula, with all weights equal to 1 simplifies to:
257
- return (n + 1 - 2 * np.sum(cumxw) / cumxw[-1]) / n
391
+
392
+ if len(x) == 0:
393
+ return np.nan
394
+ if np.any(x < 0):
395
+ # The Lorenz-based formula assumes non-negative values; with
396
+ # negatives it can return values outside [0, 1].
397
+ warnings.warn(
398
+ "gini() called on data containing negative values; the "
399
+ "result is not guaranteed to lie in [0, 1]. Pass "
400
+ "negatives='zero' or negatives='shift' to handle them.",
401
+ UserWarning,
402
+ stacklevel=2,
403
+ )
404
+
405
+ # Short-circuit degenerate cases so we don't divide by zero.
406
+ total = float((x * w).sum())
407
+ if total == 0:
408
+ return 0.0
409
+
410
+ sorter = np.argsort(x, kind="mergesort")
411
+ sorted_x = x[sorter]
412
+ sorted_w = w[sorter]
413
+ cumw = np.cumsum(sorted_w)
414
+ cumxw = np.cumsum(sorted_x * sorted_w)
415
+ # Trapezoidal approximation of the area under the Lorenz curve.
416
+ return float(
417
+ np.sum(cumxw[1:] * cumw[:-1] - cumxw[:-1] * cumw[1:])
418
+ / (cumxw[-1] * cumw[-1])
419
+ )
258
420
 
259
421
  @scalar_function
260
422
  def top_x_pct_share(self, top_x_pct: float) -> float:
261
423
  """Calculates top x% share.
262
424
 
425
+ Uses a cumulative-weight sort so that rows tied at the cutoff
426
+ contribute proportionally rather than all-or-nothing. With
427
+ constant values this correctly returns ``top_x_pct`` itself.
428
+
263
429
  :param top_x_pct: Decimal between 0 and 1 of the top %, e.g. 0.1,
264
430
  0.001.
265
431
  :type top_x_pct: float
266
432
  :returns: The weighted share held by the top x%.
267
433
  :rtype: float
268
434
  """
269
- threshold = self.quantile(1 - top_x_pct)
270
- top_x_pct_sum = self[self >= threshold].sum()
271
- total_sum = self.sum()
272
- return top_x_pct_sum / total_sum
435
+ return _weighted_top_share(
436
+ np.asarray(self._values, dtype=float),
437
+ np.asarray(self.weights.values, dtype=float),
438
+ float(top_x_pct),
439
+ )
273
440
 
274
441
  @scalar_function
275
442
  def bottom_x_pct_share(self, bottom_x_pct: float) -> float:
276
443
  """Calculates bottom x% share.
277
444
 
278
- :param bottom_x_pct: Decimal between 0 and 1 of the top %, e.g. 0.1,
445
+ :param bottom_x_pct: Decimal between 0 and 1 of the bottom %, e.g. 0.1,
279
446
  0.001.
280
447
  :type bottom_x_pct: float
281
448
  :returns: The weighted share held by the bottom x%.
@@ -350,19 +517,44 @@ class MicroSeries(pd.Series):
350
517
 
351
518
  @vector_function
352
519
  def rank(self, pct: Optional[bool] = False) -> pd.Series:
353
- weights_sum = self.weights.values.sum()
520
+ """Weighted rank of each element.
521
+
522
+ Each element's rank is the cumulative weight of all values that
523
+ are less than or equal to it. Tied values therefore share the
524
+ same rank, so downstream bucketing (``decile_rank``,
525
+ ``quintile_rank``, etc.) lands tied rows in the same bucket.
526
+
527
+ :param pct: If True, divide ranks by the total weight so they
528
+ lie in ``(0, 1]``.
529
+ :type pct: bool
530
+ :returns: MicroSeries of ranks aligned to ``self``.
531
+ :rtype: MicroSeries
532
+ """
533
+ weights_sum = np.asarray(self.weights.values, dtype=float).sum()
354
534
  if weights_sum == 0:
355
535
  raise ZeroDivisionError(
356
536
  "Cannot calculate rank with zero total weight. "
357
- "All weights in the MicroSeries are zero, which would result "
358
- "in division by zero."
537
+ "All weights in the MicroSeries are zero, which would "
538
+ "result in division by zero."
359
539
  )
360
540
 
361
- order = np.argsort(self._values)
362
- inverse_order = np.argsort(order)
363
- ranks = np.array(self.weights.values)[order].cumsum()[inverse_order]
541
+ values = np.asarray(self._values)
542
+ weights = np.asarray(self.weights.values, dtype=float)
543
+ order = np.argsort(values, kind="mergesort")
544
+ sorted_values = values[order]
545
+ sorted_weights = weights[order]
546
+ cum_w = np.cumsum(sorted_weights)
547
+ # Max rank semantics: every tied group gets the cumulative
548
+ # weight at the *end* of the group, so ties share one rank.
549
+ # searchsorted(side='right') on the sorted values finds the
550
+ # index just past each tied block in sort order.
551
+ group_end = np.searchsorted(sorted_values, sorted_values, side="right") - 1
552
+ sorted_ranks = cum_w[group_end]
553
+ # Invert the sort to put ranks back into the caller's order.
554
+ inverse_order = np.argsort(order, kind="mergesort")
555
+ ranks = sorted_ranks[inverse_order]
364
556
  if pct:
365
- ranks /= weights_sum
557
+ ranks = ranks / weights_sum
366
558
  ranks = np.where(ranks > 1.0, 1.0, ranks)
367
559
  return MicroSeries(ranks, index=self.index, weights=self.weights)
368
560
 
@@ -454,9 +646,7 @@ class MicroSeries(pd.Series):
454
646
  return MicroSeries(res, weights=self.weights)
455
647
  return self
456
648
 
457
- def round(
458
- self, decimals: Optional[int] = 0, *args, **kwargs
459
- ) -> "MicroSeries":
649
+ def round(self, decimals: Optional[int] = 0, *args, **kwargs) -> "MicroSeries":
460
650
  res = super().round(decimals=decimals, *args, **kwargs)
461
651
  return MicroSeries(res, weights=self.weights)
462
652
 
@@ -488,14 +678,10 @@ class MicroSeries(pd.Series):
488
678
  def __mul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
489
679
  return MicroSeries(super().__mul__(other), weights=self.weights)
490
680
 
491
- def __floordiv__(
492
- self, other: Union[int, float, pd.Series]
493
- ) -> "MicroSeries":
681
+ def __floordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
494
682
  return MicroSeries(super().__floordiv__(other), weights=self.weights)
495
683
 
496
- def __truediv__(
497
- self, other: Union[int, float, pd.Series]
498
- ) -> "MicroSeries":
684
+ def __truediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
499
685
  return MicroSeries(super().__truediv__(other), weights=self.weights)
500
686
 
501
687
  def __mod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
@@ -525,14 +711,10 @@ class MicroSeries(pd.Series):
525
711
  def __rmul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
526
712
  return MicroSeries(super().__rmul__(other), weights=self.weights)
527
713
 
528
- def __rfloordiv__(
529
- self, other: Union[int, float, pd.Series]
530
- ) -> "MicroSeries":
714
+ def __rfloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
531
715
  return MicroSeries(super().__rfloordiv__(other), weights=self.weights)
532
716
 
533
- def __rtruediv__(
534
- self, other: Union[int, float, pd.Series]
535
- ) -> "MicroSeries":
717
+ def __rtruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
536
718
  return MicroSeries(super().__rtruediv__(other), weights=self.weights)
537
719
 
538
720
  def __rmod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
@@ -585,17 +767,13 @@ class MicroSeries(pd.Series):
585
767
  def __imul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
586
768
  return MicroSeries(super().__imul__(other), weights=self.weights)
587
769
 
588
- def __ifloordiv__(
589
- self, other: Union[int, float, pd.Series]
590
- ) -> "MicroSeries":
770
+ def __ifloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
591
771
  return MicroSeries(super().__ifloordiv__(other), weights=self.weights)
592
772
 
593
773
  def __idiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
594
774
  return MicroSeries(super().__idiv__(other), weights=self.weights)
595
775
 
596
- def __itruediv__(
597
- self, other: Union[int, float, pd.Series]
598
- ) -> "MicroSeries":
776
+ def __itruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
599
777
  return MicroSeries(super().__itruediv__(other), weights=self.weights)
600
778
 
601
779
  def __imod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries":
@@ -666,16 +844,12 @@ class MicroSeriesGroupBy(pd.core.groupby.generic.SeriesGroupBy):
666
844
  def _init(self):
667
845
  def _weighted_agg(name) -> Callable:
668
846
  def via_micro_series(row, *args, **kwargs):
669
- return getattr(MicroSeries(row.a, weights=row.w), name)(
670
- *args, **kwargs
671
- )
847
+ return getattr(MicroSeries(row.a, weights=row.w), name)(*args, **kwargs)
672
848
 
673
849
  fn = getattr(MicroSeries, name)
674
850
 
675
851
  @wraps(fn)
676
- def _weighted_agg_fn(
677
- *args, **kwargs
678
- ) -> Union[pd.Series, pd.DataFrame]:
852
+ def _weighted_agg_fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]:
679
853
  arrays = self.apply(np.array)
680
854
  weights = self.weights.apply(np.array)
681
855
  df = pd.DataFrame(dict(a=arrays, w=weights))