microdf-python 1.5.8__py3-none-any.whl → 1.5.10__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
microdf/_docs.py ADDED
@@ -0,0 +1,144 @@
1
+ """Version-stable rendering of signatures for the API reference.
2
+
3
+ `str(inspect.signature(...))` is not stable across versions: pandas 2 renders a
4
+ Series annotation as ``pandas.core.series.Series`` and pandas 3 renders the
5
+ same annotation as ``pandas.Series``, and from Python 3.14 ``Optional[int]``
6
+ reprs as ``int | None``. Generating ``docs/api.md`` from one of those and
7
+ testing it against another is a guaranteed CI failure on some job.
8
+
9
+ So the page and its test both render signatures through the functions here,
10
+ which strip module qualifiers and put unions into a single form. There is one
11
+ renderer, imported from one place, so the page cannot disagree with the test.
12
+ """
13
+
14
+ import inspect
15
+ import re
16
+ import types
17
+ import typing
18
+
19
+ __all__ = ["render_annotation", "render_signature", "markdown_signature"]
20
+
21
+ _NoneType = type(None)
22
+
23
+ # ``pandas.core.series.Series`` -> ``Series``. Requires a dot, so string
24
+ # literals inside e.g. ``Literal['a', 'b']`` are untouched.
25
+ _QUALIFIER = re.compile(r"\b(?:[A-Za-z_][A-Za-z0-9_]*\.)+([A-Za-z_][A-Za-z0-9_]*)\b")
26
+
27
+
28
+ def _split_top_level(text, sep="|"):
29
+ parts, depth, current = [], 0, ""
30
+ for char in text:
31
+ if char in "[({":
32
+ depth += 1
33
+ elif char in "])}":
34
+ depth -= 1
35
+ if char == sep and depth == 0:
36
+ parts.append(current)
37
+ current = ""
38
+ else:
39
+ current += char
40
+ parts.append(current)
41
+ return [part.strip() for part in parts]
42
+
43
+
44
+ def _union(rendered):
45
+ """One spelling for a union, whatever the source spelling was."""
46
+ non_none = [part for part in rendered if part not in ("None", "NoneType")]
47
+ nones = len(rendered) - len(non_none)
48
+ if nones and len(non_none) == 1:
49
+ return f"Optional[{non_none[0]}]"
50
+ if nones:
51
+ return "Union[" + ", ".join(non_none + ["NoneType"]) + "]"
52
+ return "Union[" + ", ".join(non_none) + "]"
53
+
54
+
55
+ def _clean_text(text):
56
+ """Normalise an annotation that reaches us as a string.
57
+
58
+ ``from __future__ import annotations`` in pandas means many of its
59
+ signatures carry string annotations such as ``'int | None'``.
60
+ """
61
+ text = text.strip()
62
+ if len(text) > 1 and text[0] == text[-1] and text[0] in "\"'":
63
+ text = text[1:-1].strip()
64
+ text = _QUALIFIER.sub(r"\1", text)
65
+ if "|" in text:
66
+ parts = _split_top_level(text)
67
+ if len(parts) > 1:
68
+ return _union([_clean_text(part) for part in parts])
69
+ return text
70
+
71
+
72
+ def render_annotation(annotation):
73
+ """Render an annotation identically under any supported pandas/Python."""
74
+ if isinstance(annotation, str):
75
+ return _clean_text(annotation)
76
+ if isinstance(annotation, typing.ForwardRef):
77
+ return _clean_text(annotation.__forward_arg__)
78
+ if annotation is _NoneType:
79
+ return "NoneType"
80
+
81
+ origin = typing.get_origin(annotation)
82
+ if origin is not None:
83
+ args = typing.get_args(annotation)
84
+ if origin is typing.Union or origin is getattr(types, "UnionType", ()):
85
+ return _union([render_annotation(arg) for arg in args])
86
+ name = getattr(origin, "__name__", None) or _clean_text(repr(origin))
87
+ if not args:
88
+ return name
89
+ return f"{name}[{', '.join(render_annotation(arg) for arg in args)}]"
90
+
91
+ if isinstance(annotation, type):
92
+ return annotation.__name__
93
+ return _clean_text(repr(annotation))
94
+
95
+
96
+ def render_signature(func):
97
+ """Render ``func``'s signature, without ``self``, version-stably.
98
+
99
+ Parameter names, kinds and defaults come straight from
100
+ ``inspect.signature``; only the annotation text is normalised.
101
+ """
102
+ signature = inspect.signature(func)
103
+ parts, previous_kind = [], None
104
+ for parameter in signature.parameters.values():
105
+ if parameter.name == "self" and not parts:
106
+ previous_kind = parameter.kind
107
+ continue
108
+ if (
109
+ previous_kind is inspect.Parameter.POSITIONAL_ONLY
110
+ and parameter.kind is not inspect.Parameter.POSITIONAL_ONLY
111
+ ):
112
+ parts.append("/")
113
+ if parameter.kind is inspect.Parameter.KEYWORD_ONLY and previous_kind not in (
114
+ inspect.Parameter.VAR_POSITIONAL,
115
+ inspect.Parameter.KEYWORD_ONLY,
116
+ ):
117
+ parts.append("*")
118
+
119
+ rendered = parameter.name
120
+ if parameter.kind is inspect.Parameter.VAR_POSITIONAL:
121
+ rendered = "*" + rendered
122
+ elif parameter.kind is inspect.Parameter.VAR_KEYWORD:
123
+ rendered = "**" + rendered
124
+ if parameter.annotation is not inspect.Parameter.empty:
125
+ rendered += f": {render_annotation(parameter.annotation)}"
126
+ if parameter.default is not inspect.Parameter.empty:
127
+ rendered += f" = {parameter.default!r}"
128
+ elif parameter.default is not inspect.Parameter.empty:
129
+ rendered += f"={parameter.default!r}"
130
+ parts.append(rendered)
131
+ previous_kind = parameter.kind
132
+
133
+ if previous_kind is inspect.Parameter.POSITIONAL_ONLY:
134
+ parts.append("/")
135
+
136
+ text = "(" + ", ".join(parts) + ")"
137
+ if signature.return_annotation is not inspect.Signature.empty:
138
+ text += f" -> {render_annotation(signature.return_annotation)}"
139
+ return text
140
+
141
+
142
+ def markdown_signature(func):
143
+ """``render_signature`` escaped for a markdown table cell."""
144
+ return render_signature(func).replace("|", r"\|")
microdf/microdataframe.py CHANGED
@@ -41,7 +41,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
41
41
  set_weight_col.
42
42
 
43
43
  :param weights: Array of weights.
44
- :type weights: np.array
44
+ :type weights: np.ndarray
45
45
  """
46
46
  super().__init__(*args, **kwargs)
47
47
  # pandas normalizes mixed-dimensional concat inputs through this
@@ -429,7 +429,7 @@ class MicroDataFrame(WeightPropagationMixin, pd.DataFrame):
429
429
  :param weights: Array of weights.
430
430
  :param preserve_old: If True, keeps the old weights as a column when
431
431
  new weights are provided.
432
- :type weights: np.array
432
+ :type weights: np.ndarray
433
433
  """
434
434
  if preserve_old and self.weights_col is not None:
435
435
  self["old_" + self.weights_col] = self.weights
microdf/microseries.py CHANGED
@@ -181,13 +181,13 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
181
181
  # Keep inherited fallback paths working after overriding that handler.
182
182
  _HANDLED_TYPES = pd.Series._HANDLED_TYPES + (pd.Series, pd.DataFrame)
183
183
 
184
- def __init__(self, *args, weights: np.array = None, **kwargs):
184
+ def __init__(self, *args, weights: np.ndarray = None, **kwargs):
185
185
  """A Series-inheriting class for weighted microdata.
186
186
 
187
187
  Weights can be provided at initialisation, or using set_weights.
188
188
 
189
189
  :param weights: Array of weights.
190
- :type weights: np.array
190
+ :type weights: np.ndarray
191
191
  """
192
192
  super().__init__(*args, **kwargs)
193
193
  self.set_weights(weights)
@@ -347,14 +347,14 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
347
347
  return fn
348
348
 
349
349
  def set_weights(
350
- self, weights: np.array, preserve_old: Optional[bool] = False
350
+ self, weights: np.ndarray, preserve_old: Optional[bool] = False
351
351
  ) -> None:
352
352
  """Sets the weight values.
353
353
 
354
354
  :param weights: Array of weights.
355
355
  :param preserve_old: If True, keeps the old weights as a column when
356
356
  new weights are provided.
357
- :type weights: np.array.
357
+ :type weights: np.ndarray.
358
358
  """
359
359
  if weights is None:
360
360
  self.weights = weight_series(np.ones(len(self)), self.index)
@@ -622,7 +622,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
622
622
  x, y, weights, _ = pair
623
623
  return _weighted_correlation(x, y, weights)
624
624
 
625
- def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
625
+ def quantile(self, q: np.ndarray, skipna: bool = True) -> pd.Series:
626
626
  """Calculates weighted quantiles of the MicroSeries.
627
627
 
628
628
  Uses the inverse CDF method: the q-th quantile is the smallest
@@ -630,7 +630,7 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
630
630
  the default behavior of R's survey::svyquantile.
631
631
 
632
632
  :param q: Quantile(s) to calculate, must be in [0, 1].
633
- :type q: float or np.array
633
+ :type q: float or np.ndarray
634
634
  :param skipna: Exclude NaN values (default True). NaN sorts to the
635
635
  end of the array, so leaving NaN rows in would let their weight
636
636
  inflate the cumulative distribution and push the cutoff upward.
@@ -974,6 +974,11 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
974
974
 
975
975
  @vector_function
976
976
  def quintile_rank(self) -> "MicroSeries":
977
+ """Calculate weighted quintile ranks (1-5).
978
+
979
+ :returns: MicroSeries of quintile ranks.
980
+ :rtype: MicroSeries
981
+ """
977
982
  return MicroSeries(
978
983
  np.minimum(np.ceil(self.rank(pct=True) * 5), 5),
979
984
  weights=self.weights,
@@ -981,6 +986,11 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
981
986
 
982
987
  @vector_function
983
988
  def quartile_rank(self) -> "MicroSeries":
989
+ """Calculate weighted quartile ranks (1-4).
990
+
991
+ :returns: MicroSeries of quartile ranks.
992
+ :rtype: MicroSeries
993
+ """
984
994
  return MicroSeries(
985
995
  np.minimum(np.ceil(self.rank(pct=True) * 4), 4),
986
996
  weights=self.weights,
@@ -988,6 +998,11 @@ class MicroSeries(WeightPropagationMixin, pd.Series):
988
998
 
989
999
  @vector_function
990
1000
  def percentile_rank(self) -> "MicroSeries":
1001
+ """Calculate weighted percentile ranks (1-100).
1002
+
1003
+ :returns: MicroSeries of percentile ranks.
1004
+ :rtype: MicroSeries
1005
+ """
991
1006
  return MicroSeries(
992
1007
  np.minimum(np.ceil(self.rank(pct=True) * 100), 100),
993
1008
  weights=self.weights,
@@ -0,0 +1,160 @@
1
+ """The API reference must list every public method, in both directions.
2
+
3
+ A one-way check lets the page fall behind the code silently, which is how the
4
+ weighted estimators came to be missing from it.
5
+ """
6
+
7
+ import inspect
8
+ import re
9
+ from pathlib import Path
10
+
11
+ import pytest
12
+
13
+ import microdf as mdf
14
+ from microdf._docs import markdown_signature
15
+
16
+ DOCS = Path(__file__).resolve().parents[2] / "docs" / "api.md"
17
+
18
+ # docs/ is not shipped in the sdist or the wheel, so these cannot run against an
19
+ # installed copy of the package.
20
+ pytestmark = pytest.mark.skipif(
21
+ not DOCS.exists(), reason="docs/api.md is not present in the installed package"
22
+ )
23
+
24
+ # Public names that read as internals rather than API a user would call.
25
+ INTERNAL = {
26
+ "scalar_function",
27
+ "vector_function",
28
+ "override_df_functions",
29
+ "catch_series_relapse",
30
+ "get_args_as_micro_series",
31
+ }
32
+
33
+
34
+ def documented_names():
35
+ return set(re.findall(r"^\| `(\w+)` \|", DOCS.read_text(), re.M))
36
+
37
+
38
+ def public_methods(cls):
39
+ """Public names this class defines itself.
40
+
41
+ Anything inherited unchanged from pandas is pandas' to document; what
42
+ matters here is what microdf adds or overrides.
43
+ """
44
+ names = set()
45
+ for name, attr in vars(cls).items():
46
+ if name.startswith("_") or name in INTERNAL:
47
+ continue
48
+ if callable(attr) or isinstance(attr, property):
49
+ names.add(name)
50
+ return names
51
+
52
+
53
+ def test_every_documented_method_exists():
54
+ documented = documented_names()
55
+ real = (
56
+ public_methods(mdf.MicroSeries)
57
+ | public_methods(mdf.MicroDataFrame)
58
+ | {"replicate_variance", "replicate_standard_error"}
59
+ )
60
+ assert not (documented - real), f"documented but absent: {documented - real}"
61
+
62
+
63
+ def test_every_public_method_is_documented():
64
+ documented = documented_names()
65
+ for cls in (mdf.MicroSeries, mdf.MicroDataFrame):
66
+ missing = public_methods(cls) - documented
67
+ assert not missing, (
68
+ f"{cls.__name__} methods missing from docs/api.md: {sorted(missing)}"
69
+ )
70
+
71
+
72
+ def test_documented_weight_behaviour_holds():
73
+ """Pin the claims the page makes that a docstring does not enforce.
74
+
75
+ Every error found in review was in a description written by hand rather
76
+ than taken from a docstring, so the ones that remain are asserted here.
77
+ """
78
+ import numpy as np
79
+ import pandas as pd
80
+
81
+ frame = mdf.MicroDataFrame(
82
+ {"x": [1.0, 2.0, 3.0, 4.0], "y": [1.0, 4.0, 2.0, 8.0]},
83
+ weights=[1.0, 1.0, 1.0, 5.0],
84
+ )
85
+ replicated = pd.DataFrame(
86
+ {"x": [1.0, 2.0, 3.0] + [4.0] * 5, "y": [1.0, 4.0, 2.0] + [8.0] * 5}
87
+ )
88
+
89
+ # The page says these are frequency-weighted, as of #330.
90
+ assert frame.cov().loc["x", "y"] == pytest.approx(replicated.cov().loc["x", "y"])
91
+ assert frame.corr().loc["x", "y"] == pytest.approx(replicated.corr().loc["x", "y"])
92
+
93
+ # Each frame cell is the corresponding MicroSeries value.
94
+ assert frame.cov().loc["x", "y"] == pytest.approx(frame.x.cov(frame.y))
95
+ assert frame.x.cov(frame.y) == pytest.approx(replicated.cov().loc["x", "y"])
96
+
97
+ # The page says equals compares weights.
98
+ light = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1])
99
+ heavy = mdf.MicroSeries([1, 2, 3], weights=[9, 9, 9])
100
+ assert not light.equals(heavy)
101
+
102
+ # The page says cumsum drops the weights.
103
+ assert not hasattr(frame.x.cumsum(), "weights")
104
+
105
+ # The page says repeat repeats the weights alongside the values.
106
+ repeated = mdf.MicroSeries([1.0, 2.0], weights=[3.0, 4.0]).repeat(2)
107
+ assert list(np.asarray(repeated.weights)) == [3.0, 3.0, 4.0, 4.0]
108
+
109
+
110
+ def test_no_row_is_missing_its_description():
111
+ """Every documented method needs a description.
112
+
113
+ A row whose description cell is blank is the signature of a source change
114
+ that the page was never regenerated for, which happened three times in
115
+ review before this test existed.
116
+ """
117
+ blank = re.findall(
118
+ r"^\| `(\w+)` \| (?:`[^`]*`|\*attribute\*) \|\s*\|$", DOCS.read_text(), re.M
119
+ )
120
+ assert not blank, f"rows with no description: {blank}"
121
+
122
+
123
+ def test_signatures_match_the_live_ones():
124
+ """The rendered signature must be the one the code actually has.
125
+
126
+ The comparison is structural, not textual: both sides go through
127
+ `microdf._docs.markdown_signature`, which builds the text from parameter
128
+ names, kinds and defaults (stable across versions) and normalises the
129
+ annotations, so `pandas.core.series.Series` under pandas 2 and
130
+ `pandas.Series` under pandas 3 render alike, as do `Optional[int]` and the
131
+ `int | None` that Python 3.14 reprs it as. `docs/build_api.py` renders the
132
+ page with the same function, so the page and this test cannot disagree.
133
+
134
+ Rows are attributed to the class whose `## ` heading they fall under, since
135
+ several names exist on both.
136
+ """
137
+ current, mismatches = None, []
138
+ for line in DOCS.read_text().split("\n"):
139
+ if line.startswith("## MicroSeries"):
140
+ current = mdf.MicroSeries
141
+ elif line.startswith("## MicroDataFrame"):
142
+ current = mdf.MicroDataFrame
143
+ elif line.startswith("## "):
144
+ current = None
145
+ row = re.match(r"^\| `(\w+)` \| `([^`]*)` \|", line)
146
+ if not row or current is None:
147
+ continue
148
+ name, rendered = row.groups()
149
+ func = inspect.getattr_static(current, name, None)
150
+ if func is None or isinstance(func, property):
151
+ continue
152
+ try:
153
+ live = markdown_signature(func)
154
+ except (TypeError, ValueError):
155
+ continue
156
+ if live != rendered:
157
+ mismatches.append((current.__name__, name, rendered, live))
158
+ assert not mismatches, (
159
+ f"page is out of date with the code; rerun docs/build_api.py: {mismatches}"
160
+ )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microdf-python
3
- Version: 1.5.8
3
+ Version: 1.5.10
4
4
  Summary: Weighted pandas DataFrames and Series for survey microdata
5
5
  Author-email: Max Ghenis <max@policyengine.org>
6
6
  License: MIT
@@ -1,12 +1,14 @@
1
1
  microdf/__init__.py,sha256=G4m4UDiGePngG1i0DdGYhsTAXqyDELR6Mbq72OXOMG0,790
2
+ microdf/_docs.py,sha256=_mNNEUyutcOZxE1bArXnSqNNeNgXIACJQ0BzB6VTtqE,5371
2
3
  microdf/_weights.py,sha256=uBOcTZGmVJlzi27sSAALSTb4V9j56tXuaBfJB83x7j0,10303
3
- microdf/microdataframe.py,sha256=2HFXbS4gAmOQKX6eYcI5vwafCqv9aBzl3lDLODU9SXg,43436
4
- microdf/microseries.py,sha256=ZjSiUg88zOWAslUl2QpJWK0BcglbG-l7bU6_preOqNs,50039
4
+ microdf/microdataframe.py,sha256=LrNDlb6oYUk6uu3f8a7BzViAd5a_tJa0N_XU5js4gaM,43440
5
+ microdf/microseries.py,sha256=EUOQyAZS3QMqfgI-SMu7I9V45R0TGXv7sv3tOm5oFPY,50483
5
6
  microdf/replication.py,sha256=3iZ6xG24ucKofVwmXxyd7ME6rZP6iJVnN9JpWcPxg9s,7697
6
7
  microdf/tests/conftest.py,sha256=u-EMyX1-u_nM-YO0RJYCzYHQDXxUI2WQE6GkyJlErqg,150
7
8
  microdf/tests/test_aggregation_errors.py,sha256=9jJDiEyxMb2z1Zmj-o8AHDNp8LOputbEkAMefifnWaE,2013
8
9
  microdf/tests/test_binary_weight_alignment.py,sha256=d3-e7Kb6G8J22aVFTi3sOGUyDODKT1p0-XsEvMZBDvo,8115
9
10
  microdf/tests/test_dataframe_weight_storage.py,sha256=m77cbDn521ehIJWbgirjhL5Q5byBdm-UZhFkGx7mp6A,3667
11
+ microdf/tests/test_docs_api_coverage.py,sha256=-YPtY4xxMzUQ7S2Sib5iF72BpEvz1llBS2IloMh-oxI,5771
10
12
  microdf/tests/test_microseries_dataframe.py,sha256=vL0fg_NydU6a5myVVtXyHMr8eQOB_yyAkZYwLmoEnOA,30075
11
13
  microdf/tests/test_nullify_weights_index.py,sha256=kZgzMaZEa_PXbsor2S4E-6VRid3C3rcC9ufk0qa7mgY,341
12
14
  microdf/tests/test_pandas3_compatibility.py,sha256=A34Ni_WQ303sSNv-sqv5CGAQp54zj-ZSGAPEBHZslNI,8573
@@ -18,8 +20,8 @@ microdf/tests/test_ufunc_weight_dispatch.py,sha256=abQ5I4wFRM3poxO76nbR_1dQymHka
18
20
  microdf/tests/test_version_metadata.py,sha256=M1EabzHLKZZw3Djd6Zu2UuMQtDLV6rZ1zDrOU7W_jf0,227
19
21
  microdf/tests/test_weight_propagation.py,sha256=YCHZRdeWyWEwnHdpoY1jRBgFLQcBE1nLs61LrcT1wao,20359
20
22
  microdf/tests/test_weighted_cov_corr.py,sha256=Ag4YJDy0eH3r_DhOaiX44kEZ0GHuQOoQtBVAOAkBXmw,18291
21
- microdf_python-1.5.8.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
22
- microdf_python-1.5.8.dist-info/METADATA,sha256=cG5neCZ-WXOvFPEZNt1saWP2WPD4URKohDdmwn_1t-M,3535
23
- microdf_python-1.5.8.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
24
- microdf_python-1.5.8.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
25
- microdf_python-1.5.8.dist-info/RECORD,,
23
+ microdf_python-1.5.10.dist-info/licenses/LICENSE,sha256=uPs-ASYnzlldpf2z8jeRgQFeEH3FLhSuX0rw0OKWoDU,1067
24
+ microdf_python-1.5.10.dist-info/METADATA,sha256=p3-fM3i4b6dR59QzsUdNn7uxranyhhqLZmjuIQpielM,3536
25
+ microdf_python-1.5.10.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
26
+ microdf_python-1.5.10.dist-info/top_level.txt,sha256=T2WFPTygQQMdS3GF8YpZ12DKfMGrspbZ3r7z-e3KfiM,8
27
+ microdf_python-1.5.10.dist-info/RECORD,,