robustkit 0.0.1__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- robustkit-0.5.0/PKG-INFO +446 -0
- robustkit-0.5.0/README.md +406 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/pyproject.toml +1 -1
- robustkit-0.5.0/robustkit/__init__.py +103 -0
- robustkit-0.5.0/robustkit/benchmark/global_model.py +74 -0
- robustkit-0.5.0/robustkit/benchmark/robustness_map.py +160 -0
- robustkit-0.5.0/robustkit/common/quadrants.py +57 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/diagnostics.py +2 -2
- robustkit-0.5.0/robustkit/core/goodness_of_fit.py +77 -0
- robustkit-0.5.0/robustkit/core/trend.py +136 -0
- robustkit-0.5.0/robustkit/information/__init__.py +0 -0
- robustkit-0.5.0/robustkit/information/communication.py +143 -0
- robustkit-0.5.0/robustkit/information/conditional_mi.py +50 -0
- robustkit-0.5.0/robustkit/information/mutual_info.py +172 -0
- robustkit-0.5.0/robustkit/information/pairs.py +88 -0
- robustkit-0.5.0/robustkit/information/quadrants.py +56 -0
- robustkit-0.5.0/robustkit/information/utils.py +23 -0
- robustkit-0.5.0/robustkit/quantiles/__init__.py +0 -0
- robustkit-0.5.0/robustkit/quantiles/io.py +104 -0
- robustkit-0.5.0/robustkit/quantiles/reconstruct.py +272 -0
- robustkit-0.5.0/robustkit/quantiles/trend.py +85 -0
- robustkit-0.5.0/robustkit/report/__init__.py +0 -0
- robustkit-0.5.0/robustkit/report/dispersion.py +68 -0
- robustkit-0.5.0/robustkit/report/visualize_analyst.py +52 -0
- robustkit-0.5.0/robustkit/report/visualize_publisher.py +57 -0
- robustkit-0.5.0/robustkit/segmentation/__init__.py +0 -0
- robustkit-0.5.0/robustkit.egg-info/PKG-INFO +446 -0
- robustkit-0.5.0/robustkit.egg-info/SOURCES.txt +53 -0
- robustkit-0.5.0/tests/test_benchmark.py +118 -0
- robustkit-0.5.0/tests/test_common_quadrants.py +30 -0
- robustkit-0.5.0/tests/test_information_pairs.py +122 -0
- robustkit-0.5.0/tests/test_quantiles_io_trend.py +126 -0
- robustkit-0.5.0/tests/test_quantiles_reconstruct.py +90 -0
- robustkit-0.5.0/tests/test_quantiles_reconstruct_mean_only.py +78 -0
- robustkit-0.5.0/tests/test_report.py +93 -0
- robustkit-0.5.0/tests/test_trend_extras.py +92 -0
- robustkit-0.0.1/PKG-INFO +0 -169
- robustkit-0.0.1/README.md +0 -129
- robustkit-0.0.1/robustkit/__init__.py +0 -54
- robustkit-0.0.1/robustkit/core/trend.py +0 -86
- robustkit-0.0.1/robustkit/information/mutual_info.py +0 -87
- robustkit-0.0.1/robustkit/information/quadrants.py +0 -67
- robustkit-0.0.1/robustkit.egg-info/PKG-INFO +0 -169
- robustkit-0.0.1/robustkit.egg-info/SOURCES.txt +0 -27
- {robustkit-0.0.1 → robustkit-0.5.0}/LICENSE +0 -0
- {robustkit-0.0.1/robustkit/core → robustkit-0.5.0/robustkit/benchmark}/__init__.py +0 -0
- {robustkit-0.0.1/robustkit/information → robustkit-0.5.0/robustkit/common}/__init__.py +0 -0
- {robustkit-0.0.1/robustkit/segmentation → robustkit-0.5.0/robustkit/core}/__init__.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/consistency.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/stability.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/uncertainty.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/information/entropy.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/information/profile.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/information/visualization.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/segmentation/apply.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/segmentation/hierarchy.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit.egg-info/dependency_links.txt +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit.egg-info/requires.txt +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/robustkit.egg-info/top_level.txt +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/setup.cfg +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/tests/test_core.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/tests/test_information.py +0 -0
- {robustkit-0.0.1 → robustkit-0.5.0}/tests/test_segmentation.py +0 -0
robustkit-0.5.0/PKG-INFO
ADDED
|
@@ -0,0 +1,446 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: robustkit
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
|
|
5
|
+
Author: Mikael Lundqvist
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2026 Mikael Lundqvist
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Requires-Python: >=3.10
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Requires-Dist: numpy>=1.24
|
|
32
|
+
Requires-Dist: pandas>=2.0
|
|
33
|
+
Requires-Dist: scikit-learn>=1.3
|
|
34
|
+
Requires-Dist: statsmodels>=0.14
|
|
35
|
+
Requires-Dist: scipy>=1.10
|
|
36
|
+
Requires-Dist: matplotlib>=3.7
|
|
37
|
+
Provides-Extra: dev
|
|
38
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# robustkit
|
|
42
|
+
|
|
43
|
+
> ⚠️ **Under active development.** This is an early placeholder release
|
|
44
|
+
> to claim the package name on PyPI. The API is incomplete and may
|
|
45
|
+
> change without notice. Not yet recommended for production use.
|
|
46
|
+
|
|
47
|
+
Practical tools for robust analysis of a single continuous relationship:
|
|
48
|
+
y as a function of one continuous x.
|
|
49
|
+
|
|
50
|
+
The guiding idea: **a conclusion that survives multiple fitting methods
|
|
51
|
+
is more trustworthy than one that only holds under a single model.**
|
|
52
|
+
`robustkit` makes it easy to compare Huber, Tukey biweight, and OLS
|
|
53
|
+
fits side by side, identify and quantify the influence of individual
|
|
54
|
+
observations, and get honest, bias-corrected uncertainty estimates.
|
|
55
|
+
|
|
56
|
+
## Status
|
|
57
|
+
|
|
58
|
+
`robustkit.core` (trend fitting, stability, diagnostics, uncertainty,
|
|
59
|
+
consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
60
|
+
per-segment analysis), `robustkit.information` (mutual-information
|
|
61
|
+
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
62
|
+
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
63
|
+
Robustness Map), `robustkit.report` (analyst vs. publisher views,
|
|
64
|
+
dispersion measures), and `robustkit.quantiles` (generic JSON-stat
|
|
65
|
+
loading, published-quantile-trend visualization, and lognormal-
|
|
66
|
+
calibrated reconstruction of individual-level data from aggregated
|
|
67
|
+
summaries) are stable and tested.
|
|
68
|
+
|
|
69
|
+
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
70
|
+
continuous features. `mutual_information` is estimated on the
|
|
71
|
+
full-resolution continuous values, while `entropy_bits` is computed on
|
|
72
|
+
a binned version of the same feature (since `entropy()` expects
|
|
73
|
+
categorical input). Binning discards information, so `entropy_bits` is
|
|
74
|
+
a lower bound on the feature's true entropy -- an efficiency above 1.0
|
|
75
|
+
signals that the feature carries more usable information than a coarse
|
|
76
|
+
categorical summary of it would capture. This is expected behavior,
|
|
77
|
+
not a bug.
|
|
78
|
+
|
|
79
|
+
## Installation
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
git clone https://github.com/<your-username>/robustkit.git
|
|
83
|
+
cd robustkit
|
|
84
|
+
pip install -e ".[dev]"
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Quickstart
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
import numpy as np
|
|
91
|
+
from robustkit import (
|
|
92
|
+
fit_huber_trend, fit_tukey_trend, predict_trend,
|
|
93
|
+
model_stability_pct, cooks_diagnostic, cook_impact,
|
|
94
|
+
bootstrap_band, bca_bootstrap_ci,
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
# x: a single continuous predictor, y: a single continuous outcome
|
|
98
|
+
x = np.random.default_rng(0).uniform(20, 60, 200)
|
|
99
|
+
y = 1000 + 50 * x - 0.4 * x**2 + np.random.default_rng(1).normal(0, 500, 200)
|
|
100
|
+
|
|
101
|
+
fit = fit_huber_trend(x, y, degree=2)
|
|
102
|
+
y_pred = predict_trend(fit, x_new=[30, 40, 50])
|
|
103
|
+
|
|
104
|
+
stability = model_stability_pct(x, y)
|
|
105
|
+
print("Median % spread between Huber/Tukey/OLS:", stability["median_pct_diff"])
|
|
106
|
+
|
|
107
|
+
diag = cooks_diagnostic(x, y)
|
|
108
|
+
impact = cook_impact(x, y, diag["flagged_indices"])
|
|
109
|
+
print("Median % change in curve if flagged points removed:", impact["median_pct_change"])
|
|
110
|
+
|
|
111
|
+
band = bootstrap_band(x, y)
|
|
112
|
+
ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
|
|
116
|
+
|
|
117
|
+
## Trend growth rate and goodness of fit
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
|
|
121
|
+
|
|
122
|
+
fit = fit_huber_trend(df["age"], df["salary"])
|
|
123
|
+
|
|
124
|
+
# Rate of change of the trend itself (e.g. "salary growth per year of
|
|
125
|
+
# age"), not just its level
|
|
126
|
+
rates = trend_derivative(fit, x=[30, 40, 50])
|
|
127
|
+
|
|
128
|
+
# How well does this fit actually explain the variation in y?
|
|
129
|
+
goodness_of_fit(df["age"], df["salary"], degree=2)
|
|
130
|
+
|
|
131
|
+
# Don't assume a quadratic trend is always the right choice -- check
|
|
132
|
+
# empirically whether a higher degree captures meaningfully more
|
|
133
|
+
compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
**Note:** x is standardized internally before building polynomial
|
|
137
|
+
features (both here and throughout `robustkit.core`), since raw
|
|
138
|
+
polynomial features become numerically unstable at higher degrees for
|
|
139
|
+
realistic x scales (e.g. age^5 vastly outscales age^1). This is
|
|
140
|
+
transparent to callers -- `predict_trend` and `trend_derivative` still
|
|
141
|
+
take and return values in the original x scale.
|
|
142
|
+
|
|
143
|
+
## Segmentation
|
|
144
|
+
|
|
145
|
+
Run any `robustkit.core` analysis independently across subgroups of a
|
|
146
|
+
larger dataset, with automatic fallback to coarser groupings when a
|
|
147
|
+
finer one is too small to analyze reliably:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
from robustkit import hierarchical_segment, apply_by_segment, model_stability_pct
|
|
151
|
+
|
|
152
|
+
hierarchy = [["department", "level", "status"], ["level", "status"], ["status"]]
|
|
153
|
+
segmented = hierarchical_segment(df, hierarchy, min_size=20)
|
|
154
|
+
|
|
155
|
+
report = apply_by_segment(
|
|
156
|
+
segmented, segment_col="segment_id", x_col="age", y_col="value",
|
|
157
|
+
analysis_fn=model_stability_pct,
|
|
158
|
+
)
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
`apply_by_segment` works with any function shaped like
|
|
162
|
+
`analysis_fn(x, y, **kwargs) -> dict` -- built-in ones
|
|
163
|
+
(`model_stability_pct`, `cook_impact`, `bca_bootstrap_ci`, ...) or your
|
|
164
|
+
own. Only scalar values in the returned dict end up in the report
|
|
165
|
+
table; segments below `min_points` are skipped rather than causing an
|
|
166
|
+
error.
|
|
167
|
+
|
|
168
|
+
## Feature ranking (information)
|
|
169
|
+
|
|
170
|
+
Rank features by mutual information with a target, normalized by each
|
|
171
|
+
feature's own entropy, and classify them into four quadrants:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
from robustkit import rank_features, quadrant_report, plot_feature_space
|
|
175
|
+
|
|
176
|
+
ranking = rank_features(df, target="value")
|
|
177
|
+
report = quadrant_report(df, target="value") # adds a `quadrant` column
|
|
178
|
+
plot_feature_space(df, target="value") # same quadrants, visualized
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
`quadrant_report` and `plot_feature_space` always agree on quadrant
|
|
182
|
+
assignment -- both route through the same thresholding logic.
|
|
183
|
+
|
|
184
|
+
**Caveat:** default thresholds are the *median* mutual information /
|
|
185
|
+
efficiency across the ranked features. With only a handful of
|
|
186
|
+
features, this can put a genuinely weak feature in the same "high"
|
|
187
|
+
half as a strong one, since roughly half of any list sits above its
|
|
188
|
+
own median regardless of how large the actual gap is. Median
|
|
189
|
+
thresholding becomes meaningful with a reasonably large feature set;
|
|
190
|
+
for a handful of candidates, read the raw `mutual_information` /
|
|
191
|
+
`information_efficiency` values directly rather than relying on the
|
|
192
|
+
quadrant label alone.
|
|
193
|
+
|
|
194
|
+
See `examples/information_tutorial.py` for a complete walkthrough.
|
|
195
|
+
|
|
196
|
+
## Benchmarking against a global trend
|
|
197
|
+
|
|
198
|
+
Compare each segment's observed outcome against what a single global
|
|
199
|
+
robust trend predicts, with bootstrap uncertainty on the difference --
|
|
200
|
+
answers "which groups deviate from the overall trend, and by how
|
|
201
|
+
much?" rather than "how does the trend look overall?":
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
from robustkit import segment_position_report
|
|
205
|
+
|
|
206
|
+
report = segment_position_report(
|
|
207
|
+
df, segment_col="department", x_col="age", y_col="salary",
|
|
208
|
+
)
|
|
209
|
+
# segment n observed_median expected_median difference ci_lower ci_upper
|
|
210
|
+
# Finance 176 48339.70 47799.82 539.88 202.15 1031.01
|
|
211
|
+
# HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36
|
|
212
|
+
# IT 250 47226.50 47126.91 99.59 -117.56 510.81
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
A segment's confidence interval crossing zero means no clear deviation
|
|
216
|
+
from the benchmark; HR and Finance above don't cross zero, IT does.
|
|
217
|
+
|
|
218
|
+
## Robustness Map
|
|
219
|
+
|
|
220
|
+
Classify features by how much a conclusion about their relationship
|
|
221
|
+
with the target depends on (a) fitting method choice and (b) specific
|
|
222
|
+
influential observations -- two genuinely different failure modes that
|
|
223
|
+
a single diagnostic can miss:
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
from robustkit import feature_robustness_report, plot_feature_robustness
|
|
227
|
+
|
|
228
|
+
report = feature_robustness_report(df, target="value")
|
|
229
|
+
# feature stability_pct cook_impact_pct quadrant
|
|
230
|
+
# CRIM 8.9 17.1 fragile
|
|
231
|
+
# AGE 16.8 15.5 fragile
|
|
232
|
+
# RM 4.9 0.1 robust
|
|
233
|
+
# TAX 22.2 1.4 structural_sensitivity
|
|
234
|
+
|
|
235
|
+
plot_feature_robustness(report=report)
|
|
236
|
+
```
|
|
237
|
+
|
|
238
|
+
Four quadrants: **robust** (low spread, low impact), **structural
|
|
239
|
+
sensitivity** (sensitive to fitting method, not to specific points),
|
|
240
|
+
**data sensitive** (a few points drive the conclusion, method choice
|
|
241
|
+
barely matters), **fragile** (both -- least trustworthy).
|
|
242
|
+
|
|
243
|
+
`quadrant_report`/`plot_feature_space` (information) and
|
|
244
|
+
`feature_robustness_report`/`plot_feature_robustness` (benchmark) both
|
|
245
|
+
route through the same shared classifier, `robustkit.classify_quadrants`
|
|
246
|
+
-- any future quadrant-based analysis in this package will too.
|
|
247
|
+
|
|
248
|
+
## Analyst view vs. publisher view
|
|
249
|
+
|
|
250
|
+
Two visualizations that look superficially similar but answer
|
|
251
|
+
genuinely different questions:
|
|
252
|
+
|
|
253
|
+
```python
|
|
254
|
+
from robustkit import plot_analyst_view, plot_publisher_view, dispersion_ratio, iqr
|
|
255
|
+
|
|
256
|
+
# "How confident are we in the trend estimate?" -- a bootstrap
|
|
257
|
+
# confidence band that SHRINKS as sample size grows.
|
|
258
|
+
plot_analyst_view(df["age"], df["salary"])
|
|
259
|
+
|
|
260
|
+
# "How spread out are actual values in the population?" -- a median +
|
|
261
|
+
# IQR band that does NOT shrink with more data, since it reflects
|
|
262
|
+
# real dispersion, not estimation uncertainty. show_points defaults to
|
|
263
|
+
# False, since this view is meant for publishing potentially sensitive
|
|
264
|
+
# data (e.g. individual salaries) without exposing raw points.
|
|
265
|
+
plot_publisher_view(df["age"], df["salary"])
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
This distinction matters in practice: with 20x more data (same
|
|
269
|
+
underlying distribution), the analyst view's confidence band roughly
|
|
270
|
+
halves in width, while the publisher view's IQR band stays essentially
|
|
271
|
+
unchanged -- confirmed by the package's own test suite.
|
|
272
|
+
|
|
273
|
+
`dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
|
|
274
|
+
standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
275
|
+
computes both across bins of a continuous x, e.g. to check whether
|
|
276
|
+
dispersion (inequality) grows with age.
|
|
277
|
+
|
|
278
|
+
## Loading published quantile tables (SCB / JSON-stat)
|
|
279
|
+
|
|
280
|
+
Some statistics agencies (e.g. Statistics Sweden, SCB) publish
|
|
281
|
+
quantiles (Q1/median/Q3) directly, with no individual-level data
|
|
282
|
+
available at all. `robustkit.quantiles` loads these tables generically
|
|
283
|
+
via JSON-stat, a standardized dimensional-data format used by SCB and
|
|
284
|
+
other national statistics agencies -- avoiding the fragility of
|
|
285
|
+
parsing metadata out of column-name strings in a wide CSV export.
|
|
286
|
+
|
|
287
|
+
```python
|
|
288
|
+
from robustkit import load_scb_json_stat, plot_quantile_trend, quantile_trend_dispersion
|
|
289
|
+
|
|
290
|
+
df = load_scb_json_stat("some_scb_table.json")
|
|
291
|
+
|
|
292
|
+
# A real SCB quirk this loader does NOT try to guess automatically:
|
|
293
|
+
# category labels can change meaning over time (e.g. Sweden's oldest
|
|
294
|
+
# working-age bracket was labeled "65-66 år" through 2022 and
|
|
295
|
+
# "65-68 år" from 2023, following a pension-age reform). Merge such
|
|
296
|
+
# cases explicitly:
|
|
297
|
+
df = load_scb_json_stat(
|
|
298
|
+
"some_scb_table.json",
|
|
299
|
+
rename_categories={"ålder": {"65–68 år": "65–66 år"}},
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
# Once reshaped to a wide table with q1/median/q3 columns:
|
|
303
|
+
plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
304
|
+
quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
305
|
+
```
|
|
306
|
+
|
|
307
|
+
This is the "quantiles are already given" case. A complementary case
|
|
308
|
+
-- reconstructing approximate individual-level data from aggregated
|
|
309
|
+
group means, for when only summary statistics (not quantiles) are
|
|
310
|
+
available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
|
|
311
|
+
|
|
312
|
+
See `examples/quantiles_tutorial.py` for a complete walkthrough.
|
|
313
|
+
|
|
314
|
+
## Reconstructing individual-level data from aggregated summaries
|
|
315
|
+
|
|
316
|
+
For the complementary case -- only aggregated group summaries (n,
|
|
317
|
+
Q1, median, Q3) are available, not the quantile trend itself as the
|
|
318
|
+
final answer, and you want to run `robustkit.core` analyses as if
|
|
319
|
+
individual data existed:
|
|
320
|
+
|
|
321
|
+
```python
|
|
322
|
+
from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
|
|
323
|
+
|
|
324
|
+
# One row per group (e.g. year), with n/q1/median/q3 columns
|
|
325
|
+
synthetic = expand_aggregated_table(
|
|
326
|
+
summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
|
|
327
|
+
group_cols=["year"], value_name="salary",
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
# Now usable exactly like real individual-level data:
|
|
331
|
+
fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
Method: a lognormal distribution is calibrated (via the IQR) to match
|
|
335
|
+
each group's reported Q1/median/Q3, then `n` synthetic values are
|
|
336
|
+
drawn from it. Validated end-to-end against real published SCB salary
|
|
337
|
+
data: a Huber trend fitted on reconstructed pseudo-individual data
|
|
338
|
+
tracked the true published median trend within 2% across 12 years.
|
|
339
|
+
|
|
340
|
+
**Note on what this recovers:** because a Huber (or Tukey) fit on
|
|
341
|
+
right-skewed reconstructed data tracks something close to the
|
|
342
|
+
*median* trend it was calibrated against -- not the arithmetic mean --
|
|
343
|
+
this is consistent with, not a limitation of, the reconstruction
|
|
344
|
+
method. To target the mean instead, fit on `log(value)` and
|
|
345
|
+
exponentiate predictions back, which approximates the geometric mean.
|
|
346
|
+
|
|
347
|
+
Always check `check_reconstruction_quality()` before trusting a
|
|
348
|
+
reconstruction: real Q1/median/Q3 triples aren't always perfectly
|
|
349
|
+
consistent with a pure lognormal shape.
|
|
350
|
+
|
|
351
|
+
**Warning -- unbounded tail at large n:** a lognormal has no natural
|
|
352
|
+
upper limit, and its expected maximum grows with n. Reconstructing at
|
|
353
|
+
the TRUE group size from a national table (SCB salary tables can
|
|
354
|
+
report n in the hundreds of thousands to millions) can produce
|
|
355
|
+
implausibly extreme tail values -- real salaries have practical
|
|
356
|
+
ceilings a pure lognormal doesn't know about. This package's own
|
|
357
|
+
examples and tests deliberately scale n down to a few thousand for
|
|
358
|
+
demonstration; calibration quality (matching Q1/median/Q3) doesn't
|
|
359
|
+
depend on reproducing the true population size, but tail plausibility
|
|
360
|
+
does. No clipping is applied automatically.
|
|
361
|
+
|
|
362
|
+
### When only a mean is available (no quantiles at all)
|
|
363
|
+
|
|
364
|
+
Some tables (e.g. SCB's age-breakdown salary tables) report only a
|
|
365
|
+
mean per group, with no spread information. Two deliberately separate
|
|
366
|
+
methods are provided, each making a different explicit assumption --
|
|
367
|
+
compare them rather than silently picking one:
|
|
368
|
+
|
|
369
|
+
```python
|
|
370
|
+
from robustkit import (
|
|
371
|
+
expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
|
|
372
|
+
compare_reconstruction_methods,
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
# Method 1: repeat the mean n times -- zero within-group spread.
|
|
376
|
+
# Recovers between-group regression coefficients reasonably well
|
|
377
|
+
# (validated in the original technique this is based on) but
|
|
378
|
+
# understates individual-level variation.
|
|
379
|
+
flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
|
|
380
|
+
|
|
381
|
+
# Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
|
|
382
|
+
# report quantiles, and use it to imply an approximate spread around
|
|
383
|
+
# the mean. Stacks two assumptions (mean-as-median, and that the
|
|
384
|
+
# borrowed ratio transfers to this population) -- illustrative, not a
|
|
385
|
+
# substitute for genuine quantile data for this specific table.
|
|
386
|
+
borrowed = expand_aggregated_table_borrowed_dispersion(
|
|
387
|
+
df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
|
|
388
|
+
)
|
|
389
|
+
|
|
390
|
+
# Compare both for a single group directly:
|
|
391
|
+
compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
|
|
392
|
+
```
|
|
393
|
+
|
|
394
|
+
Both methods are documented with their specific assumptions rather
|
|
395
|
+
than presented as equally valid defaults -- being explicit about which
|
|
396
|
+
assumption was made lets the analyst judge how much a conclusion
|
|
397
|
+
depends on it, rather than presenting an assumption as a measurement.
|
|
398
|
+
|
|
399
|
+
## Feature pairing (information)
|
|
400
|
+
|
|
401
|
+
Beyond ranking single features, evaluate *pairs* of features together:
|
|
402
|
+
how redundant are they with each other, and does knowing one reveal
|
|
403
|
+
additional predictive value in the other (synergy, e.g. an interaction
|
|
404
|
+
effect)?
|
|
405
|
+
|
|
406
|
+
```python
|
|
407
|
+
from robustkit import (
|
|
408
|
+
conditional_mutual_information, communication_score,
|
|
409
|
+
rank_by_communication, pair_redundancy, pair_synergy,
|
|
410
|
+
rank_communicative_pairs,
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
# How communicable is a single feature -- not just predictive, but
|
|
414
|
+
# suitable for a clear chart/table (adequate group sizes, homogeneous
|
|
415
|
+
# groups, few enough categories to show at once)?
|
|
416
|
+
comm_ranking = rank_by_communication(df, target="value")
|
|
417
|
+
|
|
418
|
+
# How much does region's relevance to the target change once
|
|
419
|
+
# department is already known?
|
|
420
|
+
synergy = pair_synergy(df, feature_1="department", feature_2="region", target="value")
|
|
421
|
+
|
|
422
|
+
# Rank every candidate pair by combined relevance, penalizing
|
|
423
|
+
# redundant pairs and rewarding genuine synergy
|
|
424
|
+
pairs = rank_communicative_pairs(df, target="value")
|
|
425
|
+
```
|
|
426
|
+
|
|
427
|
+
All mutual-information-based quantities in this module (`rank_features`,
|
|
428
|
+
`conditional_mutual_information`, `pair_redundancy`, `pair_synergy`,
|
|
429
|
+
`communication_score`) are expressed in **bits**, consistent with
|
|
430
|
+
`entropy()` -- internally, scikit-learn's MI estimators return nats
|
|
431
|
+
and are converted before being used anywhere in this package.
|
|
432
|
+
|
|
433
|
+
## Design principles
|
|
434
|
+
|
|
435
|
+
- **One continuous x, one continuous y** at the core. This keeps every
|
|
436
|
+
function's output visually and numerically interpretable (a curve
|
|
437
|
+
you can plot, a band you can read).
|
|
438
|
+
- **Diagnosis and action are separate steps.** `cooks_diagnostic`
|
|
439
|
+
flags candidates; `cook_impact` tells you whether removing them
|
|
440
|
+
actually changes anything.
|
|
441
|
+
- **OLS is a reference point, not the enemy.** Comparing robust fits
|
|
442
|
+
against OLS is how you know whether robustness mattered at all.
|
|
443
|
+
|
|
444
|
+
## License
|
|
445
|
+
|
|
446
|
+
MIT -- see [LICENSE](LICENSE).
|