robustkit 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- robustkit-0.6.0/PKG-INFO +784 -0
- robustkit-0.6.0/README.md +742 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/pyproject.toml +2 -1
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/__init__.py +50 -4
- robustkit-0.6.0/robustkit/benchmark/global_model.py +162 -0
- robustkit-0.6.0/robustkit/benchmark/reporting.py +317 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/benchmark/robustness_map.py +19 -1
- robustkit-0.6.0/robustkit/core/consistency.py +90 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/core/diagnostics.py +2 -2
- robustkit-0.6.0/robustkit/core/goodness_of_fit.py +77 -0
- robustkit-0.6.0/robustkit/core/trend.py +136 -0
- robustkit-0.6.0/robustkit/core/uncertainty.py +171 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/mutual_info.py +13 -1
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/visualization.py +1 -1
- robustkit-0.6.0/robustkit/quantiles/io.py +104 -0
- robustkit-0.6.0/robustkit/quantiles/reconstruct.py +272 -0
- robustkit-0.6.0/robustkit/quantiles/trend.py +86 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/report/visualize_analyst.py +7 -2
- robustkit-0.6.0/robustkit/report/visualize_huber_iqr.py +313 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/report/visualize_publisher.py +3 -2
- robustkit-0.6.0/robustkit/segment_awareness/__init__.py +17 -0
- robustkit-0.6.0/robustkit/segment_awareness/reports.py +470 -0
- robustkit-0.6.0/robustkit/segmentation/__init__.py +0 -0
- robustkit-0.6.0/robustkit.egg-info/PKG-INFO +784 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/SOURCES.txt +25 -1
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/requires.txt +3 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_benchmark.py +23 -4
- robustkit-0.6.0/tests/test_benchmark_gaps.py +159 -0
- robustkit-0.6.0/tests/test_benchmark_reporting.py +195 -0
- robustkit-0.6.0/tests/test_bugfixes_faseA.py +102 -0
- robustkit-0.6.0/tests/test_export_no_deps.py +119 -0
- robustkit-0.6.0/tests/test_export_outlier_pdf.py +116 -0
- robustkit-0.6.0/tests/test_huber_iqr.py +197 -0
- robustkit-0.6.0/tests/test_mad_outlier_report.py +117 -0
- robustkit-0.6.0/tests/test_n_boot_auto_propagation.py +86 -0
- robustkit-0.6.0/tests/test_quantiles_io_trend.py +134 -0
- robustkit-0.6.0/tests/test_quantiles_reconstruct.py +90 -0
- robustkit-0.6.0/tests/test_quantiles_reconstruct_mean_only.py +78 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_report.py +20 -0
- robustkit-0.6.0/tests/test_segment_awareness.py +124 -0
- robustkit-0.6.0/tests/test_segment_consistency.py +72 -0
- robustkit-0.6.0/tests/test_segment_drilldown.py +84 -0
- robustkit-0.6.0/tests/test_trend_extras.py +92 -0
- robustkit-0.4.0/PKG-INFO +0 -296
- robustkit-0.4.0/README.md +0 -256
- robustkit-0.4.0/robustkit/benchmark/global_model.py +0 -74
- robustkit-0.4.0/robustkit/core/consistency.py +0 -33
- robustkit-0.4.0/robustkit/core/trend.py +0 -86
- robustkit-0.4.0/robustkit/core/uncertainty.py +0 -83
- robustkit-0.4.0/robustkit.egg-info/PKG-INFO +0 -296
- {robustkit-0.4.0 → robustkit-0.6.0}/LICENSE +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/benchmark/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/common/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/common/quadrants.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/core/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/core/stability.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/communication.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/conditional_mi.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/entropy.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/pairs.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/profile.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/quadrants.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/utils.py +0 -0
- {robustkit-0.4.0/robustkit/report → robustkit-0.6.0/robustkit/quantiles}/__init__.py +0 -0
- {robustkit-0.4.0/robustkit/segmentation → robustkit-0.6.0/robustkit/report}/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/report/dispersion.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/segmentation/apply.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/segmentation/hierarchy.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/dependency_links.txt +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/top_level.txt +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/setup.cfg +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_common_quadrants.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_core.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_information.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_information_pairs.py +0 -0
- {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_segmentation.py +0 -0
robustkit-0.6.0/PKG-INFO
ADDED
|
@@ -0,0 +1,784 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: robustkit
|
|
3
|
+
Version: 0.6.0
|
|
4
|
+
Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
|
|
5
|
+
Author: Mikael Lundqvist
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2026 Mikael Lundqvist
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Requires-Python: >=3.10
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Requires-Dist: numpy>=1.24
|
|
32
|
+
Requires-Dist: pandas>=2.0
|
|
33
|
+
Requires-Dist: scikit-learn>=1.3
|
|
34
|
+
Requires-Dist: statsmodels>=0.14
|
|
35
|
+
Requires-Dist: scipy>=1.10
|
|
36
|
+
Requires-Dist: matplotlib>=3.7
|
|
37
|
+
Provides-Extra: dev
|
|
38
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
39
|
+
Provides-Extra: excel
|
|
40
|
+
Requires-Dist: openpyxl>=3.1; extra == "excel"
|
|
41
|
+
Dynamic: license-file
|
|
42
|
+
|
|
43
|
+
# robustkit
|
|
44
|
+
|
|
45
|
+
> ⚠️ **Under active development.** This is an early placeholder release
|
|
46
|
+
> to claim the package name on PyPI. The API is incomplete and may
|
|
47
|
+
> change without notice. Not yet recommended for production use.
|
|
48
|
+
|
|
49
|
+
Practical tools for robust analysis of a single continuous relationship:
|
|
50
|
+
y as a function of one continuous x.
|
|
51
|
+
|
|
52
|
+
The guiding idea: **a conclusion that survives multiple fitting methods
|
|
53
|
+
is more trustworthy than one that only holds under a single model.**
|
|
54
|
+
`robustkit` makes it easy to compare Huber, Tukey biweight, and OLS
|
|
55
|
+
fits side by side, identify and quantify the influence of individual
|
|
56
|
+
observations, and get honest, bias-corrected uncertainty estimates.
|
|
57
|
+
|
|
58
|
+
## Status
|
|
59
|
+
|
|
60
|
+
`robustkit.core` (trend fitting, stability, diagnostics, uncertainty,
|
|
61
|
+
consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
62
|
+
per-segment analysis), `robustkit.information` (mutual-information
|
|
63
|
+
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
64
|
+
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
65
|
+
model-agnostic custom benchmarks, residual/deviation reporting, Excel
|
|
66
|
+
export, Robustness Map), `robustkit.report` (analyst vs. publisher
|
|
67
|
+
views, dispersion measures, combined Huber+IQR view),
|
|
68
|
+
`robustkit.quantiles` (generic JSON-stat loading, published-quantile-
|
|
69
|
+
trend visualization, and lognormal-calibrated reconstruction of
|
|
70
|
+
individual-level data from aggregated summaries), and
|
|
71
|
+
`robustkit.segment_awareness` (automatic hierarchical segmentation +
|
|
72
|
+
analysis, no manual hierarchy construction required) are stable and
|
|
73
|
+
tested.
|
|
74
|
+
|
|
75
|
+
**Recent fixes from real-dataset validation:**
|
|
76
|
+
- `rank_features`/`quadrant_report`/`rank_communicative_pairs` no
|
|
77
|
+
longer crash on pandas `Categorical` columns containing missing
|
|
78
|
+
values (found via OpenML's Boston Housing dataset).
|
|
79
|
+
- `bootstrap_band` (and `plot_analyst_view`, which uses it) now
|
|
80
|
+
defaults to `n_boot="auto"`, scaling iterations down for large
|
|
81
|
+
datasets since each iteration refits a full Huber model -- found to
|
|
82
|
+
become impractically slow at n_boot=200 on a ~54,000-row dataset.
|
|
83
|
+
Pass an explicit integer to opt out and always use exactly that many
|
|
84
|
+
iterations.
|
|
85
|
+
|
|
86
|
+
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
87
|
+
continuous features. `mutual_information` is estimated on the
|
|
88
|
+
full-resolution continuous values, while `entropy_bits` is computed on
|
|
89
|
+
a binned version of the same feature (since `entropy()` expects
|
|
90
|
+
categorical input). Binning discards information, so `entropy_bits` is
|
|
91
|
+
a lower bound on the feature's true entropy -- an efficiency above 1.0
|
|
92
|
+
signals that the feature carries more usable information than a coarse
|
|
93
|
+
categorical summary of it would capture. This is expected behavior,
|
|
94
|
+
not a bug.
|
|
95
|
+
|
|
96
|
+
## Installation
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
git clone https://github.com/<your-username>/robustkit.git
|
|
100
|
+
cd robustkit
|
|
101
|
+
pip install -e ".[dev]"
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## Quickstart
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
import numpy as np
|
|
108
|
+
from robustkit import (
|
|
109
|
+
fit_huber_trend, fit_tukey_trend, predict_trend,
|
|
110
|
+
model_stability_pct, cooks_diagnostic, cook_impact,
|
|
111
|
+
bootstrap_band, bca_bootstrap_ci,
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
# x: a single continuous predictor, y: a single continuous outcome
|
|
115
|
+
x = np.random.default_rng(0).uniform(20, 60, 200)
|
|
116
|
+
y = 1000 + 50 * x - 0.4 * x**2 + np.random.default_rng(1).normal(0, 500, 200)
|
|
117
|
+
|
|
118
|
+
fit = fit_huber_trend(x, y, degree=2)
|
|
119
|
+
y_pred = predict_trend(fit, x_new=[30, 40, 50])
|
|
120
|
+
|
|
121
|
+
stability = model_stability_pct(x, y)
|
|
122
|
+
print("Median % spread between Huber/Tukey/OLS:", stability["median_pct_diff"])
|
|
123
|
+
|
|
124
|
+
diag = cooks_diagnostic(x, y)
|
|
125
|
+
impact = cook_impact(x, y, diag["flagged_indices"])
|
|
126
|
+
print("Median % change in curve if flagged points removed:", impact["median_pct_change"])
|
|
127
|
+
|
|
128
|
+
band = bootstrap_band(x, y)
|
|
129
|
+
ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
|
|
133
|
+
|
|
134
|
+
## Trend growth rate and goodness of fit
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
|
|
138
|
+
|
|
139
|
+
fit = fit_huber_trend(df["age"], df["salary"])
|
|
140
|
+
|
|
141
|
+
# Rate of change of the trend itself (e.g. "salary growth per year of
|
|
142
|
+
# age"), not just its level
|
|
143
|
+
rates = trend_derivative(fit, x=[30, 40, 50])
|
|
144
|
+
|
|
145
|
+
# How well does this fit actually explain the variation in y?
|
|
146
|
+
goodness_of_fit(df["age"], df["salary"], degree=2)
|
|
147
|
+
|
|
148
|
+
# Don't assume a quadratic trend is always the right choice -- check
|
|
149
|
+
# empirically whether a higher degree captures meaningfully more
|
|
150
|
+
compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
**Note:** x is standardized internally before building polynomial
|
|
154
|
+
features (both here and throughout `robustkit.core`), since raw
|
|
155
|
+
polynomial features become numerically unstable at higher degrees for
|
|
156
|
+
realistic x scales (e.g. age^5 vastly outscales age^1). This is
|
|
157
|
+
transparent to callers -- `predict_trend` and `trend_derivative` still
|
|
158
|
+
take and return values in the original x scale.
|
|
159
|
+
|
|
160
|
+
## Segmentation
|
|
161
|
+
|
|
162
|
+
Run any `robustkit.core` analysis independently across subgroups of a
|
|
163
|
+
larger dataset, with automatic fallback to coarser groupings when a
|
|
164
|
+
finer one is too small to analyze reliably:
|
|
165
|
+
|
|
166
|
+
```python
|
|
167
|
+
from robustkit import hierarchical_segment, apply_by_segment, model_stability_pct
|
|
168
|
+
|
|
169
|
+
hierarchy = [["department", "level", "status"], ["level", "status"], ["status"]]
|
|
170
|
+
segmented = hierarchical_segment(df, hierarchy, min_size=20)
|
|
171
|
+
|
|
172
|
+
report = apply_by_segment(
|
|
173
|
+
segmented, segment_col="segment_id", x_col="age", y_col="value",
|
|
174
|
+
analysis_fn=model_stability_pct,
|
|
175
|
+
)
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
`apply_by_segment` works with any function shaped like
|
|
179
|
+
`analysis_fn(x, y, **kwargs) -> dict` -- built-in ones
|
|
180
|
+
(`model_stability_pct`, `cook_impact`, `bca_bootstrap_ci`, ...) or your
|
|
181
|
+
own. Only scalar values in the returned dict end up in the report
|
|
182
|
+
table; segments below `min_points` are skipped rather than causing an
|
|
183
|
+
error.
|
|
184
|
+
|
|
185
|
+
**Sanity-checking a segmentation before trusting it:**
|
|
186
|
+
`segment_consistency_report` runs a small battery of checks per
|
|
187
|
+
segment -- does it meet the recommended minimum size, and does fitting
|
|
188
|
+
a Huber trend on it use every row (robust methods don't need outliers
|
|
189
|
+
pre-removed, so a silently dropped row usually means a missing x/y
|
|
190
|
+
value slipped through, not intentional filtering):
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
from robustkit import segment_consistency_report
|
|
194
|
+
|
|
195
|
+
segment_consistency_report(df, segment_col="department", x_col="age", y_col="salary", min_size=20)
|
|
196
|
+
# segment n_total n_valid_xy n_dropped_missing_xy size_ok fit_ok fit_error
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
## Feature ranking (information)
|
|
200
|
+
|
|
201
|
+
Rank features by mutual information with a target, normalized by each
|
|
202
|
+
feature's own entropy, and classify them into four quadrants:
|
|
203
|
+
|
|
204
|
+
```python
|
|
205
|
+
from robustkit import rank_features, quadrant_report, plot_feature_space
|
|
206
|
+
|
|
207
|
+
ranking = rank_features(df, target="value")
|
|
208
|
+
report = quadrant_report(df, target="value") # adds a `quadrant` column
|
|
209
|
+
plot_feature_space(df, target="value") # same quadrants, visualized
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
`quadrant_report` and `plot_feature_space` always agree on quadrant
|
|
213
|
+
assignment -- both route through the same thresholding logic.
|
|
214
|
+
|
|
215
|
+
**Caveat:** default thresholds are the *median* mutual information /
|
|
216
|
+
efficiency across the ranked features. With only a handful of
|
|
217
|
+
features, this can put a genuinely weak feature in the same "high"
|
|
218
|
+
half as a strong one, since roughly half of any list sits above its
|
|
219
|
+
own median regardless of how large the actual gap is. Median
|
|
220
|
+
thresholding becomes meaningful with a reasonably large feature set;
|
|
221
|
+
for a handful of candidates, read the raw `mutual_information` /
|
|
222
|
+
`information_efficiency` values directly rather than relying on the
|
|
223
|
+
quadrant label alone.
|
|
224
|
+
|
|
225
|
+
See `examples/information_tutorial.py` for a complete walkthrough.
|
|
226
|
+
|
|
227
|
+
## Benchmarking against a global trend
|
|
228
|
+
|
|
229
|
+
Compare each segment's observed outcome against what a benchmark model
|
|
230
|
+
predicts, with bootstrap uncertainty on the difference -- answers
|
|
231
|
+
"which groups deviate from the overall trend, and by how much?" rather
|
|
232
|
+
than "how does the trend look overall?":
|
|
233
|
+
|
|
234
|
+
```python
|
|
235
|
+
from robustkit import segment_position_report
|
|
236
|
+
|
|
237
|
+
# Default: a single global Huber trend on one continuous x
|
|
238
|
+
report = segment_position_report(
|
|
239
|
+
df, segment_col="department", x_col="age", y_col="salary",
|
|
240
|
+
)
|
|
241
|
+
# segment n observed_median expected_median difference ci_lower ci_upper ci_available
|
|
242
|
+
# Finance 176 48339.70 47799.82 539.88 202.15 1031.01 True
|
|
243
|
+
# HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36 True
|
|
244
|
+
# IT 250 47226.50 47126.91 99.59 -117.56 510.81 True
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
A segment's confidence interval crossing zero means no clear deviation
|
|
248
|
+
from the benchmark; HR and Finance above don't cross zero, IT does.
|
|
249
|
+
|
|
250
|
+
**Custom, model-agnostic benchmarks:** the default single-column Huber
|
|
251
|
+
trend can be replaced with any richer model -- e.g. one using age,
|
|
252
|
+
age-squared, job level, overtime status, and a reference cluster
|
|
253
|
+
together, rather than a single x. Provide any object exposing
|
|
254
|
+
`predict(dataframe) -> array`:
|
|
255
|
+
|
|
256
|
+
```python
|
|
257
|
+
report = segment_position_report(
|
|
258
|
+
df, segment_col="department", y_col="salary", benchmark_fit=my_richer_model,
|
|
259
|
+
)
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
`segment_position_report` never inspects what the model uses
|
|
263
|
+
internally -- it only calls `predict()`.
|
|
264
|
+
|
|
265
|
+
**Small segments:** groups with fewer than `MIN_POINTS_FOR_CI` (default
|
|
266
|
+
20) observations still get `observed_median` / `expected_median` /
|
|
267
|
+
`difference`, but `ci_lower` / `ci_upper` are `NaN` and
|
|
268
|
+
`ci_available` is `False` -- a BCa bootstrap confidence interval (which
|
|
269
|
+
relies on a jackknife step) is not attempted for populations that
|
|
270
|
+
small, since it can fail outright or become statistically meaningless.
|
|
271
|
+
For segments at or above the threshold, the interval is a full BCa
|
|
272
|
+
(bias-corrected and accelerated) bootstrap interval, via the same
|
|
273
|
+
`bca_bootstrap_ci_by_index` primitive used elsewhere in the package --
|
|
274
|
+
not a plain percentile bootstrap.
|
|
275
|
+
|
|
276
|
+
## Reporting: residuals, individual deviations, batch runs, and Excel export
|
|
277
|
+
|
|
278
|
+
Four functions built on the same benchmark contract as
|
|
279
|
+
`segment_position_report`, for turning a benchmark into something a
|
|
280
|
+
non-technical audience (or a spreadsheet) can use directly:
|
|
281
|
+
|
|
282
|
+
```python
|
|
283
|
+
from robustkit import (
|
|
284
|
+
residual_summary, deviation_report,
|
|
285
|
+
benchmark_report_suite, export_benchmark_excel,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
# Per-segment residual SHAPE (not just the median difference):
|
|
289
|
+
residual_summary(df, segment_col="job_family", y_col="salary", x_col="age")
|
|
290
|
+
# segment n median_residual mad_residual p10_residual p90_residual
|
|
291
|
+
|
|
292
|
+
# Individuals furthest from the benchmark, sorted by residual --
|
|
293
|
+
# material for a conversation, not an automatic flag:
|
|
294
|
+
deviation_report(
|
|
295
|
+
df, y_col="salary", x_col="age", top_n=50, id_cols=["employee_id"],
|
|
296
|
+
)
|
|
297
|
+
# employee_id actual expected residual
|
|
298
|
+
|
|
299
|
+
# direction="negative" (default, furthest below), "positive" (furthest
|
|
300
|
+
# above), or "two_sided" (largest |residual| either direction).
|
|
301
|
+
|
|
302
|
+
# Run the same benchmark across several grouping columns at once,
|
|
303
|
+
# reusing ONE fitted benchmark so results are directly comparable:
|
|
304
|
+
reports = benchmark_report_suite(
|
|
305
|
+
df, group_columns=["gender", "job_family", "location"], y_col="salary", x_col="age",
|
|
306
|
+
)
|
|
307
|
+
# -> {"gender": DataFrame, "job_family": DataFrame, "location": DataFrame}
|
|
308
|
+
|
|
309
|
+
# Every report as its own sheet in one workbook:
|
|
310
|
+
export_benchmark_excel(reports, "salary_report.xlsx")
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
All four accept the same `benchmark_fit` / `x_col` contract as
|
|
314
|
+
`segment_position_report` (default single-column Huber trend, or any
|
|
315
|
+
custom model exposing `predict(dataframe)`).
|
|
316
|
+
|
|
317
|
+
**Dependency-free Excel export:** `export_benchmark_excel` requires
|
|
318
|
+
`openpyxl` (an optional dependency). In an offline/air-gapped
|
|
319
|
+
environment where installing it isn't possible, use
|
|
320
|
+
`export_benchmark_excel_no_deps` instead -- identical interface,
|
|
321
|
+
implemented with only the Python standard library (writes valid
|
|
322
|
+
`.xlsx` files via `zipfile` and OOXML templating directly, no
|
|
323
|
+
third-party package required):
|
|
324
|
+
|
|
325
|
+
```python
|
|
326
|
+
from robustkit import export_benchmark_excel_no_deps
|
|
327
|
+
|
|
328
|
+
export_benchmark_excel_no_deps(reports, "salary_report.xlsx")
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
Prefer `export_benchmark_excel` when `openpyxl` is available -- it's a
|
|
332
|
+
more complete, better-tested implementation of the Excel format.
|
|
333
|
+
|
|
334
|
+
## Robustness Map
|
|
335
|
+
|
|
336
|
+
Classify features by how much a conclusion about their relationship
|
|
337
|
+
with the target depends on (a) fitting method choice and (b) specific
|
|
338
|
+
influential observations -- two genuinely different failure modes that
|
|
339
|
+
a single diagnostic can miss:
|
|
340
|
+
|
|
341
|
+
```python
|
|
342
|
+
from robustkit import feature_robustness_report, plot_feature_robustness
|
|
343
|
+
|
|
344
|
+
report = feature_robustness_report(df, target="value")
|
|
345
|
+
# feature stability_pct cook_impact_pct quadrant
|
|
346
|
+
# CRIM 8.9 17.1 fragile
|
|
347
|
+
# AGE 16.8 15.5 fragile
|
|
348
|
+
# RM 4.9 0.1 robust
|
|
349
|
+
# TAX 22.2 1.4 structural_sensitivity
|
|
350
|
+
|
|
351
|
+
plot_feature_robustness(report=report)
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
Four quadrants: **robust** (low spread, low impact), **structural
|
|
355
|
+
sensitivity** (sensitive to fitting method, not to specific points),
|
|
356
|
+
**data sensitive** (a few points drive the conclusion, method choice
|
|
357
|
+
barely matters), **fragile** (both -- least trustworthy).
|
|
358
|
+
|
|
359
|
+
`quadrant_report`/`plot_feature_space` (information) and
|
|
360
|
+
`feature_robustness_report`/`plot_feature_robustness` (benchmark) both
|
|
361
|
+
route through the same shared classifier, `robustkit.classify_quadrants`
|
|
362
|
+
-- any future quadrant-based analysis in this package will too.
|
|
363
|
+
|
|
364
|
+
## Analyst view vs. publisher view
|
|
365
|
+
|
|
366
|
+
Two visualizations that look superficially similar but answer
|
|
367
|
+
genuinely different questions:
|
|
368
|
+
|
|
369
|
+
```python
|
|
370
|
+
from robustkit import plot_analyst_view, plot_publisher_view, dispersion_ratio, iqr
|
|
371
|
+
|
|
372
|
+
# "How confident are we in the trend estimate?" -- a bootstrap
|
|
373
|
+
# confidence band that SHRINKS as sample size grows.
|
|
374
|
+
plot_analyst_view(df["age"], df["salary"])
|
|
375
|
+
|
|
376
|
+
# "How spread out are actual values in the population?" -- a median +
|
|
377
|
+
# IQR band that does NOT shrink with more data, since it reflects
|
|
378
|
+
# real dispersion, not estimation uncertainty. show_points defaults to
|
|
379
|
+
# False, since this view is meant for publishing potentially sensitive
|
|
380
|
+
# data (e.g. individual salaries) without exposing raw points.
|
|
381
|
+
plot_publisher_view(df["age"], df["salary"])
|
|
382
|
+
```
|
|
383
|
+
|
|
384
|
+
This distinction matters in practice: with 20x more data (same
|
|
385
|
+
underlying distribution), the analyst view's confidence band roughly
|
|
386
|
+
halves in width, while the publisher view's IQR band stays essentially
|
|
387
|
+
unchanged -- confirmed by the package's own test suite.
|
|
388
|
+
|
|
389
|
+
Both accept an optional `title=None` to override the default title
|
|
390
|
+
(e.g. `plot_analyst_view(x, y, title="Q3 salary review")`), as does
|
|
391
|
+
`plot_quantile_trend`.
|
|
392
|
+
|
|
393
|
+
`dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
|
|
394
|
+
standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
395
|
+
computes both across bins of a continuous x, e.g. to check whether
|
|
396
|
+
dispersion (inequality) grows with age.
|
|
397
|
+
|
|
398
|
+
## Segment awareness: automatic hierarchical grouping + analysis
|
|
399
|
+
|
|
400
|
+
`segment_stability_report` and `segment_benchmark_report` build the
|
|
401
|
+
hierarchical segmentation automatically from a flat, most-specific-
|
|
402
|
+
first list of columns, then run an existing analysis within the
|
|
403
|
+
result -- no separate `hierarchical_segment(...)` + `apply_by_segment(...)`
|
|
404
|
+
preparation step required:
|
|
405
|
+
|
|
406
|
+
```python
|
|
407
|
+
from robustkit import segment_stability_report, segment_benchmark_report
|
|
408
|
+
|
|
409
|
+
# hierarchy built automatically: [JobFamily, Level, OT] -> [Level, OT] -> [OT] -> ALL
|
|
410
|
+
report = segment_stability_report(
|
|
411
|
+
df, x_col="age", y_col="salary",
|
|
412
|
+
segment_cols=["JobFamily", "Level", "OT"], min_size=20,
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
report = segment_benchmark_report(
|
|
416
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"],
|
|
417
|
+
x_col="age", min_size=20, # or benchmark_fit=my_custom_model
|
|
418
|
+
)
|
|
419
|
+
```
|
|
420
|
+
|
|
421
|
+
Both add a `segment_level` column showing which tier of the hierarchy
|
|
422
|
+
each reported segment actually landed on (0 = finest), so a fallback
|
|
423
|
+
to a coarser grouping is visible rather than silent. These are pure
|
|
424
|
+
convenience wrappers -- identical results to building the hierarchy
|
|
425
|
+
by hand with `hierarchical_segment` and calling `apply_by_segment` /
|
|
426
|
+
`segment_position_report` directly.
|
|
427
|
+
|
|
428
|
+
`mad_outlier_report` flags individuals whose residual is an outlier
|
|
429
|
+
relative to their OWN segment's typical spread (MAD), not the whole
|
|
430
|
+
population -- built on the same automatic hierarchical segmentation as
|
|
431
|
+
above, so even someone in a small segment is compared against a
|
|
432
|
+
sensibly-sized reference group rather than an irrelevant one:
|
|
433
|
+
|
|
434
|
+
```python
|
|
435
|
+
from robustkit import mad_outlier_report
|
|
436
|
+
|
|
437
|
+
report = mad_outlier_report(
|
|
438
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age",
|
|
439
|
+
min_size=20, k=3.0, direction="negative", id_cols=["employee_id"],
|
|
440
|
+
)
|
|
441
|
+
# employee_id actual expected residual residual_pct segment segment_level segment_mad threshold flagged
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
`direction`: `"negative"` (default -- flag underperformance relative
|
|
445
|
+
to the benchmark), `"positive"`, or `"two_sided"`. Flagging compares
|
|
446
|
+
each residual to `k` MADs from its *own segment's* median residual
|
|
447
|
+
(not literally zero), so a segment the benchmark is systematically
|
|
448
|
+
biased for doesn't get every member flagged just for that bias.
|
|
449
|
+
Segments with zero MAD (a degenerate case, usually a tiny segment
|
|
450
|
+
where every residual happens to match) are treated as having an
|
|
451
|
+
infinite threshold rather than flagging everyone in them.
|
|
452
|
+
|
|
453
|
+
`export_outlier_pdf` renders one chart per segment -- built from the
|
|
454
|
+
same segmentation and flagging as `mad_outlier_report` -- as a
|
|
455
|
+
one-page-per-segment PDF, for visual verification alongside the
|
|
456
|
+
numeric report:
|
|
457
|
+
|
|
458
|
+
```python
|
|
459
|
+
from robustkit import export_outlier_pdf
|
|
460
|
+
|
|
461
|
+
export_outlier_pdf(
|
|
462
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age",
|
|
463
|
+
path="outliers.pdf", min_size=20, k=3.0, id_cols=["employee_id"],
|
|
464
|
+
)
|
|
465
|
+
```
|
|
466
|
+
|
|
467
|
+
Each page plots every observation in that segment, the benchmark's
|
|
468
|
+
expected values (the exact same values used for flagging, not a
|
|
469
|
+
separately re-fit curve), and flagged outliers marked distinctly.
|
|
470
|
+
Segments with fewer than `min_points_to_plot` (default 5) observations
|
|
471
|
+
are skipped in the PDF -- a chart with a handful of points isn't
|
|
472
|
+
meaningfully verifiable -- but still appear in `mad_outlier_report`'s
|
|
473
|
+
numeric output. The idea: a numeric flag and a visual confirmation are
|
|
474
|
+
two independent checks, and agreement between them is stronger
|
|
475
|
+
evidence than either alone.
|
|
476
|
+
|
|
477
|
+
### Drilldown reports: every hierarchy level at once, without exclusive assignment
|
|
478
|
+
|
|
479
|
+
`segment_stability_report`, `segment_benchmark_report`, and
|
|
480
|
+
`mad_outlier_report` each assign every individual to exactly ONE
|
|
481
|
+
segment (their most specific grouping meeting `min_size`). That
|
|
482
|
+
answers "what is the single most relevant reference population for
|
|
483
|
+
THIS individual?"
|
|
484
|
+
|
|
485
|
+
`segment_benchmark_drilldown_report` and `mad_outlier_drilldown_report`
|
|
486
|
+
answer a different question -- "what does every granularity level look
|
|
487
|
+
like on its own?" -- by reporting EVERY level of the hierarchy
|
|
488
|
+
independently, without exclusive assignment. The same individual can
|
|
489
|
+
appear in multiple rows (e.g. once in a `JobFamily x Level x OT` row,
|
|
490
|
+
and again in the broader `Level x OT` row), whenever both groupings
|
|
491
|
+
independently meet `min_size`:
|
|
492
|
+
|
|
493
|
+
```python
|
|
494
|
+
from robustkit import segment_benchmark_drilldown_report, mad_outlier_drilldown_report
|
|
495
|
+
|
|
496
|
+
segment_benchmark_drilldown_report(
|
|
497
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age", min_size=20,
|
|
498
|
+
)
|
|
499
|
+
mad_outlier_drilldown_report(
|
|
500
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age", k=3.0,
|
|
501
|
+
)
|
|
502
|
+
```
|
|
503
|
+
|
|
504
|
+
Both add a `segment_level` column, and the sum of `n` across rows will
|
|
505
|
+
exceed the population size -- that's the expected signature of
|
|
506
|
+
overlap, not a bug. Use the exclusive functions when you need to route
|
|
507
|
+
each individual to one home; use the drilldown functions when you want
|
|
508
|
+
to see every level side by side.
|
|
509
|
+
|
|
510
|
+
## Combined model + spread view
|
|
511
|
+
|
|
512
|
+
`plot_analyst_view` and `plot_publisher_view` each show one thing --
|
|
513
|
+
estimation uncertainty, or population spread -- deliberately kept
|
|
514
|
+
separate. `plot_huber_iqr` shows both together: one or more trend
|
|
515
|
+
curves overlaid with median + IQR error bars, plus an optional
|
|
516
|
+
residual-quality box, matching the combined model-and-spread diagram
|
|
517
|
+
style common in salary/wage analysis reporting:
|
|
518
|
+
|
|
519
|
+
```python
|
|
520
|
+
from robustkit import plot_huber_iqr
|
|
521
|
+
|
|
522
|
+
result = plot_huber_iqr(df["age"], df["salary"], degree=2, bins=15)
|
|
523
|
+
# result["grid"], result["huber"], result["binned"]
|
|
524
|
+
```
|
|
525
|
+
|
|
526
|
+
`show_points` defaults to `False`, consistent with `plot_publisher_view`.
|
|
527
|
+
|
|
528
|
+
**Multiple curves, bootstrap bands, and full style control:**
|
|
529
|
+
|
|
530
|
+
```python
|
|
531
|
+
plot_huber_iqr(
|
|
532
|
+
df["age"], df["salary"],
|
|
533
|
+
methods=("huber", "tukey", "ols", "median_ensemble"), # overlay all four
|
|
534
|
+
show_bootstrap_band=True, bootstrap_levels=(95, 50), # nested confidence bands
|
|
535
|
+
cap_style="manual", # hand-drawn boxplot-style Q1/Q3 "hats" instead of matplotlib's default caps
|
|
536
|
+
residual_box_metric="mdape", # MdAPE + IQR(resid) instead of R^2/MAE/RMSE
|
|
537
|
+
ylim="dynamic", # y-limits set from the data (min*0.95, max*1.05)
|
|
538
|
+
style={
|
|
539
|
+
"huber_line": {"color": "red", "linewidth": 2, "linestyle": "-", "label": "Huber poly(2)"},
|
|
540
|
+
"iqr_color": "black", "cap_width": 0.15,
|
|
541
|
+
},
|
|
542
|
+
)
|
|
543
|
+
```
|
|
544
|
+
|
|
545
|
+
`methods` selects which trend curve(s) to draw (`"tukey"` and
|
|
546
|
+
`"median_ensemble"` -- the pointwise median of Huber/Tukey/OLS --
|
|
547
|
+
require statsmodels). `style` overrides individual colors, line
|
|
548
|
+
widths, and other visual details without needing to touch anything
|
|
549
|
+
else; every new parameter here defaults to the original, simpler
|
|
550
|
+
single-Huber-curve appearance, so existing calls are unaffected.
|
|
551
|
+
|
|
552
|
+
**Two grouping strategies:** `grouping="bin"` (default) uses quantile-
|
|
553
|
+
based binning for stable estimates even in small populations.
|
|
554
|
+
`grouping="unique"` instead groups by each EXACT x value (e.g. every
|
|
555
|
+
individual age in years) -- matching a workbook-style `groupby(x)`
|
|
556
|
+
aggregation -- and, when a given x value has fewer than
|
|
557
|
+
`min_n_for_iqr` (default 5) observations, omits its IQR error bar
|
|
558
|
+
entirely rather than showing an unreliable one:
|
|
559
|
+
|
|
560
|
+
**Caveat, found via validation against a real dataset:** `grouping="unique"`
|
|
561
|
+
only makes sense for x values with natural repetition (e.g. integer
|
|
562
|
+
ages) -- for a genuinely continuous, high-precision measurement (e.g.
|
|
563
|
+
carat weight to several decimal places), nearly every x value is
|
|
564
|
+
unique, so almost nothing meets `min_n_for_iqr` and the result shows
|
|
565
|
+
no IQR bars at all. Use `grouping="bin"` (the default) for
|
|
566
|
+
high-precision continuous x; reserve `grouping="unique"` for x values
|
|
567
|
+
that naturally repeat.
|
|
568
|
+
|
|
569
|
+
```python
|
|
570
|
+
plot_huber_iqr(
|
|
571
|
+
df["age"], df["salary"], grouping="unique", min_n_for_iqr=5,
|
|
572
|
+
)
|
|
573
|
+
```
|
|
574
|
+
|
|
575
|
+
The absence of an error bar at a given age is itself information --
|
|
576
|
+
it signals the sample at that exact value is too small to say
|
|
577
|
+
anything about spread, not just a plotting simplification.
|
|
578
|
+
|
|
579
|
+
## Loading published quantile tables (SCB / JSON-stat)
|
|
580
|
+
|
|
581
|
+
Some statistics agencies (e.g. Statistics Sweden, SCB) publish
|
|
582
|
+
quantiles (Q1/median/Q3) directly, with no individual-level data
|
|
583
|
+
available at all. `robustkit.quantiles` loads these tables generically
|
|
584
|
+
via JSON-stat, a standardized dimensional-data format used by SCB and
|
|
585
|
+
other national statistics agencies -- avoiding the fragility of
|
|
586
|
+
parsing metadata out of column-name strings in a wide CSV export.
|
|
587
|
+
|
|
588
|
+
```python
|
|
589
|
+
from robustkit import load_json_stat, plot_quantile_trend, quantile_trend_dispersion
|
|
590
|
+
|
|
591
|
+
df = load_json_stat("some_scb_table.json")
|
|
592
|
+
|
|
593
|
+
# A real SCB quirk this loader does NOT try to guess automatically:
|
|
594
|
+
# category labels can change meaning over time (e.g. Sweden's oldest
|
|
595
|
+
# working-age bracket was labeled "65-66 år" through 2022 and
|
|
596
|
+
# "65-68 år" from 2023, following a pension-age reform). Merge such
|
|
597
|
+
# cases explicitly:
|
|
598
|
+
df = load_json_stat(
|
|
599
|
+
"some_scb_table.json",
|
|
600
|
+
rename_categories={"ålder": {"65–68 år": "65–66 år"}},
|
|
601
|
+
)
|
|
602
|
+
|
|
603
|
+
# Once reshaped to a wide table with q1/median/q3 columns:
|
|
604
|
+
plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
605
|
+
quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
606
|
+
```
|
|
607
|
+
|
|
608
|
+
This is the "quantiles are already given" case. A complementary case
|
|
609
|
+
-- reconstructing approximate individual-level data from aggregated
|
|
610
|
+
group means, for when only summary statistics (not quantiles) are
|
|
611
|
+
available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
|
|
612
|
+
|
|
613
|
+
See `examples/quantiles_tutorial.py` for a complete walkthrough.
|
|
614
|
+
|
|
615
|
+
## Reconstructing individual-level data from aggregated summaries
|
|
616
|
+
|
|
617
|
+
For the complementary case -- only aggregated group summaries (n,
|
|
618
|
+
Q1, median, Q3) are available, not the quantile trend itself as the
|
|
619
|
+
final answer, and you want to run `robustkit.core` analyses as if
|
|
620
|
+
individual data existed:
|
|
621
|
+
|
|
622
|
+
```python
|
|
623
|
+
from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
|
|
624
|
+
|
|
625
|
+
# One row per group (e.g. year), with n/q1/median/q3 columns
|
|
626
|
+
synthetic = expand_aggregated_table(
|
|
627
|
+
summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
|
|
628
|
+
group_cols=["year"], value_name="salary",
|
|
629
|
+
)
|
|
630
|
+
|
|
631
|
+
# Now usable exactly like real individual-level data:
|
|
632
|
+
fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
|
|
633
|
+
```
|
|
634
|
+
|
|
635
|
+
Method: a lognormal distribution is calibrated (via the IQR) to match
|
|
636
|
+
each group's reported Q1/median/Q3, then `n` synthetic values are
|
|
637
|
+
drawn from it. Validated end-to-end against real published SCB salary
|
|
638
|
+
data: a Huber trend fitted on reconstructed pseudo-individual data
|
|
639
|
+
tracked the true published median trend within 2% across 12 years.
|
|
640
|
+
|
|
641
|
+
**Note on what this recovers:** because a Huber (or Tukey) fit on
|
|
642
|
+
right-skewed reconstructed data tracks something close to the
|
|
643
|
+
*median* trend it was calibrated against -- not the arithmetic mean --
|
|
644
|
+
this is consistent with, not a limitation of, the reconstruction
|
|
645
|
+
method. To target the mean instead, fit on `log(value)` and
|
|
646
|
+
exponentiate predictions back, which approximates the geometric mean.
|
|
647
|
+
|
|
648
|
+
Always check `check_reconstruction_quality()` before trusting a
|
|
649
|
+
reconstruction: real Q1/median/Q3 triples aren't always perfectly
|
|
650
|
+
consistent with a pure lognormal shape.
|
|
651
|
+
|
|
652
|
+
**Warning -- unbounded tail at large n:** a lognormal has no natural
|
|
653
|
+
upper limit, and its expected maximum grows with n. Reconstructing at
|
|
654
|
+
the TRUE group size from a national table (SCB salary tables can
|
|
655
|
+
report n in the hundreds of thousands to millions) can produce
|
|
656
|
+
implausibly extreme tail values -- real salaries have practical
|
|
657
|
+
ceilings a pure lognormal doesn't know about. This package's own
|
|
658
|
+
examples and tests deliberately scale n down to a few thousand for
|
|
659
|
+
demonstration; calibration quality (matching Q1/median/Q3) doesn't
|
|
660
|
+
depend on reproducing the true population size, but tail plausibility
|
|
661
|
+
does. No clipping is applied automatically.
|
|
662
|
+
|
|
663
|
+
### When only a mean is available (no quantiles at all)
|
|
664
|
+
|
|
665
|
+
Some tables (e.g. SCB's age-breakdown salary tables) report only a
|
|
666
|
+
mean per group, with no spread information. Two deliberately separate
|
|
667
|
+
methods are provided, each making a different explicit assumption --
|
|
668
|
+
compare them rather than silently picking one:
|
|
669
|
+
|
|
670
|
+
```python
|
|
671
|
+
from robustkit import (
|
|
672
|
+
expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
|
|
673
|
+
compare_reconstruction_methods,
|
|
674
|
+
)
|
|
675
|
+
|
|
676
|
+
# Method 1: repeat the mean n times -- zero within-group spread.
|
|
677
|
+
# Recovers between-group regression coefficients reasonably well
|
|
678
|
+
# (validated in the original technique this is based on) but
|
|
679
|
+
# understates individual-level variation.
|
|
680
|
+
flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
|
|
681
|
+
|
|
682
|
+
# Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
|
|
683
|
+
# report quantiles, and use it to imply an approximate spread around
|
|
684
|
+
# the mean. Stacks two assumptions (mean-as-median, and that the
|
|
685
|
+
# borrowed ratio transfers to this population) -- illustrative, not a
|
|
686
|
+
# substitute for genuine quantile data for this specific table.
|
|
687
|
+
borrowed = expand_aggregated_table_borrowed_dispersion(
|
|
688
|
+
df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
|
|
689
|
+
)
|
|
690
|
+
|
|
691
|
+
# Compare both for a single group directly:
|
|
692
|
+
compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
|
|
693
|
+
```
|
|
694
|
+
|
|
695
|
+
Both methods are documented with their specific assumptions rather
|
|
696
|
+
than presented as equally valid defaults -- being explicit about which
|
|
697
|
+
assumption was made lets the analyst judge how much a conclusion
|
|
698
|
+
depends on it, rather than presenting an assumption as a measurement.
|
|
699
|
+
|
|
700
|
+
## Feature pairing (information)
|
|
701
|
+
|
|
702
|
+
Beyond ranking single features, evaluate *pairs* of features together:
|
|
703
|
+
how redundant are they with each other, and does knowing one reveal
|
|
704
|
+
additional predictive value in the other (synergy, e.g. an interaction
|
|
705
|
+
effect)?
|
|
706
|
+
|
|
707
|
+
```python
|
|
708
|
+
from robustkit import (
|
|
709
|
+
conditional_mutual_information, communication_score,
|
|
710
|
+
rank_by_communication, pair_redundancy, pair_synergy,
|
|
711
|
+
rank_communicative_pairs,
|
|
712
|
+
)
|
|
713
|
+
|
|
714
|
+
# How communicable is a single feature -- not just predictive, but
|
|
715
|
+
# suitable for a clear chart/table (adequate group sizes, homogeneous
|
|
716
|
+
# groups, few enough categories to show at once)?
|
|
717
|
+
comm_ranking = rank_by_communication(df, target="value")
|
|
718
|
+
|
|
719
|
+
# How much does region's relevance to the target change once
|
|
720
|
+
# department is already known?
|
|
721
|
+
synergy = pair_synergy(df, feature_1="department", feature_2="region", target="value")
|
|
722
|
+
|
|
723
|
+
# Rank every candidate pair by combined relevance, penalizing
|
|
724
|
+
# redundant pairs and rewarding genuine synergy
|
|
725
|
+
pairs = rank_communicative_pairs(df, target="value")
|
|
726
|
+
```
|
|
727
|
+
|
|
728
|
+
All mutual-information-based quantities in this module (`rank_features`,
|
|
729
|
+
`conditional_mutual_information`, `pair_redundancy`, `pair_synergy`,
|
|
730
|
+
`communication_score`) are expressed in **bits**, consistent with
|
|
731
|
+
`entropy()` -- internally, scikit-learn's MI estimators return nats
|
|
732
|
+
and are converted before being used anywhere in this package.
|
|
733
|
+
|
|
734
|
+
## Design principles
|
|
735
|
+
|
|
736
|
+
- **One continuous x, one continuous y** at the core. This keeps every
|
|
737
|
+
function's output visually and numerically interpretable (a curve
|
|
738
|
+
you can plot, a band you can read).
|
|
739
|
+
- **Diagnosis and action are separate steps.** `cooks_diagnostic`
|
|
740
|
+
flags candidates; `cook_impact` tells you whether removing them
|
|
741
|
+
actually changes anything.
|
|
742
|
+
- **OLS is a reference point, not the enemy.** Comparing robust fits
|
|
743
|
+
against OLS is how you know whether robustness mattered at all.
|
|
744
|
+
|
|
745
|
+
## Naming conventions
|
|
746
|
+
|
|
747
|
+
A few parameter/column names look similar across the package but mean
|
|
748
|
+
different things -- documented here explicitly so the difference reads
|
|
749
|
+
as intentional, not as an inconsistency to "fix":
|
|
750
|
+
|
|
751
|
+
- **`residual` vs. `difference`:** `residual` is an INDIVIDUAL-level
|
|
752
|
+
quantity (`actual - expected` for one row) -- used by
|
|
753
|
+
`mad_outlier_report`, `mad_outlier_drilldown_report`, and
|
|
754
|
+
`deviation_report`. `difference` is a SEGMENT/GROUP-level quantity
|
|
755
|
+
(typically the median residual within a group) -- used by
|
|
756
|
+
`segment_position_report`, `segment_benchmark_report`,
|
|
757
|
+
`segment_benchmark_drilldown_report`, and `benchmark_report_suite`.
|
|
758
|
+
- **`target` vs. `y_col`:** `robustkit.information` uses `target` for
|
|
759
|
+
the column being explained, since it works with arbitrary features
|
|
760
|
+
(not necessarily a continuous regression outcome).
|
|
761
|
+
`robustkit.benchmark`, `robustkit.segment_awareness`, and
|
|
762
|
+
`robustkit.quantiles` use `y_col`, since they specifically model a
|
|
763
|
+
continuous `y` as a function of `x`.
|
|
764
|
+
- **`segment_cols` vs. `group_columns`:** `segment_cols` (throughout
|
|
765
|
+
`robustkit.segment_awareness`) is an ORDERED, most-specific-first
|
|
766
|
+
list used to build a fallback HIERARCHY (see `hierarchical_segment`).
|
|
767
|
+
`group_columns` (`benchmark_report_suite`) is a FLAT list of
|
|
768
|
+
independent groupings, run separately with no hierarchy or fallback
|
|
769
|
+
between them. Different structure, different name on purpose.
|
|
770
|
+
- **`min_size` vs. `min_points` vs. `min_stratum_size` vs.
|
|
771
|
+
`min_group_size`:** all mean "minimum group size," but at different
|
|
772
|
+
stages: `min_size` (`hierarchical_segment` and everything built on
|
|
773
|
+
it) gates whether a hierarchy LEVEL gets created at all;
|
|
774
|
+
`min_points` (`apply_by_segment`) gates whether an already-built
|
|
775
|
+
segment gets ANALYZED; `min_stratum_size`
|
|
776
|
+
(`conditional_mutual_information`) and `min_group_size`
|
|
777
|
+
(`communication_score`, `rank_by_communication`) are specific to
|
|
778
|
+
those `robustkit.information` calculations. Kept separate rather
|
|
779
|
+
than unified to one name, since collapsing them would obscure which
|
|
780
|
+
stage of a pipeline each threshold actually applies to.
|
|
781
|
+
|
|
782
|
+
## License
|
|
783
|
+
|
|
784
|
+
MIT -- see [LICENSE](LICENSE).
|