robustkit 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {robustkit-0.5.0/robustkit.egg-info → robustkit-0.6.0}/PKG-INFO +355 -17
- {robustkit-0.5.0 → robustkit-0.6.0}/README.md +352 -16
- {robustkit-0.5.0 → robustkit-0.6.0}/pyproject.toml +2 -1
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/__init__.py +27 -5
- robustkit-0.6.0/robustkit/benchmark/global_model.py +162 -0
- robustkit-0.6.0/robustkit/benchmark/reporting.py +317 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/benchmark/robustness_map.py +1 -1
- robustkit-0.6.0/robustkit/core/consistency.py +90 -0
- robustkit-0.6.0/robustkit/core/uncertainty.py +171 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/mutual_info.py +13 -1
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/visualization.py +1 -1
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/quantiles/io.py +1 -1
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/quantiles/trend.py +3 -2
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/report/visualize_analyst.py +7 -2
- robustkit-0.6.0/robustkit/report/visualize_huber_iqr.py +313 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/report/visualize_publisher.py +3 -2
- robustkit-0.6.0/robustkit/segment_awareness/__init__.py +17 -0
- robustkit-0.6.0/robustkit/segment_awareness/reports.py +470 -0
- {robustkit-0.5.0 → robustkit-0.6.0/robustkit.egg-info}/PKG-INFO +355 -17
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit.egg-info/SOURCES.txt +15 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit.egg-info/requires.txt +3 -0
- robustkit-0.6.0/tests/test_benchmark_gaps.py +159 -0
- robustkit-0.6.0/tests/test_benchmark_reporting.py +195 -0
- robustkit-0.6.0/tests/test_bugfixes_faseA.py +102 -0
- robustkit-0.6.0/tests/test_export_no_deps.py +119 -0
- robustkit-0.6.0/tests/test_export_outlier_pdf.py +116 -0
- robustkit-0.6.0/tests/test_huber_iqr.py +197 -0
- robustkit-0.6.0/tests/test_mad_outlier_report.py +117 -0
- robustkit-0.6.0/tests/test_n_boot_auto_propagation.py +86 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_quantiles_io_trend.py +17 -9
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_report.py +20 -0
- robustkit-0.6.0/tests/test_segment_awareness.py +124 -0
- robustkit-0.6.0/tests/test_segment_consistency.py +72 -0
- robustkit-0.6.0/tests/test_segment_drilldown.py +84 -0
- robustkit-0.5.0/robustkit/benchmark/global_model.py +0 -74
- robustkit-0.5.0/robustkit/core/consistency.py +0 -33
- robustkit-0.5.0/robustkit/core/uncertainty.py +0 -83
- {robustkit-0.5.0 → robustkit-0.6.0}/LICENSE +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/benchmark/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/common/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/common/quadrants.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/core/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/core/diagnostics.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/core/goodness_of_fit.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/core/stability.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/core/trend.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/communication.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/conditional_mi.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/entropy.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/pairs.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/profile.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/quadrants.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/information/utils.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/quantiles/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/quantiles/reconstruct.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/report/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/report/dispersion.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/segmentation/__init__.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/segmentation/apply.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit/segmentation/hierarchy.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit.egg-info/dependency_links.txt +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/robustkit.egg-info/top_level.txt +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/setup.cfg +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_benchmark.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_common_quadrants.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_core.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_information.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_information_pairs.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_quantiles_reconstruct.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_quantiles_reconstruct_mean_only.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_segmentation.py +0 -0
- {robustkit-0.5.0 → robustkit-0.6.0}/tests/test_trend_extras.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: robustkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
|
|
5
5
|
Author: Mikael Lundqvist
|
|
6
6
|
License: MIT License
|
|
@@ -36,6 +36,8 @@ Requires-Dist: scipy>=1.10
|
|
|
36
36
|
Requires-Dist: matplotlib>=3.7
|
|
37
37
|
Provides-Extra: dev
|
|
38
38
|
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
39
|
+
Provides-Extra: excel
|
|
40
|
+
Requires-Dist: openpyxl>=3.1; extra == "excel"
|
|
39
41
|
Dynamic: license-file
|
|
40
42
|
|
|
41
43
|
# robustkit
|
|
@@ -60,11 +62,26 @@ consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
|
60
62
|
per-segment analysis), `robustkit.information` (mutual-information
|
|
61
63
|
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
62
64
|
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
65
|
+
model-agnostic custom benchmarks, residual/deviation reporting, Excel
|
|
66
|
+
export, Robustness Map), `robustkit.report` (analyst vs. publisher
|
|
67
|
+
views, dispersion measures, combined Huber+IQR view),
|
|
68
|
+
`robustkit.quantiles` (generic JSON-stat loading, published-quantile-
|
|
69
|
+
trend visualization, and lognormal-calibrated reconstruction of
|
|
70
|
+
individual-level data from aggregated summaries), and
|
|
71
|
+
`robustkit.segment_awareness` (automatic hierarchical segmentation +
|
|
72
|
+
analysis, no manual hierarchy construction required) are stable and
|
|
73
|
+
tested.
|
|
74
|
+
|
|
75
|
+
**Recent fixes from real-dataset validation:**
|
|
76
|
+
- `rank_features`/`quadrant_report`/`rank_communicative_pairs` no
|
|
77
|
+
longer crash on pandas `Categorical` columns containing missing
|
|
78
|
+
values (found via OpenML's Boston Housing dataset).
|
|
79
|
+
- `bootstrap_band` (and `plot_analyst_view`, which uses it) now
|
|
80
|
+
defaults to `n_boot="auto"`, scaling iterations down for large
|
|
81
|
+
datasets since each iteration refits a full Huber model -- found to
|
|
82
|
+
become impractically slow at n_boot=200 on a ~54,000-row dataset.
|
|
83
|
+
Pass an explicit integer to opt out and always use exactly that many
|
|
84
|
+
iterations.
|
|
68
85
|
|
|
69
86
|
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
70
87
|
continuous features. `mutual_information` is estimated on the
|
|
@@ -165,6 +182,20 @@ own. Only scalar values in the returned dict end up in the report
|
|
|
165
182
|
table; segments below `min_points` are skipped rather than causing an
|
|
166
183
|
error.
|
|
167
184
|
|
|
185
|
+
**Sanity-checking a segmentation before trusting it:**
|
|
186
|
+
`segment_consistency_report` runs a small battery of checks per
|
|
187
|
+
segment -- does it meet the recommended minimum size, and does fitting
|
|
188
|
+
a Huber trend on it use every row (robust methods don't need outliers
|
|
189
|
+
pre-removed, so a silently dropped row usually means a missing x/y
|
|
190
|
+
value slipped through, not intentional filtering):
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
from robustkit import segment_consistency_report
|
|
194
|
+
|
|
195
|
+
segment_consistency_report(df, segment_col="department", x_col="age", y_col="salary", min_size=20)
|
|
196
|
+
# segment n_total n_valid_xy n_dropped_missing_xy size_ok fit_ok fit_error
|
|
197
|
+
```
|
|
198
|
+
|
|
168
199
|
## Feature ranking (information)
|
|
169
200
|
|
|
170
201
|
Rank features by mutual information with a target, normalized by each
|
|
@@ -195,26 +226,111 @@ See `examples/information_tutorial.py` for a complete walkthrough.
|
|
|
195
226
|
|
|
196
227
|
## Benchmarking against a global trend
|
|
197
228
|
|
|
198
|
-
Compare each segment's observed outcome against what a
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
229
|
+
Compare each segment's observed outcome against what a benchmark model
|
|
230
|
+
predicts, with bootstrap uncertainty on the difference -- answers
|
|
231
|
+
"which groups deviate from the overall trend, and by how much?" rather
|
|
232
|
+
than "how does the trend look overall?":
|
|
202
233
|
|
|
203
234
|
```python
|
|
204
235
|
from robustkit import segment_position_report
|
|
205
236
|
|
|
237
|
+
# Default: a single global Huber trend on one continuous x
|
|
206
238
|
report = segment_position_report(
|
|
207
239
|
df, segment_col="department", x_col="age", y_col="salary",
|
|
208
240
|
)
|
|
209
|
-
# segment n observed_median expected_median difference ci_lower ci_upper
|
|
210
|
-
# Finance 176 48339.70 47799.82 539.88 202.15 1031.01
|
|
211
|
-
# HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36
|
|
212
|
-
# IT 250 47226.50 47126.91 99.59 -117.56 510.81
|
|
241
|
+
# segment n observed_median expected_median difference ci_lower ci_upper ci_available
|
|
242
|
+
# Finance 176 48339.70 47799.82 539.88 202.15 1031.01 True
|
|
243
|
+
# HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36 True
|
|
244
|
+
# IT 250 47226.50 47126.91 99.59 -117.56 510.81 True
|
|
213
245
|
```
|
|
214
246
|
|
|
215
247
|
A segment's confidence interval crossing zero means no clear deviation
|
|
216
248
|
from the benchmark; HR and Finance above don't cross zero, IT does.
|
|
217
249
|
|
|
250
|
+
**Custom, model-agnostic benchmarks:** the default single-column Huber
|
|
251
|
+
trend can be replaced with any richer model -- e.g. one using age,
|
|
252
|
+
age-squared, job level, overtime status, and a reference cluster
|
|
253
|
+
together, rather than a single x. Provide any object exposing
|
|
254
|
+
`predict(dataframe) -> array`:
|
|
255
|
+
|
|
256
|
+
```python
|
|
257
|
+
report = segment_position_report(
|
|
258
|
+
df, segment_col="department", y_col="salary", benchmark_fit=my_richer_model,
|
|
259
|
+
)
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
`segment_position_report` never inspects what the model uses
|
|
263
|
+
internally -- it only calls `predict()`.
|
|
264
|
+
|
|
265
|
+
**Small segments:** groups with fewer than `MIN_POINTS_FOR_CI` (default
|
|
266
|
+
20) observations still get `observed_median` / `expected_median` /
|
|
267
|
+
`difference`, but `ci_lower` / `ci_upper` are `NaN` and
|
|
268
|
+
`ci_available` is `False` -- a BCa bootstrap confidence interval (which
|
|
269
|
+
relies on a jackknife step) is not attempted for populations that
|
|
270
|
+
small, since it can fail outright or become statistically meaningless.
|
|
271
|
+
For segments at or above the threshold, the interval is a full BCa
|
|
272
|
+
(bias-corrected and accelerated) bootstrap interval, via the same
|
|
273
|
+
`bca_bootstrap_ci_by_index` primitive used elsewhere in the package --
|
|
274
|
+
not a plain percentile bootstrap.
|
|
275
|
+
|
|
276
|
+
## Reporting: residuals, individual deviations, batch runs, and Excel export
|
|
277
|
+
|
|
278
|
+
Four functions built on the same benchmark contract as
|
|
279
|
+
`segment_position_report`, for turning a benchmark into something a
|
|
280
|
+
non-technical audience (or a spreadsheet) can use directly:
|
|
281
|
+
|
|
282
|
+
```python
|
|
283
|
+
from robustkit import (
|
|
284
|
+
residual_summary, deviation_report,
|
|
285
|
+
benchmark_report_suite, export_benchmark_excel,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
# Per-segment residual SHAPE (not just the median difference):
|
|
289
|
+
residual_summary(df, segment_col="job_family", y_col="salary", x_col="age")
|
|
290
|
+
# segment n median_residual mad_residual p10_residual p90_residual
|
|
291
|
+
|
|
292
|
+
# Individuals furthest from the benchmark, sorted by residual --
|
|
293
|
+
# material for a conversation, not an automatic flag:
|
|
294
|
+
deviation_report(
|
|
295
|
+
df, y_col="salary", x_col="age", top_n=50, id_cols=["employee_id"],
|
|
296
|
+
)
|
|
297
|
+
# employee_id actual expected residual
|
|
298
|
+
|
|
299
|
+
# direction="negative" (default, furthest below), "positive" (furthest
|
|
300
|
+
# above), or "two_sided" (largest |residual| either direction).
|
|
301
|
+
|
|
302
|
+
# Run the same benchmark across several grouping columns at once,
|
|
303
|
+
# reusing ONE fitted benchmark so results are directly comparable:
|
|
304
|
+
reports = benchmark_report_suite(
|
|
305
|
+
df, group_columns=["gender", "job_family", "location"], y_col="salary", x_col="age",
|
|
306
|
+
)
|
|
307
|
+
# -> {"gender": DataFrame, "job_family": DataFrame, "location": DataFrame}
|
|
308
|
+
|
|
309
|
+
# Every report as its own sheet in one workbook:
|
|
310
|
+
export_benchmark_excel(reports, "salary_report.xlsx")
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
All four accept the same `benchmark_fit` / `x_col` contract as
|
|
314
|
+
`segment_position_report` (default single-column Huber trend, or any
|
|
315
|
+
custom model exposing `predict(dataframe)`).
|
|
316
|
+
|
|
317
|
+
**Dependency-free Excel export:** `export_benchmark_excel` requires
|
|
318
|
+
`openpyxl` (an optional dependency). In an offline/air-gapped
|
|
319
|
+
environment where installing it isn't possible, use
|
|
320
|
+
`export_benchmark_excel_no_deps` instead -- identical interface,
|
|
321
|
+
implemented with only the Python standard library (writes valid
|
|
322
|
+
`.xlsx` files via `zipfile` and OOXML templating directly, no
|
|
323
|
+
third-party package required):
|
|
324
|
+
|
|
325
|
+
```python
|
|
326
|
+
from robustkit import export_benchmark_excel_no_deps
|
|
327
|
+
|
|
328
|
+
export_benchmark_excel_no_deps(reports, "salary_report.xlsx")
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
Prefer `export_benchmark_excel` when `openpyxl` is available -- it's a
|
|
332
|
+
more complete, better-tested implementation of the Excel format.
|
|
333
|
+
|
|
218
334
|
## Robustness Map
|
|
219
335
|
|
|
220
336
|
Classify features by how much a conclusion about their relationship
|
|
@@ -270,11 +386,196 @@ underlying distribution), the analyst view's confidence band roughly
|
|
|
270
386
|
halves in width, while the publisher view's IQR band stays essentially
|
|
271
387
|
unchanged -- confirmed by the package's own test suite.
|
|
272
388
|
|
|
389
|
+
Both accept an optional `title=None` to override the default title
|
|
390
|
+
(e.g. `plot_analyst_view(x, y, title="Q3 salary review")`), as does
|
|
391
|
+
`plot_quantile_trend`.
|
|
392
|
+
|
|
273
393
|
`dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
|
|
274
394
|
standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
275
395
|
computes both across bins of a continuous x, e.g. to check whether
|
|
276
396
|
dispersion (inequality) grows with age.
|
|
277
397
|
|
|
398
|
+
## Segment awareness: automatic hierarchical grouping + analysis
|
|
399
|
+
|
|
400
|
+
`segment_stability_report` and `segment_benchmark_report` build the
|
|
401
|
+
hierarchical segmentation automatically from a flat, most-specific-
|
|
402
|
+
first list of columns, then run an existing analysis within the
|
|
403
|
+
result -- no separate `hierarchical_segment(...)` + `apply_by_segment(...)`
|
|
404
|
+
preparation step required:
|
|
405
|
+
|
|
406
|
+
```python
|
|
407
|
+
from robustkit import segment_stability_report, segment_benchmark_report
|
|
408
|
+
|
|
409
|
+
# hierarchy built automatically: [JobFamily, Level, OT] -> [Level, OT] -> [OT] -> ALL
|
|
410
|
+
report = segment_stability_report(
|
|
411
|
+
df, x_col="age", y_col="salary",
|
|
412
|
+
segment_cols=["JobFamily", "Level", "OT"], min_size=20,
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
report = segment_benchmark_report(
|
|
416
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"],
|
|
417
|
+
x_col="age", min_size=20, # or benchmark_fit=my_custom_model
|
|
418
|
+
)
|
|
419
|
+
```
|
|
420
|
+
|
|
421
|
+
Both add a `segment_level` column showing which tier of the hierarchy
|
|
422
|
+
each reported segment actually landed on (0 = finest), so a fallback
|
|
423
|
+
to a coarser grouping is visible rather than silent. These are pure
|
|
424
|
+
convenience wrappers -- identical results to building the hierarchy
|
|
425
|
+
by hand with `hierarchical_segment` and calling `apply_by_segment` /
|
|
426
|
+
`segment_position_report` directly.
|
|
427
|
+
|
|
428
|
+
`mad_outlier_report` flags individuals whose residual is an outlier
|
|
429
|
+
relative to their OWN segment's typical spread (MAD), not the whole
|
|
430
|
+
population -- built on the same automatic hierarchical segmentation as
|
|
431
|
+
above, so even someone in a small segment is compared against a
|
|
432
|
+
sensibly-sized reference group rather than an irrelevant one:
|
|
433
|
+
|
|
434
|
+
```python
|
|
435
|
+
from robustkit import mad_outlier_report
|
|
436
|
+
|
|
437
|
+
report = mad_outlier_report(
|
|
438
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age",
|
|
439
|
+
min_size=20, k=3.0, direction="negative", id_cols=["employee_id"],
|
|
440
|
+
)
|
|
441
|
+
# employee_id actual expected residual residual_pct segment segment_level segment_mad threshold flagged
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
`direction`: `"negative"` (default -- flag underperformance relative
|
|
445
|
+
to the benchmark), `"positive"`, or `"two_sided"`. Flagging compares
|
|
446
|
+
each residual to `k` MADs from its *own segment's* median residual
|
|
447
|
+
(not literally zero), so a segment the benchmark is systematically
|
|
448
|
+
biased for doesn't get every member flagged just for that bias.
|
|
449
|
+
Segments with zero MAD (a degenerate case, usually a tiny segment
|
|
450
|
+
where every residual happens to match) are treated as having an
|
|
451
|
+
infinite threshold rather than flagging everyone in them.
|
|
452
|
+
|
|
453
|
+
`export_outlier_pdf` renders one chart per segment -- built from the
|
|
454
|
+
same segmentation and flagging as `mad_outlier_report` -- as a
|
|
455
|
+
one-page-per-segment PDF, for visual verification alongside the
|
|
456
|
+
numeric report:
|
|
457
|
+
|
|
458
|
+
```python
|
|
459
|
+
from robustkit import export_outlier_pdf
|
|
460
|
+
|
|
461
|
+
export_outlier_pdf(
|
|
462
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age",
|
|
463
|
+
path="outliers.pdf", min_size=20, k=3.0, id_cols=["employee_id"],
|
|
464
|
+
)
|
|
465
|
+
```
|
|
466
|
+
|
|
467
|
+
Each page plots every observation in that segment, the benchmark's
|
|
468
|
+
expected values (the exact same values used for flagging, not a
|
|
469
|
+
separately re-fit curve), and flagged outliers marked distinctly.
|
|
470
|
+
Segments with fewer than `min_points_to_plot` (default 5) observations
|
|
471
|
+
are skipped in the PDF -- a chart with a handful of points isn't
|
|
472
|
+
meaningfully verifiable -- but still appear in `mad_outlier_report`'s
|
|
473
|
+
numeric output. The idea: a numeric flag and a visual confirmation are
|
|
474
|
+
two independent checks, and agreement between them is stronger
|
|
475
|
+
evidence than either alone.
|
|
476
|
+
|
|
477
|
+
### Drilldown reports: every hierarchy level at once, without exclusive assignment
|
|
478
|
+
|
|
479
|
+
`segment_stability_report`, `segment_benchmark_report`, and
|
|
480
|
+
`mad_outlier_report` each assign every individual to exactly ONE
|
|
481
|
+
segment (their most specific grouping meeting `min_size`). That
|
|
482
|
+
answers "what is the single most relevant reference population for
|
|
483
|
+
THIS individual?"
|
|
484
|
+
|
|
485
|
+
`segment_benchmark_drilldown_report` and `mad_outlier_drilldown_report`
|
|
486
|
+
answer a different question -- "what does every granularity level look
|
|
487
|
+
like on its own?" -- by reporting EVERY level of the hierarchy
|
|
488
|
+
independently, without exclusive assignment. The same individual can
|
|
489
|
+
appear in multiple rows (e.g. once in a `JobFamily x Level x OT` row,
|
|
490
|
+
and again in the broader `Level x OT` row), whenever both groupings
|
|
491
|
+
independently meet `min_size`:
|
|
492
|
+
|
|
493
|
+
```python
|
|
494
|
+
from robustkit import segment_benchmark_drilldown_report, mad_outlier_drilldown_report
|
|
495
|
+
|
|
496
|
+
segment_benchmark_drilldown_report(
|
|
497
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age", min_size=20,
|
|
498
|
+
)
|
|
499
|
+
mad_outlier_drilldown_report(
|
|
500
|
+
df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age", k=3.0,
|
|
501
|
+
)
|
|
502
|
+
```
|
|
503
|
+
|
|
504
|
+
Both add a `segment_level` column, and the sum of `n` across rows will
|
|
505
|
+
exceed the population size -- that's the expected signature of
|
|
506
|
+
overlap, not a bug. Use the exclusive functions when you need to route
|
|
507
|
+
each individual to one home; use the drilldown functions when you want
|
|
508
|
+
to see every level side by side.
|
|
509
|
+
|
|
510
|
+
## Combined model + spread view
|
|
511
|
+
|
|
512
|
+
`plot_analyst_view` and `plot_publisher_view` each show one thing --
|
|
513
|
+
estimation uncertainty, or population spread -- deliberately kept
|
|
514
|
+
separate. `plot_huber_iqr` shows both together: one or more trend
|
|
515
|
+
curves overlaid with median + IQR error bars, plus an optional
|
|
516
|
+
residual-quality box, matching the combined model-and-spread diagram
|
|
517
|
+
style common in salary/wage analysis reporting:
|
|
518
|
+
|
|
519
|
+
```python
|
|
520
|
+
from robustkit import plot_huber_iqr
|
|
521
|
+
|
|
522
|
+
result = plot_huber_iqr(df["age"], df["salary"], degree=2, bins=15)
|
|
523
|
+
# result["grid"], result["huber"], result["binned"]
|
|
524
|
+
```
|
|
525
|
+
|
|
526
|
+
`show_points` defaults to `False`, consistent with `plot_publisher_view`.
|
|
527
|
+
|
|
528
|
+
**Multiple curves, bootstrap bands, and full style control:**
|
|
529
|
+
|
|
530
|
+
```python
|
|
531
|
+
plot_huber_iqr(
|
|
532
|
+
df["age"], df["salary"],
|
|
533
|
+
methods=("huber", "tukey", "ols", "median_ensemble"), # overlay all four
|
|
534
|
+
show_bootstrap_band=True, bootstrap_levels=(95, 50), # nested confidence bands
|
|
535
|
+
cap_style="manual", # hand-drawn boxplot-style Q1/Q3 "hats" instead of matplotlib's default caps
|
|
536
|
+
residual_box_metric="mdape", # MdAPE + IQR(resid) instead of R^2/MAE/RMSE
|
|
537
|
+
ylim="dynamic", # y-limits set from the data (min*0.95, max*1.05)
|
|
538
|
+
style={
|
|
539
|
+
"huber_line": {"color": "red", "linewidth": 2, "linestyle": "-", "label": "Huber poly(2)"},
|
|
540
|
+
"iqr_color": "black", "cap_width": 0.15,
|
|
541
|
+
},
|
|
542
|
+
)
|
|
543
|
+
```
|
|
544
|
+
|
|
545
|
+
`methods` selects which trend curve(s) to draw (`"tukey"` and
|
|
546
|
+
`"median_ensemble"` -- the pointwise median of Huber/Tukey/OLS --
|
|
547
|
+
require statsmodels). `style` overrides individual colors, line
|
|
548
|
+
widths, and other visual details without needing to touch anything
|
|
549
|
+
else; every new parameter here defaults to the original, simpler
|
|
550
|
+
single-Huber-curve appearance, so existing calls are unaffected.
|
|
551
|
+
|
|
552
|
+
**Two grouping strategies:** `grouping="bin"` (default) uses quantile-
|
|
553
|
+
based binning for stable estimates even in small populations.
|
|
554
|
+
`grouping="unique"` instead groups by each EXACT x value (e.g. every
|
|
555
|
+
individual age in years) -- matching a workbook-style `groupby(x)`
|
|
556
|
+
aggregation -- and, when a given x value has fewer than
|
|
557
|
+
`min_n_for_iqr` (default 5) observations, omits its IQR error bar
|
|
558
|
+
entirely rather than showing an unreliable one:
|
|
559
|
+
|
|
560
|
+
**Caveat, found via validation against a real dataset:** `grouping="unique"`
|
|
561
|
+
only makes sense for x values with natural repetition (e.g. integer
|
|
562
|
+
ages) -- for a genuinely continuous, high-precision measurement (e.g.
|
|
563
|
+
carat weight to several decimal places), nearly every x value is
|
|
564
|
+
unique, so almost nothing meets `min_n_for_iqr` and the result shows
|
|
565
|
+
no IQR bars at all. Use `grouping="bin"` (the default) for
|
|
566
|
+
high-precision continuous x; reserve `grouping="unique"` for x values
|
|
567
|
+
that naturally repeat.
|
|
568
|
+
|
|
569
|
+
```python
|
|
570
|
+
plot_huber_iqr(
|
|
571
|
+
df["age"], df["salary"], grouping="unique", min_n_for_iqr=5,
|
|
572
|
+
)
|
|
573
|
+
```
|
|
574
|
+
|
|
575
|
+
The absence of an error bar at a given age is itself information --
|
|
576
|
+
it signals the sample at that exact value is too small to say
|
|
577
|
+
anything about spread, not just a plotting simplification.
|
|
578
|
+
|
|
278
579
|
## Loading published quantile tables (SCB / JSON-stat)
|
|
279
580
|
|
|
280
581
|
Some statistics agencies (e.g. Statistics Sweden, SCB) publish
|
|
@@ -285,16 +586,16 @@ other national statistics agencies -- avoiding the fragility of
|
|
|
285
586
|
parsing metadata out of column-name strings in a wide CSV export.
|
|
286
587
|
|
|
287
588
|
```python
|
|
288
|
-
from robustkit import
|
|
589
|
+
from robustkit import load_json_stat, plot_quantile_trend, quantile_trend_dispersion
|
|
289
590
|
|
|
290
|
-
df =
|
|
591
|
+
df = load_json_stat("some_scb_table.json")
|
|
291
592
|
|
|
292
593
|
# A real SCB quirk this loader does NOT try to guess automatically:
|
|
293
594
|
# category labels can change meaning over time (e.g. Sweden's oldest
|
|
294
595
|
# working-age bracket was labeled "65-66 år" through 2022 and
|
|
295
596
|
# "65-68 år" from 2023, following a pension-age reform). Merge such
|
|
296
597
|
# cases explicitly:
|
|
297
|
-
df =
|
|
598
|
+
df = load_json_stat(
|
|
298
599
|
"some_scb_table.json",
|
|
299
600
|
rename_categories={"ålder": {"65–68 år": "65–66 år"}},
|
|
300
601
|
)
|
|
@@ -441,6 +742,43 @@ and are converted before being used anywhere in this package.
|
|
|
441
742
|
- **OLS is a reference point, not the enemy.** Comparing robust fits
|
|
442
743
|
against OLS is how you know whether robustness mattered at all.
|
|
443
744
|
|
|
745
|
+
## Naming conventions
|
|
746
|
+
|
|
747
|
+
A few parameter/column names look similar across the package but mean
|
|
748
|
+
different things -- documented here explicitly so the difference reads
|
|
749
|
+
as intentional, not as an inconsistency to "fix":
|
|
750
|
+
|
|
751
|
+
- **`residual` vs. `difference`:** `residual` is an INDIVIDUAL-level
|
|
752
|
+
quantity (`actual - expected` for one row) -- used by
|
|
753
|
+
`mad_outlier_report`, `mad_outlier_drilldown_report`, and
|
|
754
|
+
`deviation_report`. `difference` is a SEGMENT/GROUP-level quantity
|
|
755
|
+
(typically the median residual within a group) -- used by
|
|
756
|
+
`segment_position_report`, `segment_benchmark_report`,
|
|
757
|
+
`segment_benchmark_drilldown_report`, and `benchmark_report_suite`.
|
|
758
|
+
- **`target` vs. `y_col`:** `robustkit.information` uses `target` for
|
|
759
|
+
the column being explained, since it works with arbitrary features
|
|
760
|
+
(not necessarily a continuous regression outcome).
|
|
761
|
+
`robustkit.benchmark`, `robustkit.segment_awareness`, and
|
|
762
|
+
`robustkit.quantiles` use `y_col`, since they specifically model a
|
|
763
|
+
continuous `y` as a function of `x`.
|
|
764
|
+
- **`segment_cols` vs. `group_columns`:** `segment_cols` (throughout
|
|
765
|
+
`robustkit.segment_awareness`) is an ORDERED, most-specific-first
|
|
766
|
+
list used to build a fallback HIERARCHY (see `hierarchical_segment`).
|
|
767
|
+
`group_columns` (`benchmark_report_suite`) is a FLAT list of
|
|
768
|
+
independent groupings, run separately with no hierarchy or fallback
|
|
769
|
+
between them. Different structure, different name on purpose.
|
|
770
|
+
- **`min_size` vs. `min_points` vs. `min_stratum_size` vs.
|
|
771
|
+
`min_group_size`:** all mean "minimum group size," but at different
|
|
772
|
+
stages: `min_size` (`hierarchical_segment` and everything built on
|
|
773
|
+
it) gates whether a hierarchy LEVEL gets created at all;
|
|
774
|
+
`min_points` (`apply_by_segment`) gates whether an already-built
|
|
775
|
+
segment gets ANALYZED; `min_stratum_size`
|
|
776
|
+
(`conditional_mutual_information`) and `min_group_size`
|
|
777
|
+
(`communication_score`, `rank_by_communication`) are specific to
|
|
778
|
+
those `robustkit.information` calculations. Kept separate rather
|
|
779
|
+
than unified to one name, since collapsing them would obscure which
|
|
780
|
+
stage of a pipeline each threshold actually applies to.
|
|
781
|
+
|
|
444
782
|
## License
|
|
445
783
|
|
|
446
784
|
MIT -- see [LICENSE](LICENSE).
|