robustkit 0.4.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. robustkit-0.6.0/PKG-INFO +784 -0
  2. robustkit-0.6.0/README.md +742 -0
  3. {robustkit-0.4.0 → robustkit-0.6.0}/pyproject.toml +2 -1
  4. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/__init__.py +50 -4
  5. robustkit-0.6.0/robustkit/benchmark/global_model.py +162 -0
  6. robustkit-0.6.0/robustkit/benchmark/reporting.py +317 -0
  7. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/benchmark/robustness_map.py +19 -1
  8. robustkit-0.6.0/robustkit/core/consistency.py +90 -0
  9. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/core/diagnostics.py +2 -2
  10. robustkit-0.6.0/robustkit/core/goodness_of_fit.py +77 -0
  11. robustkit-0.6.0/robustkit/core/trend.py +136 -0
  12. robustkit-0.6.0/robustkit/core/uncertainty.py +171 -0
  13. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/mutual_info.py +13 -1
  14. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/visualization.py +1 -1
  15. robustkit-0.6.0/robustkit/quantiles/io.py +104 -0
  16. robustkit-0.6.0/robustkit/quantiles/reconstruct.py +272 -0
  17. robustkit-0.6.0/robustkit/quantiles/trend.py +86 -0
  18. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/report/visualize_analyst.py +7 -2
  19. robustkit-0.6.0/robustkit/report/visualize_huber_iqr.py +313 -0
  20. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/report/visualize_publisher.py +3 -2
  21. robustkit-0.6.0/robustkit/segment_awareness/__init__.py +17 -0
  22. robustkit-0.6.0/robustkit/segment_awareness/reports.py +470 -0
  23. robustkit-0.6.0/robustkit/segmentation/__init__.py +0 -0
  24. robustkit-0.6.0/robustkit.egg-info/PKG-INFO +784 -0
  25. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/SOURCES.txt +25 -1
  26. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/requires.txt +3 -0
  27. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_benchmark.py +23 -4
  28. robustkit-0.6.0/tests/test_benchmark_gaps.py +159 -0
  29. robustkit-0.6.0/tests/test_benchmark_reporting.py +195 -0
  30. robustkit-0.6.0/tests/test_bugfixes_faseA.py +102 -0
  31. robustkit-0.6.0/tests/test_export_no_deps.py +119 -0
  32. robustkit-0.6.0/tests/test_export_outlier_pdf.py +116 -0
  33. robustkit-0.6.0/tests/test_huber_iqr.py +197 -0
  34. robustkit-0.6.0/tests/test_mad_outlier_report.py +117 -0
  35. robustkit-0.6.0/tests/test_n_boot_auto_propagation.py +86 -0
  36. robustkit-0.6.0/tests/test_quantiles_io_trend.py +134 -0
  37. robustkit-0.6.0/tests/test_quantiles_reconstruct.py +90 -0
  38. robustkit-0.6.0/tests/test_quantiles_reconstruct_mean_only.py +78 -0
  39. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_report.py +20 -0
  40. robustkit-0.6.0/tests/test_segment_awareness.py +124 -0
  41. robustkit-0.6.0/tests/test_segment_consistency.py +72 -0
  42. robustkit-0.6.0/tests/test_segment_drilldown.py +84 -0
  43. robustkit-0.6.0/tests/test_trend_extras.py +92 -0
  44. robustkit-0.4.0/PKG-INFO +0 -296
  45. robustkit-0.4.0/README.md +0 -256
  46. robustkit-0.4.0/robustkit/benchmark/global_model.py +0 -74
  47. robustkit-0.4.0/robustkit/core/consistency.py +0 -33
  48. robustkit-0.4.0/robustkit/core/trend.py +0 -86
  49. robustkit-0.4.0/robustkit/core/uncertainty.py +0 -83
  50. robustkit-0.4.0/robustkit.egg-info/PKG-INFO +0 -296
  51. {robustkit-0.4.0 → robustkit-0.6.0}/LICENSE +0 -0
  52. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/benchmark/__init__.py +0 -0
  53. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/common/__init__.py +0 -0
  54. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/common/quadrants.py +0 -0
  55. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/core/__init__.py +0 -0
  56. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/core/stability.py +0 -0
  57. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/__init__.py +0 -0
  58. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/communication.py +0 -0
  59. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/conditional_mi.py +0 -0
  60. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/entropy.py +0 -0
  61. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/pairs.py +0 -0
  62. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/profile.py +0 -0
  63. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/quadrants.py +0 -0
  64. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/information/utils.py +0 -0
  65. {robustkit-0.4.0/robustkit/report → robustkit-0.6.0/robustkit/quantiles}/__init__.py +0 -0
  66. {robustkit-0.4.0/robustkit/segmentation → robustkit-0.6.0/robustkit/report}/__init__.py +0 -0
  67. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/report/dispersion.py +0 -0
  68. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/segmentation/apply.py +0 -0
  69. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit/segmentation/hierarchy.py +0 -0
  70. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/dependency_links.txt +0 -0
  71. {robustkit-0.4.0 → robustkit-0.6.0}/robustkit.egg-info/top_level.txt +0 -0
  72. {robustkit-0.4.0 → robustkit-0.6.0}/setup.cfg +0 -0
  73. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_common_quadrants.py +0 -0
  74. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_core.py +0 -0
  75. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_information.py +0 -0
  76. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_information_pairs.py +0 -0
  77. {robustkit-0.4.0 → robustkit-0.6.0}/tests/test_segmentation.py +0 -0
@@ -0,0 +1,784 @@
1
+ Metadata-Version: 2.4
2
+ Name: robustkit
3
+ Version: 0.6.0
4
+ Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
5
+ Author: Mikael Lundqvist
6
+ License: MIT License
7
+
8
+ Copyright (c) 2026 Mikael Lundqvist
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Requires-Python: >=3.10
29
+ Description-Content-Type: text/markdown
30
+ License-File: LICENSE
31
+ Requires-Dist: numpy>=1.24
32
+ Requires-Dist: pandas>=2.0
33
+ Requires-Dist: scikit-learn>=1.3
34
+ Requires-Dist: statsmodels>=0.14
35
+ Requires-Dist: scipy>=1.10
36
+ Requires-Dist: matplotlib>=3.7
37
+ Provides-Extra: dev
38
+ Requires-Dist: pytest>=7.0; extra == "dev"
39
+ Provides-Extra: excel
40
+ Requires-Dist: openpyxl>=3.1; extra == "excel"
41
+ Dynamic: license-file
42
+
43
+ # robustkit
44
+
45
+ > ⚠️ **Under active development.** This is an early placeholder release
46
+ > to claim the package name on PyPI. The API is incomplete and may
47
+ > change without notice. Not yet recommended for production use.
48
+
49
+ Practical tools for robust analysis of a single continuous relationship:
50
+ y as a function of one continuous x.
51
+
52
+ The guiding idea: **a conclusion that survives multiple fitting methods
53
+ is more trustworthy than one that only holds under a single model.**
54
+ `robustkit` makes it easy to compare Huber, Tukey biweight, and OLS
55
+ fits side by side, identify and quantify the influence of individual
56
+ observations, and get honest, bias-corrected uncertainty estimates.
57
+
58
+ ## Status
59
+
60
+ `robustkit.core` (trend fitting, stability, diagnostics, uncertainty,
61
+ consistency checks), `robustkit.segmentation` (hierarchical grouping,
62
+ per-segment analysis), `robustkit.information` (mutual-information
63
+ feature ranking, quadrant classification, pairwise redundancy/synergy
64
+ scoring), `robustkit.benchmark` (global-trend segment comparison,
65
+ model-agnostic custom benchmarks, residual/deviation reporting, Excel
66
+ export, Robustness Map), `robustkit.report` (analyst vs. publisher
67
+ views, dispersion measures, combined Huber+IQR view),
68
+ `robustkit.quantiles` (generic JSON-stat loading, published-quantile-
69
+ trend visualization, and lognormal-calibrated reconstruction of
70
+ individual-level data from aggregated summaries), and
71
+ `robustkit.segment_awareness` (automatic hierarchical segmentation +
72
+ analysis, no manual hierarchy construction required) are stable and
73
+ tested.
74
+
75
+ **Recent fixes from real-dataset validation:**
76
+ - `rank_features`/`quadrant_report`/`rank_communicative_pairs` no
77
+ longer crash on pandas `Categorical` columns containing missing
78
+ values (found via OpenML's Boston Housing dataset).
79
+ - `bootstrap_band` (and `plot_analyst_view`, which uses it) now
80
+ defaults to `n_boot="auto"`, scaling iterations down for large
81
+ datasets since each iteration refits a full Huber model -- found to
82
+ become impractically slow at n_boot=200 on a ~54,000-row dataset.
83
+ Pass an explicit integer to opt out and always use exactly that many
84
+ iterations.
85
+
86
+ **Note on `information_efficiency`:** values can exceed 1.0 for
87
+ continuous features. `mutual_information` is estimated on the
88
+ full-resolution continuous values, while `entropy_bits` is computed on
89
+ a binned version of the same feature (since `entropy()` expects
90
+ categorical input). Binning discards information, so `entropy_bits` is
91
+ a lower bound on the feature's true entropy -- an efficiency above 1.0
92
+ signals that the feature carries more usable information than a coarse
93
+ categorical summary of it would capture. This is expected behavior,
94
+ not a bug.
95
+
96
+ ## Installation
97
+
98
+ ```bash
99
+ git clone https://github.com/<your-username>/robustkit.git
100
+ cd robustkit
101
+ pip install -e ".[dev]"
102
+ ```
103
+
104
+ ## Quickstart
105
+
106
+ ```python
107
+ import numpy as np
108
+ from robustkit import (
109
+ fit_huber_trend, fit_tukey_trend, predict_trend,
110
+ model_stability_pct, cooks_diagnostic, cook_impact,
111
+ bootstrap_band, bca_bootstrap_ci,
112
+ )
113
+
114
+ # x: a single continuous predictor, y: a single continuous outcome
115
+ x = np.random.default_rng(0).uniform(20, 60, 200)
116
+ y = 1000 + 50 * x - 0.4 * x**2 + np.random.default_rng(1).normal(0, 500, 200)
117
+
118
+ fit = fit_huber_trend(x, y, degree=2)
119
+ y_pred = predict_trend(fit, x_new=[30, 40, 50])
120
+
121
+ stability = model_stability_pct(x, y)
122
+ print("Median % spread between Huber/Tukey/OLS:", stability["median_pct_diff"])
123
+
124
+ diag = cooks_diagnostic(x, y)
125
+ impact = cook_impact(x, y, diag["flagged_indices"])
126
+ print("Median % change in curve if flagged points removed:", impact["median_pct_change"])
127
+
128
+ band = bootstrap_band(x, y)
129
+ ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
130
+ ```
131
+
132
+ See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
133
+
134
+ ## Trend growth rate and goodness of fit
135
+
136
+ ```python
137
+ from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
138
+
139
+ fit = fit_huber_trend(df["age"], df["salary"])
140
+
141
+ # Rate of change of the trend itself (e.g. "salary growth per year of
142
+ # age"), not just its level
143
+ rates = trend_derivative(fit, x=[30, 40, 50])
144
+
145
+ # How well does this fit actually explain the variation in y?
146
+ goodness_of_fit(df["age"], df["salary"], degree=2)
147
+
148
+ # Don't assume a quadratic trend is always the right choice -- check
149
+ # empirically whether a higher degree captures meaningfully more
150
+ compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
151
+ ```
152
+
153
+ **Note:** x is standardized internally before building polynomial
154
+ features (both here and throughout `robustkit.core`), since raw
155
+ polynomial features become numerically unstable at higher degrees for
156
+ realistic x scales (e.g. age^5 vastly outscales age^1). This is
157
+ transparent to callers -- `predict_trend` and `trend_derivative` still
158
+ take and return values in the original x scale.
159
+
160
+ ## Segmentation
161
+
162
+ Run any `robustkit.core` analysis independently across subgroups of a
163
+ larger dataset, with automatic fallback to coarser groupings when a
164
+ finer one is too small to analyze reliably:
165
+
166
+ ```python
167
+ from robustkit import hierarchical_segment, apply_by_segment, model_stability_pct
168
+
169
+ hierarchy = [["department", "level", "status"], ["level", "status"], ["status"]]
170
+ segmented = hierarchical_segment(df, hierarchy, min_size=20)
171
+
172
+ report = apply_by_segment(
173
+ segmented, segment_col="segment_id", x_col="age", y_col="value",
174
+ analysis_fn=model_stability_pct,
175
+ )
176
+ ```
177
+
178
+ `apply_by_segment` works with any function shaped like
179
+ `analysis_fn(x, y, **kwargs) -> dict` -- built-in ones
180
+ (`model_stability_pct`, `cook_impact`, `bca_bootstrap_ci`, ...) or your
181
+ own. Only scalar values in the returned dict end up in the report
182
+ table; segments below `min_points` are skipped rather than causing an
183
+ error.
184
+
185
+ **Sanity-checking a segmentation before trusting it:**
186
+ `segment_consistency_report` runs a small battery of checks per
187
+ segment -- does it meet the recommended minimum size, and does fitting
188
+ a Huber trend on it use every row (robust methods don't need outliers
189
+ pre-removed, so a silently dropped row usually means a missing x/y
190
+ value slipped through, not intentional filtering):
191
+
192
+ ```python
193
+ from robustkit import segment_consistency_report
194
+
195
+ segment_consistency_report(df, segment_col="department", x_col="age", y_col="salary", min_size=20)
196
+ # segment n_total n_valid_xy n_dropped_missing_xy size_ok fit_ok fit_error
197
+ ```
198
+
199
+ ## Feature ranking (information)
200
+
201
+ Rank features by mutual information with a target, normalized by each
202
+ feature's own entropy, and classify them into four quadrants:
203
+
204
+ ```python
205
+ from robustkit import rank_features, quadrant_report, plot_feature_space
206
+
207
+ ranking = rank_features(df, target="value")
208
+ report = quadrant_report(df, target="value") # adds a `quadrant` column
209
+ plot_feature_space(df, target="value") # same quadrants, visualized
210
+ ```
211
+
212
+ `quadrant_report` and `plot_feature_space` always agree on quadrant
213
+ assignment -- both route through the same thresholding logic.
214
+
215
+ **Caveat:** default thresholds are the *median* mutual information /
216
+ efficiency across the ranked features. With only a handful of
217
+ features, this can put a genuinely weak feature in the same "high"
218
+ half as a strong one, since roughly half of any list sits above its
219
+ own median regardless of how large the actual gap is. Median
220
+ thresholding becomes meaningful with a reasonably large feature set;
221
+ for a handful of candidates, read the raw `mutual_information` /
222
+ `information_efficiency` values directly rather than relying on the
223
+ quadrant label alone.
224
+
225
+ See `examples/information_tutorial.py` for a complete walkthrough.
226
+
227
+ ## Benchmarking against a global trend
228
+
229
+ Compare each segment's observed outcome against what a benchmark model
230
+ predicts, with bootstrap uncertainty on the difference -- answers
231
+ "which groups deviate from the overall trend, and by how much?" rather
232
+ than "how does the trend look overall?":
233
+
234
+ ```python
235
+ from robustkit import segment_position_report
236
+
237
+ # Default: a single global Huber trend on one continuous x
238
+ report = segment_position_report(
239
+ df, segment_col="department", x_col="age", y_col="salary",
240
+ )
241
+ # segment n observed_median expected_median difference ci_lower ci_upper ci_available
242
+ # Finance 176 48339.70 47799.82 539.88 202.15 1031.01 True
243
+ # HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36 True
244
+ # IT 250 47226.50 47126.91 99.59 -117.56 510.81 True
245
+ ```
246
+
247
+ A segment's confidence interval crossing zero means no clear deviation
248
+ from the benchmark; HR and Finance above don't cross zero, IT does.
249
+
250
+ **Custom, model-agnostic benchmarks:** the default single-column Huber
251
+ trend can be replaced with any richer model -- e.g. one using age,
252
+ age-squared, job level, overtime status, and a reference cluster
253
+ together, rather than a single x. Provide any object exposing
254
+ `predict(dataframe) -> array`:
255
+
256
+ ```python
257
+ report = segment_position_report(
258
+ df, segment_col="department", y_col="salary", benchmark_fit=my_richer_model,
259
+ )
260
+ ```
261
+
262
+ `segment_position_report` never inspects what the model uses
263
+ internally -- it only calls `predict()`.
264
+
265
+ **Small segments:** groups with fewer than `MIN_POINTS_FOR_CI` (default
266
+ 20) observations still get `observed_median` / `expected_median` /
267
+ `difference`, but `ci_lower` / `ci_upper` are `NaN` and
268
+ `ci_available` is `False` -- a BCa bootstrap confidence interval (which
269
+ relies on a jackknife step) is not attempted for populations that
270
+ small, since it can fail outright or become statistically meaningless.
271
+ For segments at or above the threshold, the interval is a full BCa
272
+ (bias-corrected and accelerated) bootstrap interval, via the same
273
+ `bca_bootstrap_ci_by_index` primitive used elsewhere in the package --
274
+ not a plain percentile bootstrap.
275
+
276
+ ## Reporting: residuals, individual deviations, batch runs, and Excel export
277
+
278
+ Four functions built on the same benchmark contract as
279
+ `segment_position_report`, for turning a benchmark into something a
280
+ non-technical audience (or a spreadsheet) can use directly:
281
+
282
+ ```python
283
+ from robustkit import (
284
+ residual_summary, deviation_report,
285
+ benchmark_report_suite, export_benchmark_excel,
286
+ )
287
+
288
+ # Per-segment residual SHAPE (not just the median difference):
289
+ residual_summary(df, segment_col="job_family", y_col="salary", x_col="age")
290
+ # segment n median_residual mad_residual p10_residual p90_residual
291
+
292
+ # Individuals furthest from the benchmark, sorted by residual --
293
+ # material for a conversation, not an automatic flag:
294
+ deviation_report(
295
+ df, y_col="salary", x_col="age", top_n=50, id_cols=["employee_id"],
296
+ )
297
+ # employee_id actual expected residual
298
+
299
+ # direction="negative" (default, furthest below), "positive" (furthest
300
+ # above), or "two_sided" (largest |residual| either direction).
301
+
302
+ # Run the same benchmark across several grouping columns at once,
303
+ # reusing ONE fitted benchmark so results are directly comparable:
304
+ reports = benchmark_report_suite(
305
+ df, group_columns=["gender", "job_family", "location"], y_col="salary", x_col="age",
306
+ )
307
+ # -> {"gender": DataFrame, "job_family": DataFrame, "location": DataFrame}
308
+
309
+ # Every report as its own sheet in one workbook:
310
+ export_benchmark_excel(reports, "salary_report.xlsx")
311
+ ```
312
+
313
+ All four accept the same `benchmark_fit` / `x_col` contract as
314
+ `segment_position_report` (default single-column Huber trend, or any
315
+ custom model exposing `predict(dataframe)`).
316
+
317
+ **Dependency-free Excel export:** `export_benchmark_excel` requires
318
+ `openpyxl` (an optional dependency). In an offline/air-gapped
319
+ environment where installing it isn't possible, use
320
+ `export_benchmark_excel_no_deps` instead -- identical interface,
321
+ implemented with only the Python standard library (writes valid
322
+ `.xlsx` files via `zipfile` and OOXML templating directly, no
323
+ third-party package required):
324
+
325
+ ```python
326
+ from robustkit import export_benchmark_excel_no_deps
327
+
328
+ export_benchmark_excel_no_deps(reports, "salary_report.xlsx")
329
+ ```
330
+
331
+ Prefer `export_benchmark_excel` when `openpyxl` is available -- it's a
332
+ more complete, better-tested implementation of the Excel format.
333
+
334
+ ## Robustness Map
335
+
336
+ Classify features by how much a conclusion about their relationship
337
+ with the target depends on (a) fitting method choice and (b) specific
338
+ influential observations -- two genuinely different failure modes that
339
+ a single diagnostic can miss:
340
+
341
+ ```python
342
+ from robustkit import feature_robustness_report, plot_feature_robustness
343
+
344
+ report = feature_robustness_report(df, target="value")
345
+ # feature stability_pct cook_impact_pct quadrant
346
+ # CRIM 8.9 17.1 fragile
347
+ # AGE 16.8 15.5 fragile
348
+ # RM 4.9 0.1 robust
349
+ # TAX 22.2 1.4 structural_sensitivity
350
+
351
+ plot_feature_robustness(report=report)
352
+ ```
353
+
354
+ Four quadrants: **robust** (low spread, low impact), **structural
355
+ sensitivity** (sensitive to fitting method, not to specific points),
356
+ **data sensitive** (a few points drive the conclusion, method choice
357
+ barely matters), **fragile** (both -- least trustworthy).
358
+
359
+ `quadrant_report`/`plot_feature_space` (information) and
360
+ `feature_robustness_report`/`plot_feature_robustness` (benchmark) both
361
+ route through the same shared classifier, `robustkit.classify_quadrants`
362
+ -- any future quadrant-based analysis in this package will too.
363
+
364
+ ## Analyst view vs. publisher view
365
+
366
+ Two visualizations that look superficially similar but answer
367
+ genuinely different questions:
368
+
369
+ ```python
370
+ from robustkit import plot_analyst_view, plot_publisher_view, dispersion_ratio, iqr
371
+
372
+ # "How confident are we in the trend estimate?" -- a bootstrap
373
+ # confidence band that SHRINKS as sample size grows.
374
+ plot_analyst_view(df["age"], df["salary"])
375
+
376
+ # "How spread out are actual values in the population?" -- a median +
377
+ # IQR band that does NOT shrink with more data, since it reflects
378
+ # real dispersion, not estimation uncertainty. show_points defaults to
379
+ # False, since this view is meant for publishing potentially sensitive
380
+ # data (e.g. individual salaries) without exposing raw points.
381
+ plot_publisher_view(df["age"], df["salary"])
382
+ ```
383
+
384
+ This distinction matters in practice: with 20x more data (same
385
+ underlying distribution), the analyst view's confidence band roughly
386
+ halves in width, while the publisher view's IQR band stays essentially
387
+ unchanged -- confirmed by the package's own test suite.
388
+
389
+ Both accept an optional `title=None` to override the default title
390
+ (e.g. `plot_analyst_view(x, y, title="Q3 salary review")`), as does
391
+ `plot_quantile_trend`.
392
+
393
+ `dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
394
+ standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
395
+ computes both across bins of a continuous x, e.g. to check whether
396
+ dispersion (inequality) grows with age.
397
+
398
+ ## Segment awareness: automatic hierarchical grouping + analysis
399
+
400
+ `segment_stability_report` and `segment_benchmark_report` build the
401
+ hierarchical segmentation automatically from a flat, most-specific-
402
+ first list of columns, then run an existing analysis within the
403
+ result -- no separate `hierarchical_segment(...)` + `apply_by_segment(...)`
404
+ preparation step required:
405
+
406
+ ```python
407
+ from robustkit import segment_stability_report, segment_benchmark_report
408
+
409
+ # hierarchy built automatically: [JobFamily, Level, OT] -> [Level, OT] -> [OT] -> ALL
410
+ report = segment_stability_report(
411
+ df, x_col="age", y_col="salary",
412
+ segment_cols=["JobFamily", "Level", "OT"], min_size=20,
413
+ )
414
+
415
+ report = segment_benchmark_report(
416
+ df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"],
417
+ x_col="age", min_size=20, # or benchmark_fit=my_custom_model
418
+ )
419
+ ```
420
+
421
+ Both add a `segment_level` column showing which tier of the hierarchy
422
+ each reported segment actually landed on (0 = finest), so a fallback
423
+ to a coarser grouping is visible rather than silent. These are pure
424
+ convenience wrappers -- identical results to building the hierarchy
425
+ by hand with `hierarchical_segment` and calling `apply_by_segment` /
426
+ `segment_position_report` directly.
427
+
428
+ `mad_outlier_report` flags individuals whose residual is an outlier
429
+ relative to their OWN segment's typical spread (MAD), not the whole
430
+ population -- built on the same automatic hierarchical segmentation as
431
+ above, so even someone in a small segment is compared against a
432
+ sensibly-sized reference group rather than an irrelevant one:
433
+
434
+ ```python
435
+ from robustkit import mad_outlier_report
436
+
437
+ report = mad_outlier_report(
438
+ df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age",
439
+ min_size=20, k=3.0, direction="negative", id_cols=["employee_id"],
440
+ )
441
+ # employee_id actual expected residual residual_pct segment segment_level segment_mad threshold flagged
442
+ ```
443
+
444
+ `direction`: `"negative"` (default -- flag underperformance relative
445
+ to the benchmark), `"positive"`, or `"two_sided"`. Flagging compares
446
+ each residual to `k` MADs from its *own segment's* median residual
447
+ (not literally zero), so a segment the benchmark is systematically
448
+ biased for doesn't get every member flagged just for that bias.
449
+ Segments with zero MAD (a degenerate case, usually a tiny segment
450
+ where every residual happens to match) are treated as having an
451
+ infinite threshold rather than flagging everyone in them.
452
+
453
+ `export_outlier_pdf` renders one chart per segment -- built from the
454
+ same segmentation and flagging as `mad_outlier_report` -- as a
455
+ one-page-per-segment PDF, for visual verification alongside the
456
+ numeric report:
457
+
458
+ ```python
459
+ from robustkit import export_outlier_pdf
460
+
461
+ export_outlier_pdf(
462
+ df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age",
463
+ path="outliers.pdf", min_size=20, k=3.0, id_cols=["employee_id"],
464
+ )
465
+ ```
466
+
467
+ Each page plots every observation in that segment, the benchmark's
468
+ expected values (the exact same values used for flagging, not a
469
+ separately re-fit curve), and flagged outliers marked distinctly.
470
+ Segments with fewer than `min_points_to_plot` (default 5) observations
471
+ are skipped in the PDF -- a chart with a handful of points isn't
472
+ meaningfully verifiable -- but still appear in `mad_outlier_report`'s
473
+ numeric output. The idea: a numeric flag and a visual confirmation are
474
+ two independent checks, and agreement between them is stronger
475
+ evidence than either alone.
476
+
477
+ ### Drilldown reports: every hierarchy level at once, without exclusive assignment
478
+
479
+ `segment_stability_report`, `segment_benchmark_report`, and
480
+ `mad_outlier_report` each assign every individual to exactly ONE
481
+ segment (their most specific grouping meeting `min_size`). That
482
+ answers "what is the single most relevant reference population for
483
+ THIS individual?"
484
+
485
+ `segment_benchmark_drilldown_report` and `mad_outlier_drilldown_report`
486
+ answer a different question -- "what does every granularity level look
487
+ like on its own?" -- by reporting EVERY level of the hierarchy
488
+ independently, without exclusive assignment. The same individual can
489
+ appear in multiple rows (e.g. once in a `JobFamily x Level x OT` row,
490
+ and again in the broader `Level x OT` row), whenever both groupings
491
+ independently meet `min_size`:
492
+
493
+ ```python
494
+ from robustkit import segment_benchmark_drilldown_report, mad_outlier_drilldown_report
495
+
496
+ segment_benchmark_drilldown_report(
497
+ df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age", min_size=20,
498
+ )
499
+ mad_outlier_drilldown_report(
500
+ df, y_col="salary", segment_cols=["JobFamily", "Level", "OT"], x_col="age", k=3.0,
501
+ )
502
+ ```
503
+
504
+ Both add a `segment_level` column, and the sum of `n` across rows will
505
+ exceed the population size -- that's the expected signature of
506
+ overlap, not a bug. Use the exclusive functions when you need to route
507
+ each individual to one home; use the drilldown functions when you want
508
+ to see every level side by side.
509
+
510
+ ## Combined model + spread view
511
+
512
+ `plot_analyst_view` and `plot_publisher_view` each show one thing --
513
+ estimation uncertainty, or population spread -- deliberately kept
514
+ separate. `plot_huber_iqr` shows both together: one or more trend
515
+ curves overlaid with median + IQR error bars, plus an optional
516
+ residual-quality box, matching the combined model-and-spread diagram
517
+ style common in salary/wage analysis reporting:
518
+
519
+ ```python
520
+ from robustkit import plot_huber_iqr
521
+
522
+ result = plot_huber_iqr(df["age"], df["salary"], degree=2, bins=15)
523
+ # result["grid"], result["huber"], result["binned"]
524
+ ```
525
+
526
+ `show_points` defaults to `False`, consistent with `plot_publisher_view`.
527
+
528
+ **Multiple curves, bootstrap bands, and full style control:**
529
+
530
+ ```python
531
+ plot_huber_iqr(
532
+ df["age"], df["salary"],
533
+ methods=("huber", "tukey", "ols", "median_ensemble"), # overlay all four
534
+ show_bootstrap_band=True, bootstrap_levels=(95, 50), # nested confidence bands
535
+ cap_style="manual", # hand-drawn boxplot-style Q1/Q3 "hats" instead of matplotlib's default caps
536
+ residual_box_metric="mdape", # MdAPE + IQR(resid) instead of R^2/MAE/RMSE
537
+ ylim="dynamic", # y-limits set from the data (min*0.95, max*1.05)
538
+ style={
539
+ "huber_line": {"color": "red", "linewidth": 2, "linestyle": "-", "label": "Huber poly(2)"},
540
+ "iqr_color": "black", "cap_width": 0.15,
541
+ },
542
+ )
543
+ ```
544
+
545
+ `methods` selects which trend curve(s) to draw (`"tukey"` and
546
+ `"median_ensemble"` -- the pointwise median of Huber/Tukey/OLS --
547
+ require statsmodels). `style` overrides individual colors, line
548
+ widths, and other visual details without needing to touch anything
549
+ else; every new parameter here defaults to the original, simpler
550
+ single-Huber-curve appearance, so existing calls are unaffected.
551
+
552
+ **Two grouping strategies:** `grouping="bin"` (default) uses quantile-
553
+ based binning for stable estimates even in small populations.
554
+ `grouping="unique"` instead groups by each EXACT x value (e.g. every
555
+ individual age in years) -- matching a workbook-style `groupby(x)`
556
+ aggregation -- and, when a given x value has fewer than
557
+ `min_n_for_iqr` (default 5) observations, omits its IQR error bar
558
+ entirely rather than showing an unreliable one:
559
+
560
+ **Caveat, found via validation against a real dataset:** `grouping="unique"`
561
+ only makes sense for x values with natural repetition (e.g. integer
562
+ ages) -- for a genuinely continuous, high-precision measurement (e.g.
563
+ carat weight to several decimal places), nearly every x value is
564
+ unique, so almost nothing meets `min_n_for_iqr` and the result shows
565
+ no IQR bars at all. Use `grouping="bin"` (the default) for
566
+ high-precision continuous x; reserve `grouping="unique"` for x values
567
+ that naturally repeat.
568
+
569
+ ```python
570
+ plot_huber_iqr(
571
+ df["age"], df["salary"], grouping="unique", min_n_for_iqr=5,
572
+ )
573
+ ```
574
+
575
+ The absence of an error bar at a given age is itself information --
576
+ it signals the sample at that exact value is too small to say
577
+ anything about spread, not just a plotting simplification.
578
+
579
+ ## Loading published quantile tables (SCB / JSON-stat)
580
+
581
+ Some statistics agencies (e.g. Statistics Sweden, SCB) publish
582
+ quantiles (Q1/median/Q3) directly, with no individual-level data
583
+ available at all. `robustkit.quantiles` loads these tables generically
584
+ via JSON-stat, a standardized dimensional-data format used by SCB and
585
+ other national statistics agencies -- avoiding the fragility of
586
+ parsing metadata out of column-name strings in a wide CSV export.
587
+
588
+ ```python
589
+ from robustkit import load_json_stat, plot_quantile_trend, quantile_trend_dispersion
590
+
591
+ df = load_json_stat("some_scb_table.json")
592
+
593
+ # A real SCB quirk this loader does NOT try to guess automatically:
594
+ # category labels can change meaning over time (e.g. Sweden's oldest
595
+ # working-age bracket was labeled "65-66 år" through 2022 and
596
+ # "65-68 år" from 2023, following a pension-age reform). Merge such
597
+ # cases explicitly:
598
+ df = load_json_stat(
599
+ "some_scb_table.json",
600
+ rename_categories={"ålder": {"65–68 år": "65–66 år"}},
601
+ )
602
+
603
+ # Once reshaped to a wide table with q1/median/q3 columns:
604
+ plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
605
+ quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
606
+ ```
607
+
608
+ This is the "quantiles are already given" case. A complementary case
609
+ -- reconstructing approximate individual-level data from aggregated
610
+ group means, for when only summary statistics (not quantiles) are
611
+ available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
612
+
613
+ See `examples/quantiles_tutorial.py` for a complete walkthrough.
614
+
615
+ ## Reconstructing individual-level data from aggregated summaries
616
+
617
+ For the complementary case -- only aggregated group summaries (n,
618
+ Q1, median, Q3) are available, not the quantile trend itself as the
619
+ final answer, and you want to run `robustkit.core` analyses as if
620
+ individual data existed:
621
+
622
+ ```python
623
+ from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
624
+
625
+ # One row per group (e.g. year), with n/q1/median/q3 columns
626
+ synthetic = expand_aggregated_table(
627
+ summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
628
+ group_cols=["year"], value_name="salary",
629
+ )
630
+
631
+ # Now usable exactly like real individual-level data:
632
+ fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
633
+ ```
634
+
635
+ Method: a lognormal distribution is calibrated (via the IQR) to match
636
+ each group's reported Q1/median/Q3, then `n` synthetic values are
637
+ drawn from it. Validated end-to-end against real published SCB salary
638
+ data: a Huber trend fitted on reconstructed pseudo-individual data
639
+ tracked the true published median trend within 2% across 12 years.
640
+
641
+ **Note on what this recovers:** because a Huber (or Tukey) fit on
642
+ right-skewed reconstructed data tracks something close to the
643
+ *median* trend it was calibrated against -- not the arithmetic mean --
644
+ this is consistent with, not a limitation of, the reconstruction
645
+ method. To target the mean instead, fit on `log(value)` and
646
+ exponentiate predictions back, which approximates the geometric mean.
647
+
648
+ Always check `check_reconstruction_quality()` before trusting a
649
+ reconstruction: real Q1/median/Q3 triples aren't always perfectly
650
+ consistent with a pure lognormal shape.
651
+
652
+ **Warning -- unbounded tail at large n:** a lognormal has no natural
653
+ upper limit, and its expected maximum grows with n. Reconstructing at
654
+ the TRUE group size from a national table (SCB salary tables can
655
+ report n in the hundreds of thousands to millions) can produce
656
+ implausibly extreme tail values -- real salaries have practical
657
+ ceilings a pure lognormal doesn't know about. This package's own
658
+ examples and tests deliberately scale n down to a few thousand for
659
+ demonstration; calibration quality (matching Q1/median/Q3) doesn't
660
+ depend on reproducing the true population size, but tail plausibility
661
+ does. No clipping is applied automatically.
662
+
663
+ ### When only a mean is available (no quantiles at all)
664
+
665
+ Some tables (e.g. SCB's age-breakdown salary tables) report only a
666
+ mean per group, with no spread information. Two deliberately separate
667
+ methods are provided, each making a different explicit assumption --
668
+ compare them rather than silently picking one:
669
+
670
+ ```python
671
+ from robustkit import (
672
+ expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
673
+ compare_reconstruction_methods,
674
+ )
675
+
676
+ # Method 1: repeat the mean n times -- zero within-group spread.
677
+ # Recovers between-group regression coefficients reasonably well
678
+ # (validated in the original technique this is based on) but
679
+ # understates individual-level variation.
680
+ flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
681
+
682
+ # Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
683
+ # report quantiles, and use it to imply an approximate spread around
684
+ # the mean. Stacks two assumptions (mean-as-median, and that the
685
+ # borrowed ratio transfers to this population) -- illustrative, not a
686
+ # substitute for genuine quantile data for this specific table.
687
+ borrowed = expand_aggregated_table_borrowed_dispersion(
688
+ df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
689
+ )
690
+
691
+ # Compare both for a single group directly:
692
+ compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
693
+ ```
694
+
695
+ Both methods are documented with their specific assumptions rather
696
+ than presented as equally valid defaults -- being explicit about which
697
+ assumption was made lets the analyst judge how much a conclusion
698
+ depends on it, rather than presenting an assumption as a measurement.
699
+
700
+ ## Feature pairing (information)
701
+
702
+ Beyond ranking single features, evaluate *pairs* of features together:
703
+ how redundant are they with each other, and does knowing one reveal
704
+ additional predictive value in the other (synergy, e.g. an interaction
705
+ effect)?
706
+
707
+ ```python
708
+ from robustkit import (
709
+ conditional_mutual_information, communication_score,
710
+ rank_by_communication, pair_redundancy, pair_synergy,
711
+ rank_communicative_pairs,
712
+ )
713
+
714
+ # How communicable is a single feature -- not just predictive, but
715
+ # suitable for a clear chart/table (adequate group sizes, homogeneous
716
+ # groups, few enough categories to show at once)?
717
+ comm_ranking = rank_by_communication(df, target="value")
718
+
719
+ # How much does region's relevance to the target change once
720
+ # department is already known?
721
+ synergy = pair_synergy(df, feature_1="department", feature_2="region", target="value")
722
+
723
+ # Rank every candidate pair by combined relevance, penalizing
724
+ # redundant pairs and rewarding genuine synergy
725
+ pairs = rank_communicative_pairs(df, target="value")
726
+ ```
727
+
728
+ All mutual-information-based quantities in this module (`rank_features`,
729
+ `conditional_mutual_information`, `pair_redundancy`, `pair_synergy`,
730
+ `communication_score`) are expressed in **bits**, consistent with
731
+ `entropy()` -- internally, scikit-learn's MI estimators return nats
732
+ and are converted before being used anywhere in this package.
733
+
734
+ ## Design principles
735
+
736
+ - **One continuous x, one continuous y** at the core. This keeps every
737
+ function's output visually and numerically interpretable (a curve
738
+ you can plot, a band you can read).
739
+ - **Diagnosis and action are separate steps.** `cooks_diagnostic`
740
+ flags candidates; `cook_impact` tells you whether removing them
741
+ actually changes anything.
742
+ - **OLS is a reference point, not the enemy.** Comparing robust fits
743
+ against OLS is how you know whether robustness mattered at all.
744
+
745
+ ## Naming conventions
746
+
747
+ A few parameter/column names look similar across the package but mean
748
+ different things -- documented here explicitly so the difference reads
749
+ as intentional, not as an inconsistency to "fix":
750
+
751
+ - **`residual` vs. `difference`:** `residual` is an INDIVIDUAL-level
752
+ quantity (`actual - expected` for one row) -- used by
753
+ `mad_outlier_report`, `mad_outlier_drilldown_report`, and
754
+ `deviation_report`. `difference` is a SEGMENT/GROUP-level quantity
755
+ (typically the median residual within a group) -- used by
756
+ `segment_position_report`, `segment_benchmark_report`,
757
+ `segment_benchmark_drilldown_report`, and `benchmark_report_suite`.
758
+ - **`target` vs. `y_col`:** `robustkit.information` uses `target` for
759
+ the column being explained, since it works with arbitrary features
760
+ (not necessarily a continuous regression outcome).
761
+ `robustkit.benchmark`, `robustkit.segment_awareness`, and
762
+ `robustkit.quantiles` use `y_col`, since they specifically model a
763
+ continuous `y` as a function of `x`.
764
+ - **`segment_cols` vs. `group_columns`:** `segment_cols` (throughout
765
+ `robustkit.segment_awareness`) is an ORDERED, most-specific-first
766
+ list used to build a fallback HIERARCHY (see `hierarchical_segment`).
767
+ `group_columns` (`benchmark_report_suite`) is a FLAT list of
768
+ independent groupings, run separately with no hierarchy or fallback
769
+ between them. Different structure, different name on purpose.
770
+ - **`min_size` vs. `min_points` vs. `min_stratum_size` vs.
771
+ `min_group_size`:** all mean "minimum group size," but at different
772
+ stages: `min_size` (`hierarchical_segment` and everything built on
773
+ it) gates whether a hierarchy LEVEL gets created at all;
774
+ `min_points` (`apply_by_segment`) gates whether an already-built
775
+ segment gets ANALYZED; `min_stratum_size`
776
+ (`conditional_mutual_information`) and `min_group_size`
777
+ (`communication_score`, `rank_by_communication`) are specific to
778
+ those `robustkit.information` calculations. Kept separate rather
779
+ than unified to one name, since collapsing them would obscure which
780
+ stage of a pipeline each threshold actually applies to.
781
+
782
+ ## License
783
+
784
+ MIT -- see [LICENSE](LICENSE).