robustkit 0.0.1__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. robustkit-0.5.0/PKG-INFO +446 -0
  2. robustkit-0.5.0/README.md +406 -0
  3. {robustkit-0.0.1 → robustkit-0.5.0}/pyproject.toml +1 -1
  4. robustkit-0.5.0/robustkit/__init__.py +103 -0
  5. robustkit-0.5.0/robustkit/benchmark/global_model.py +74 -0
  6. robustkit-0.5.0/robustkit/benchmark/robustness_map.py +160 -0
  7. robustkit-0.5.0/robustkit/common/quadrants.py +57 -0
  8. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/diagnostics.py +2 -2
  9. robustkit-0.5.0/robustkit/core/goodness_of_fit.py +77 -0
  10. robustkit-0.5.0/robustkit/core/trend.py +136 -0
  11. robustkit-0.5.0/robustkit/information/__init__.py +0 -0
  12. robustkit-0.5.0/robustkit/information/communication.py +143 -0
  13. robustkit-0.5.0/robustkit/information/conditional_mi.py +50 -0
  14. robustkit-0.5.0/robustkit/information/mutual_info.py +172 -0
  15. robustkit-0.5.0/robustkit/information/pairs.py +88 -0
  16. robustkit-0.5.0/robustkit/information/quadrants.py +56 -0
  17. robustkit-0.5.0/robustkit/information/utils.py +23 -0
  18. robustkit-0.5.0/robustkit/quantiles/__init__.py +0 -0
  19. robustkit-0.5.0/robustkit/quantiles/io.py +104 -0
  20. robustkit-0.5.0/robustkit/quantiles/reconstruct.py +272 -0
  21. robustkit-0.5.0/robustkit/quantiles/trend.py +85 -0
  22. robustkit-0.5.0/robustkit/report/__init__.py +0 -0
  23. robustkit-0.5.0/robustkit/report/dispersion.py +68 -0
  24. robustkit-0.5.0/robustkit/report/visualize_analyst.py +52 -0
  25. robustkit-0.5.0/robustkit/report/visualize_publisher.py +57 -0
  26. robustkit-0.5.0/robustkit/segmentation/__init__.py +0 -0
  27. robustkit-0.5.0/robustkit.egg-info/PKG-INFO +446 -0
  28. robustkit-0.5.0/robustkit.egg-info/SOURCES.txt +53 -0
  29. robustkit-0.5.0/tests/test_benchmark.py +118 -0
  30. robustkit-0.5.0/tests/test_common_quadrants.py +30 -0
  31. robustkit-0.5.0/tests/test_information_pairs.py +122 -0
  32. robustkit-0.5.0/tests/test_quantiles_io_trend.py +126 -0
  33. robustkit-0.5.0/tests/test_quantiles_reconstruct.py +90 -0
  34. robustkit-0.5.0/tests/test_quantiles_reconstruct_mean_only.py +78 -0
  35. robustkit-0.5.0/tests/test_report.py +93 -0
  36. robustkit-0.5.0/tests/test_trend_extras.py +92 -0
  37. robustkit-0.0.1/PKG-INFO +0 -169
  38. robustkit-0.0.1/README.md +0 -129
  39. robustkit-0.0.1/robustkit/__init__.py +0 -54
  40. robustkit-0.0.1/robustkit/core/trend.py +0 -86
  41. robustkit-0.0.1/robustkit/information/mutual_info.py +0 -87
  42. robustkit-0.0.1/robustkit/information/quadrants.py +0 -67
  43. robustkit-0.0.1/robustkit.egg-info/PKG-INFO +0 -169
  44. robustkit-0.0.1/robustkit.egg-info/SOURCES.txt +0 -27
  45. {robustkit-0.0.1 → robustkit-0.5.0}/LICENSE +0 -0
  46. {robustkit-0.0.1/robustkit/core → robustkit-0.5.0/robustkit/benchmark}/__init__.py +0 -0
  47. {robustkit-0.0.1/robustkit/information → robustkit-0.5.0/robustkit/common}/__init__.py +0 -0
  48. {robustkit-0.0.1/robustkit/segmentation → robustkit-0.5.0/robustkit/core}/__init__.py +0 -0
  49. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/consistency.py +0 -0
  50. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/stability.py +0 -0
  51. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/core/uncertainty.py +0 -0
  52. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/information/entropy.py +0 -0
  53. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/information/profile.py +0 -0
  54. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/information/visualization.py +0 -0
  55. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/segmentation/apply.py +0 -0
  56. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit/segmentation/hierarchy.py +0 -0
  57. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit.egg-info/dependency_links.txt +0 -0
  58. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit.egg-info/requires.txt +0 -0
  59. {robustkit-0.0.1 → robustkit-0.5.0}/robustkit.egg-info/top_level.txt +0 -0
  60. {robustkit-0.0.1 → robustkit-0.5.0}/setup.cfg +0 -0
  61. {robustkit-0.0.1 → robustkit-0.5.0}/tests/test_core.py +0 -0
  62. {robustkit-0.0.1 → robustkit-0.5.0}/tests/test_information.py +0 -0
  63. {robustkit-0.0.1 → robustkit-0.5.0}/tests/test_segmentation.py +0 -0
@@ -0,0 +1,446 @@
1
+ Metadata-Version: 2.4
2
+ Name: robustkit
3
+ Version: 0.5.0
4
+ Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
5
+ Author: Mikael Lundqvist
6
+ License: MIT License
7
+
8
+ Copyright (c) 2026 Mikael Lundqvist
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Requires-Python: >=3.10
29
+ Description-Content-Type: text/markdown
30
+ License-File: LICENSE
31
+ Requires-Dist: numpy>=1.24
32
+ Requires-Dist: pandas>=2.0
33
+ Requires-Dist: scikit-learn>=1.3
34
+ Requires-Dist: statsmodels>=0.14
35
+ Requires-Dist: scipy>=1.10
36
+ Requires-Dist: matplotlib>=3.7
37
+ Provides-Extra: dev
38
+ Requires-Dist: pytest>=7.0; extra == "dev"
39
+ Dynamic: license-file
40
+
41
+ # robustkit
42
+
43
+ > ⚠️ **Under active development.** This is an early placeholder release
44
+ > to claim the package name on PyPI. The API is incomplete and may
45
+ > change without notice. Not yet recommended for production use.
46
+
47
+ Practical tools for robust analysis of a single continuous relationship:
48
+ y as a function of one continuous x.
49
+
50
+ The guiding idea: **a conclusion that survives multiple fitting methods
51
+ is more trustworthy than one that only holds under a single model.**
52
+ `robustkit` makes it easy to compare Huber, Tukey biweight, and OLS
53
+ fits side by side, identify and quantify the influence of individual
54
+ observations, and get honest, bias-corrected uncertainty estimates.
55
+
56
+ ## Status
57
+
58
+ `robustkit.core` (trend fitting, stability, diagnostics, uncertainty,
59
+ consistency checks), `robustkit.segmentation` (hierarchical grouping,
60
+ per-segment analysis), `robustkit.information` (mutual-information
61
+ feature ranking, quadrant classification, pairwise redundancy/synergy
62
+ scoring), `robustkit.benchmark` (global-trend segment comparison,
63
+ Robustness Map), `robustkit.report` (analyst vs. publisher views,
64
+ dispersion measures), and `robustkit.quantiles` (generic JSON-stat
65
+ loading, published-quantile-trend visualization, and lognormal-
66
+ calibrated reconstruction of individual-level data from aggregated
67
+ summaries) are stable and tested.
68
+
69
+ **Note on `information_efficiency`:** values can exceed 1.0 for
70
+ continuous features. `mutual_information` is estimated on the
71
+ full-resolution continuous values, while `entropy_bits` is computed on
72
+ a binned version of the same feature (since `entropy()` expects
73
+ categorical input). Binning discards information, so `entropy_bits` is
74
+ a lower bound on the feature's true entropy -- an efficiency above 1.0
75
+ signals that the feature carries more usable information than a coarse
76
+ categorical summary of it would capture. This is expected behavior,
77
+ not a bug.
78
+
79
+ ## Installation
80
+
81
+ ```bash
82
+ git clone https://github.com/<your-username>/robustkit.git
83
+ cd robustkit
84
+ pip install -e ".[dev]"
85
+ ```
86
+
87
+ ## Quickstart
88
+
89
+ ```python
90
+ import numpy as np
91
+ from robustkit import (
92
+ fit_huber_trend, fit_tukey_trend, predict_trend,
93
+ model_stability_pct, cooks_diagnostic, cook_impact,
94
+ bootstrap_band, bca_bootstrap_ci,
95
+ )
96
+
97
+ # x: a single continuous predictor, y: a single continuous outcome
98
+ x = np.random.default_rng(0).uniform(20, 60, 200)
99
+ y = 1000 + 50 * x - 0.4 * x**2 + np.random.default_rng(1).normal(0, 500, 200)
100
+
101
+ fit = fit_huber_trend(x, y, degree=2)
102
+ y_pred = predict_trend(fit, x_new=[30, 40, 50])
103
+
104
+ stability = model_stability_pct(x, y)
105
+ print("Median % spread between Huber/Tukey/OLS:", stability["median_pct_diff"])
106
+
107
+ diag = cooks_diagnostic(x, y)
108
+ impact = cook_impact(x, y, diag["flagged_indices"])
109
+ print("Median % change in curve if flagged points removed:", impact["median_pct_change"])
110
+
111
+ band = bootstrap_band(x, y)
112
+ ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
113
+ ```
114
+
115
+ See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
116
+
117
+ ## Trend growth rate and goodness of fit
118
+
119
+ ```python
120
+ from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
121
+
122
+ fit = fit_huber_trend(df["age"], df["salary"])
123
+
124
+ # Rate of change of the trend itself (e.g. "salary growth per year of
125
+ # age"), not just its level
126
+ rates = trend_derivative(fit, x=[30, 40, 50])
127
+
128
+ # How well does this fit actually explain the variation in y?
129
+ goodness_of_fit(df["age"], df["salary"], degree=2)
130
+
131
+ # Don't assume a quadratic trend is always the right choice -- check
132
+ # empirically whether a higher degree captures meaningfully more
133
+ compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
134
+ ```
135
+
136
+ **Note:** x is standardized internally before building polynomial
137
+ features (both here and throughout `robustkit.core`), since raw
138
+ polynomial features become numerically unstable at higher degrees for
139
+ realistic x scales (e.g. age^5 vastly outscales age^1). This is
140
+ transparent to callers -- `predict_trend` and `trend_derivative` still
141
+ take and return values in the original x scale.
142
+
143
+ ## Segmentation
144
+
145
+ Run any `robustkit.core` analysis independently across subgroups of a
146
+ larger dataset, with automatic fallback to coarser groupings when a
147
+ finer one is too small to analyze reliably:
148
+
149
+ ```python
150
+ from robustkit import hierarchical_segment, apply_by_segment, model_stability_pct
151
+
152
+ hierarchy = [["department", "level", "status"], ["level", "status"], ["status"]]
153
+ segmented = hierarchical_segment(df, hierarchy, min_size=20)
154
+
155
+ report = apply_by_segment(
156
+ segmented, segment_col="segment_id", x_col="age", y_col="value",
157
+ analysis_fn=model_stability_pct,
158
+ )
159
+ ```
160
+
161
+ `apply_by_segment` works with any function shaped like
162
+ `analysis_fn(x, y, **kwargs) -> dict` -- built-in ones
163
+ (`model_stability_pct`, `cook_impact`, `bca_bootstrap_ci`, ...) or your
164
+ own. Only scalar values in the returned dict end up in the report
165
+ table; segments below `min_points` are skipped rather than causing an
166
+ error.
167
+
168
+ ## Feature ranking (information)
169
+
170
+ Rank features by mutual information with a target, normalized by each
171
+ feature's own entropy, and classify them into four quadrants:
172
+
173
+ ```python
174
+ from robustkit import rank_features, quadrant_report, plot_feature_space
175
+
176
+ ranking = rank_features(df, target="value")
177
+ report = quadrant_report(df, target="value") # adds a `quadrant` column
178
+ plot_feature_space(df, target="value") # same quadrants, visualized
179
+ ```
180
+
181
+ `quadrant_report` and `plot_feature_space` always agree on quadrant
182
+ assignment -- both route through the same thresholding logic.
183
+
184
+ **Caveat:** default thresholds are the *median* mutual information /
185
+ efficiency across the ranked features. With only a handful of
186
+ features, this can put a genuinely weak feature in the same "high"
187
+ half as a strong one, since roughly half of any list sits above its
188
+ own median regardless of how large the actual gap is. Median
189
+ thresholding becomes meaningful with a reasonably large feature set;
190
+ for a handful of candidates, read the raw `mutual_information` /
191
+ `information_efficiency` values directly rather than relying on the
192
+ quadrant label alone.
193
+
194
+ See `examples/information_tutorial.py` for a complete walkthrough.
195
+
196
+ ## Benchmarking against a global trend
197
+
198
+ Compare each segment's observed outcome against what a single global
199
+ robust trend predicts, with bootstrap uncertainty on the difference --
200
+ answers "which groups deviate from the overall trend, and by how
201
+ much?" rather than "how does the trend look overall?":
202
+
203
+ ```python
204
+ from robustkit import segment_position_report
205
+
206
+ report = segment_position_report(
207
+ df, segment_col="department", x_col="age", y_col="salary",
208
+ )
209
+ # segment n observed_median expected_median difference ci_lower ci_upper
210
+ # Finance 176 48339.70 47799.82 539.88 202.15 1031.01
211
+ # HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36
212
+ # IT 250 47226.50 47126.91 99.59 -117.56 510.81
213
+ ```
214
+
215
+ A segment's confidence interval crossing zero means no clear deviation
216
+ from the benchmark; HR and Finance above don't cross zero, IT does.
217
+
218
+ ## Robustness Map
219
+
220
+ Classify features by how much a conclusion about their relationship
221
+ with the target depends on (a) fitting method choice and (b) specific
222
+ influential observations -- two genuinely different failure modes that
223
+ a single diagnostic can miss:
224
+
225
+ ```python
226
+ from robustkit import feature_robustness_report, plot_feature_robustness
227
+
228
+ report = feature_robustness_report(df, target="value")
229
+ # feature stability_pct cook_impact_pct quadrant
230
+ # CRIM 8.9 17.1 fragile
231
+ # AGE 16.8 15.5 fragile
232
+ # RM 4.9 0.1 robust
233
+ # TAX 22.2 1.4 structural_sensitivity
234
+
235
+ plot_feature_robustness(report=report)
236
+ ```
237
+
238
+ Four quadrants: **robust** (low spread, low impact), **structural
239
+ sensitivity** (sensitive to fitting method, not to specific points),
240
+ **data sensitive** (a few points drive the conclusion, method choice
241
+ barely matters), **fragile** (both -- least trustworthy).
242
+
243
+ `quadrant_report`/`plot_feature_space` (information) and
244
+ `feature_robustness_report`/`plot_feature_robustness` (benchmark) both
245
+ route through the same shared classifier, `robustkit.classify_quadrants`
246
+ -- any future quadrant-based analysis in this package will too.
247
+
248
+ ## Analyst view vs. publisher view
249
+
250
+ Two visualizations that look superficially similar but answer
251
+ genuinely different questions:
252
+
253
+ ```python
254
+ from robustkit import plot_analyst_view, plot_publisher_view, dispersion_ratio, iqr
255
+
256
+ # "How confident are we in the trend estimate?" -- a bootstrap
257
+ # confidence band that SHRINKS as sample size grows.
258
+ plot_analyst_view(df["age"], df["salary"])
259
+
260
+ # "How spread out are actual values in the population?" -- a median +
261
+ # IQR band that does NOT shrink with more data, since it reflects
262
+ # real dispersion, not estimation uncertainty. show_points defaults to
263
+ # False, since this view is meant for publishing potentially sensitive
264
+ # data (e.g. individual salaries) without exposing raw points.
265
+ plot_publisher_view(df["age"], df["salary"])
266
+ ```
267
+
268
+ This distinction matters in practice: with 20x more data (same
269
+ underlying distribution), the analyst view's confidence band roughly
270
+ halves in width, while the publisher view's IQR band stays essentially
271
+ unchanged -- confirmed by the package's own test suite.
272
+
273
+ `dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
274
+ standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
275
+ computes both across bins of a continuous x, e.g. to check whether
276
+ dispersion (inequality) grows with age.
277
+
278
+ ## Loading published quantile tables (SCB / JSON-stat)
279
+
280
+ Some statistics agencies (e.g. Statistics Sweden, SCB) publish
281
+ quantiles (Q1/median/Q3) directly, with no individual-level data
282
+ available at all. `robustkit.quantiles` loads these tables generically
283
+ via JSON-stat, a standardized dimensional-data format used by SCB and
284
+ other national statistics agencies -- avoiding the fragility of
285
+ parsing metadata out of column-name strings in a wide CSV export.
286
+
287
+ ```python
288
+ from robustkit import load_scb_json_stat, plot_quantile_trend, quantile_trend_dispersion
289
+
290
+ df = load_scb_json_stat("some_scb_table.json")
291
+
292
+ # A real SCB quirk this loader does NOT try to guess automatically:
293
+ # category labels can change meaning over time (e.g. Sweden's oldest
294
+ # working-age bracket was labeled "65-66 år" through 2022 and
295
+ # "65-68 år" from 2023, following a pension-age reform). Merge such
296
+ # cases explicitly:
297
+ df = load_scb_json_stat(
298
+ "some_scb_table.json",
299
+ rename_categories={"ålder": {"65–68 år": "65–66 år"}},
300
+ )
301
+
302
+ # Once reshaped to a wide table with q1/median/q3 columns:
303
+ plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
304
+ quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
305
+ ```
306
+
307
+ This is the "quantiles are already given" case. A complementary case
308
+ -- reconstructing approximate individual-level data from aggregated
309
+ group means, for when only summary statistics (not quantiles) are
310
+ available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
311
+
312
+ See `examples/quantiles_tutorial.py` for a complete walkthrough.
313
+
314
+ ## Reconstructing individual-level data from aggregated summaries
315
+
316
+ For the complementary case -- only aggregated group summaries (n,
317
+ Q1, median, Q3) are available, not the quantile trend itself as the
318
+ final answer, and you want to run `robustkit.core` analyses as if
319
+ individual data existed:
320
+
321
+ ```python
322
+ from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
323
+
324
+ # One row per group (e.g. year), with n/q1/median/q3 columns
325
+ synthetic = expand_aggregated_table(
326
+ summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
327
+ group_cols=["year"], value_name="salary",
328
+ )
329
+
330
+ # Now usable exactly like real individual-level data:
331
+ fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
332
+ ```
333
+
334
+ Method: a lognormal distribution is calibrated (via the IQR) to match
335
+ each group's reported Q1/median/Q3, then `n` synthetic values are
336
+ drawn from it. Validated end-to-end against real published SCB salary
337
+ data: a Huber trend fitted on reconstructed pseudo-individual data
338
+ tracked the true published median trend within 2% across 12 years.
339
+
340
+ **Note on what this recovers:** because a Huber (or Tukey) fit on
341
+ right-skewed reconstructed data tracks something close to the
342
+ *median* trend it was calibrated against -- not the arithmetic mean --
343
+ this is consistent with, not a limitation of, the reconstruction
344
+ method. To target the mean instead, fit on `log(value)` and
345
+ exponentiate predictions back, which approximates the geometric mean.
346
+
347
+ Always check `check_reconstruction_quality()` before trusting a
348
+ reconstruction: real Q1/median/Q3 triples aren't always perfectly
349
+ consistent with a pure lognormal shape.
350
+
351
+ **Warning -- unbounded tail at large n:** a lognormal has no natural
352
+ upper limit, and its expected maximum grows with n. Reconstructing at
353
+ the TRUE group size from a national table (SCB salary tables can
354
+ report n in the hundreds of thousands to millions) can produce
355
+ implausibly extreme tail values -- real salaries have practical
356
+ ceilings a pure lognormal doesn't know about. This package's own
357
+ examples and tests deliberately scale n down to a few thousand for
358
+ demonstration; calibration quality (matching Q1/median/Q3) doesn't
359
+ depend on reproducing the true population size, but tail plausibility
360
+ does. No clipping is applied automatically.
361
+
362
+ ### When only a mean is available (no quantiles at all)
363
+
364
+ Some tables (e.g. SCB's age-breakdown salary tables) report only a
365
+ mean per group, with no spread information. Two deliberately separate
366
+ methods are provided, each making a different explicit assumption --
367
+ compare them rather than silently picking one:
368
+
369
+ ```python
370
+ from robustkit import (
371
+ expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
372
+ compare_reconstruction_methods,
373
+ )
374
+
375
+ # Method 1: repeat the mean n times -- zero within-group spread.
376
+ # Recovers between-group regression coefficients reasonably well
377
+ # (validated in the original technique this is based on) but
378
+ # understates individual-level variation.
379
+ flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
380
+
381
+ # Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
382
+ # report quantiles, and use it to imply an approximate spread around
383
+ # the mean. Stacks two assumptions (mean-as-median, and that the
384
+ # borrowed ratio transfers to this population) -- illustrative, not a
385
+ # substitute for genuine quantile data for this specific table.
386
+ borrowed = expand_aggregated_table_borrowed_dispersion(
387
+ df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
388
+ )
389
+
390
+ # Compare both for a single group directly:
391
+ compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
392
+ ```
393
+
394
+ Both methods are documented with their specific assumptions rather
395
+ than presented as equally valid defaults -- being explicit about which
396
+ assumption was made lets the analyst judge how much a conclusion
397
+ depends on it, rather than presenting an assumption as a measurement.
398
+
399
+ ## Feature pairing (information)
400
+
401
+ Beyond ranking single features, evaluate *pairs* of features together:
402
+ how redundant are they with each other, and does knowing one reveal
403
+ additional predictive value in the other (synergy, e.g. an interaction
404
+ effect)?
405
+
406
+ ```python
407
+ from robustkit import (
408
+ conditional_mutual_information, communication_score,
409
+ rank_by_communication, pair_redundancy, pair_synergy,
410
+ rank_communicative_pairs,
411
+ )
412
+
413
+ # How communicable is a single feature -- not just predictive, but
414
+ # suitable for a clear chart/table (adequate group sizes, homogeneous
415
+ # groups, few enough categories to show at once)?
416
+ comm_ranking = rank_by_communication(df, target="value")
417
+
418
+ # How much does region's relevance to the target change once
419
+ # department is already known?
420
+ synergy = pair_synergy(df, feature_1="department", feature_2="region", target="value")
421
+
422
+ # Rank every candidate pair by combined relevance, penalizing
423
+ # redundant pairs and rewarding genuine synergy
424
+ pairs = rank_communicative_pairs(df, target="value")
425
+ ```
426
+
427
+ All mutual-information-based quantities in this module (`rank_features`,
428
+ `conditional_mutual_information`, `pair_redundancy`, `pair_synergy`,
429
+ `communication_score`) are expressed in **bits**, consistent with
430
+ `entropy()` -- internally, scikit-learn's MI estimators return nats
431
+ and are converted before being used anywhere in this package.
432
+
433
+ ## Design principles
434
+
435
+ - **One continuous x, one continuous y** at the core. This keeps every
436
+ function's output visually and numerically interpretable (a curve
437
+ you can plot, a band you can read).
438
+ - **Diagnosis and action are separate steps.** `cooks_diagnostic`
439
+ flags candidates; `cook_impact` tells you whether removing them
440
+ actually changes anything.
441
+ - **OLS is a reference point, not the enemy.** Comparing robust fits
442
+ against OLS is how you know whether robustness mattered at all.
443
+
444
+ ## License
445
+
446
+ MIT -- see [LICENSE](LICENSE).