robustkit 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {robustkit-0.4.0/robustkit.egg-info → robustkit-0.5.0}/PKG-INFO +153 -3
  2. {robustkit-0.4.0 → robustkit-0.5.0}/README.md +152 -2
  3. {robustkit-0.4.0 → robustkit-0.5.0}/pyproject.toml +1 -1
  4. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/__init__.py +25 -1
  5. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/benchmark/robustness_map.py +18 -0
  6. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/diagnostics.py +2 -2
  7. robustkit-0.5.0/robustkit/core/goodness_of_fit.py +77 -0
  8. robustkit-0.5.0/robustkit/core/trend.py +136 -0
  9. robustkit-0.5.0/robustkit/quantiles/io.py +104 -0
  10. robustkit-0.5.0/robustkit/quantiles/reconstruct.py +272 -0
  11. robustkit-0.5.0/robustkit/quantiles/trend.py +85 -0
  12. robustkit-0.5.0/robustkit/segmentation/__init__.py +0 -0
  13. {robustkit-0.4.0 → robustkit-0.5.0/robustkit.egg-info}/PKG-INFO +153 -3
  14. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/SOURCES.txt +10 -1
  15. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_benchmark.py +23 -4
  16. robustkit-0.5.0/tests/test_quantiles_io_trend.py +126 -0
  17. robustkit-0.5.0/tests/test_quantiles_reconstruct.py +90 -0
  18. robustkit-0.5.0/tests/test_quantiles_reconstruct_mean_only.py +78 -0
  19. robustkit-0.5.0/tests/test_trend_extras.py +92 -0
  20. robustkit-0.4.0/robustkit/core/trend.py +0 -86
  21. {robustkit-0.4.0 → robustkit-0.5.0}/LICENSE +0 -0
  22. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/benchmark/__init__.py +0 -0
  23. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/benchmark/global_model.py +0 -0
  24. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/common/__init__.py +0 -0
  25. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/common/quadrants.py +0 -0
  26. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/__init__.py +0 -0
  27. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/consistency.py +0 -0
  28. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/stability.py +0 -0
  29. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/uncertainty.py +0 -0
  30. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/__init__.py +0 -0
  31. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/communication.py +0 -0
  32. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/conditional_mi.py +0 -0
  33. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/entropy.py +0 -0
  34. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/mutual_info.py +0 -0
  35. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/pairs.py +0 -0
  36. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/profile.py +0 -0
  37. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/quadrants.py +0 -0
  38. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/utils.py +0 -0
  39. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/visualization.py +0 -0
  40. {robustkit-0.4.0/robustkit/report → robustkit-0.5.0/robustkit/quantiles}/__init__.py +0 -0
  41. {robustkit-0.4.0/robustkit/segmentation → robustkit-0.5.0/robustkit/report}/__init__.py +0 -0
  42. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/report/dispersion.py +0 -0
  43. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/report/visualize_analyst.py +0 -0
  44. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/report/visualize_publisher.py +0 -0
  45. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/segmentation/apply.py +0 -0
  46. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/segmentation/hierarchy.py +0 -0
  47. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/dependency_links.txt +0 -0
  48. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/requires.txt +0 -0
  49. {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/top_level.txt +0 -0
  50. {robustkit-0.4.0 → robustkit-0.5.0}/setup.cfg +0 -0
  51. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_common_quadrants.py +0 -0
  52. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_core.py +0 -0
  53. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_information.py +0 -0
  54. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_information_pairs.py +0 -0
  55. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_report.py +0 -0
  56. {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_segmentation.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: robustkit
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
5
5
  Author: Mikael Lundqvist
6
6
  License: MIT License
@@ -60,8 +60,11 @@ consistency checks), `robustkit.segmentation` (hierarchical grouping,
60
60
  per-segment analysis), `robustkit.information` (mutual-information
61
61
  feature ranking, quadrant classification, pairwise redundancy/synergy
62
62
  scoring), `robustkit.benchmark` (global-trend segment comparison,
63
- Robustness Map), and `robustkit.report` (analyst vs. publisher views,
64
- dispersion measures) are stable and tested.
63
+ Robustness Map), `robustkit.report` (analyst vs. publisher views,
64
+ dispersion measures), and `robustkit.quantiles` (generic JSON-stat
65
+ loading, published-quantile-trend visualization, and lognormal-
66
+ calibrated reconstruction of individual-level data from aggregated
67
+ summaries) are stable and tested.
65
68
 
66
69
  **Note on `information_efficiency`:** values can exceed 1.0 for
67
70
  continuous features. `mutual_information` is estimated on the
@@ -111,6 +114,32 @@ ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
111
114
 
112
115
  See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
113
116
 
117
+ ## Trend growth rate and goodness of fit
118
+
119
+ ```python
120
+ from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
121
+
122
+ fit = fit_huber_trend(df["age"], df["salary"])
123
+
124
+ # Rate of change of the trend itself (e.g. "salary growth per year of
125
+ # age"), not just its level
126
+ rates = trend_derivative(fit, x=[30, 40, 50])
127
+
128
+ # How well does this fit actually explain the variation in y?
129
+ goodness_of_fit(df["age"], df["salary"], degree=2)
130
+
131
+ # Don't assume a quadratic trend is always the right choice -- check
132
+ # empirically whether a higher degree captures meaningfully more
133
+ compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
134
+ ```
135
+
136
+ **Note:** x is standardized internally before building polynomial
137
+ features (both here and throughout `robustkit.core`), since raw
138
+ polynomial features become numerically unstable at higher degrees for
139
+ realistic x scales (e.g. age^5 vastly outscales age^1). This is
140
+ transparent to callers -- `predict_trend` and `trend_derivative` still
141
+ take and return values in the original x scale.
142
+
114
143
  ## Segmentation
115
144
 
116
145
  Run any `robustkit.core` analysis independently across subgroups of a
@@ -246,6 +275,127 @@ standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
246
275
  computes both across bins of a continuous x, e.g. to check whether
247
276
  dispersion (inequality) grows with age.
248
277
 
278
+ ## Loading published quantile tables (SCB / JSON-stat)
279
+
280
+ Some statistics agencies (e.g. Statistics Sweden, SCB) publish
281
+ quantiles (Q1/median/Q3) directly, with no individual-level data
282
+ available at all. `robustkit.quantiles` loads these tables generically
283
+ via JSON-stat, a standardized dimensional-data format used by SCB and
284
+ other national statistics agencies -- avoiding the fragility of
285
+ parsing metadata out of column-name strings in a wide CSV export.
286
+
287
+ ```python
288
+ from robustkit import load_scb_json_stat, plot_quantile_trend, quantile_trend_dispersion
289
+
290
+ df = load_scb_json_stat("some_scb_table.json")
291
+
292
+ # A real SCB quirk this loader does NOT try to guess automatically:
293
+ # category labels can change meaning over time (e.g. Sweden's oldest
294
+ # working-age bracket was labeled "65-66 år" through 2022 and
295
+ # "65-68 år" from 2023, following a pension-age reform). Merge such
296
+ # cases explicitly:
297
+ df = load_scb_json_stat(
298
+ "some_scb_table.json",
299
+ rename_categories={"ålder": {"65–68 år": "65–66 år"}},
300
+ )
301
+
302
+ # Once reshaped to a wide table with q1/median/q3 columns:
303
+ plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
304
+ quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
305
+ ```
306
+
307
+ This is the "quantiles are already given" case. A complementary case
308
+ -- reconstructing approximate individual-level data from aggregated
309
+ group means, for when only summary statistics (not quantiles) are
310
+ available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
311
+
312
+ See `examples/quantiles_tutorial.py` for a complete walkthrough.
313
+
314
+ ## Reconstructing individual-level data from aggregated summaries
315
+
316
+ For the complementary case -- only aggregated group summaries (n,
317
+ Q1, median, Q3) are available, not the quantile trend itself as the
318
+ final answer, and you want to run `robustkit.core` analyses as if
319
+ individual data existed:
320
+
321
+ ```python
322
+ from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
323
+
324
+ # One row per group (e.g. year), with n/q1/median/q3 columns
325
+ synthetic = expand_aggregated_table(
326
+ summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
327
+ group_cols=["year"], value_name="salary",
328
+ )
329
+
330
+ # Now usable exactly like real individual-level data:
331
+ fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
332
+ ```
333
+
334
+ Method: a lognormal distribution is calibrated (via the IQR) to match
335
+ each group's reported Q1/median/Q3, then `n` synthetic values are
336
+ drawn from it. Validated end-to-end against real published SCB salary
337
+ data: a Huber trend fitted on reconstructed pseudo-individual data
338
+ tracked the true published median trend within 2% across 12 years.
339
+
340
+ **Note on what this recovers:** because a Huber (or Tukey) fit on
341
+ right-skewed reconstructed data tracks something close to the
342
+ *median* trend it was calibrated against -- not the arithmetic mean --
343
+ this is consistent with, not a limitation of, the reconstruction
344
+ method. To target the mean instead, fit on `log(value)` and
345
+ exponentiate predictions back, which approximates the geometric mean.
346
+
347
+ Always check `check_reconstruction_quality()` before trusting a
348
+ reconstruction: real Q1/median/Q3 triples aren't always perfectly
349
+ consistent with a pure lognormal shape.
350
+
351
+ **Warning -- unbounded tail at large n:** a lognormal has no natural
352
+ upper limit, and its expected maximum grows with n. Reconstructing at
353
+ the TRUE group size from a national table (SCB salary tables can
354
+ report n in the hundreds of thousands to millions) can produce
355
+ implausibly extreme tail values -- real salaries have practical
356
+ ceilings a pure lognormal doesn't know about. This package's own
357
+ examples and tests deliberately scale n down to a few thousand for
358
+ demonstration; calibration quality (matching Q1/median/Q3) doesn't
359
+ depend on reproducing the true population size, but tail plausibility
360
+ does. No clipping is applied automatically.
361
+
362
+ ### When only a mean is available (no quantiles at all)
363
+
364
+ Some tables (e.g. SCB's age-breakdown salary tables) report only a
365
+ mean per group, with no spread information. Two deliberately separate
366
+ methods are provided, each making a different explicit assumption --
367
+ compare them rather than silently picking one:
368
+
369
+ ```python
370
+ from robustkit import (
371
+ expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
372
+ compare_reconstruction_methods,
373
+ )
374
+
375
+ # Method 1: repeat the mean n times -- zero within-group spread.
376
+ # Recovers between-group regression coefficients reasonably well
377
+ # (validated in the original technique this is based on) but
378
+ # understates individual-level variation.
379
+ flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
380
+
381
+ # Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
382
+ # report quantiles, and use it to imply an approximate spread around
383
+ # the mean. Stacks two assumptions (mean-as-median, and that the
384
+ # borrowed ratio transfers to this population) -- illustrative, not a
385
+ # substitute for genuine quantile data for this specific table.
386
+ borrowed = expand_aggregated_table_borrowed_dispersion(
387
+ df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
388
+ )
389
+
390
+ # Compare both for a single group directly:
391
+ compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
392
+ ```
393
+
394
+ Both methods are documented with their specific assumptions rather
395
+ than presented as equally valid defaults -- being explicit about which
396
+ assumption was made lets the analyst judge how much a conclusion
397
+ depends on it, rather than presenting an assumption as a measurement.
398
+
249
399
  ## Feature pairing (information)
250
400
 
251
401
  Beyond ranking single features, evaluate *pairs* of features together:
@@ -20,8 +20,11 @@ consistency checks), `robustkit.segmentation` (hierarchical grouping,
20
20
  per-segment analysis), `robustkit.information` (mutual-information
21
21
  feature ranking, quadrant classification, pairwise redundancy/synergy
22
22
  scoring), `robustkit.benchmark` (global-trend segment comparison,
23
- Robustness Map), and `robustkit.report` (analyst vs. publisher views,
24
- dispersion measures) are stable and tested.
23
+ Robustness Map), `robustkit.report` (analyst vs. publisher views,
24
+ dispersion measures), and `robustkit.quantiles` (generic JSON-stat
25
+ loading, published-quantile-trend visualization, and lognormal-
26
+ calibrated reconstruction of individual-level data from aggregated
27
+ summaries) are stable and tested.
25
28
 
26
29
  **Note on `information_efficiency`:** values can exceed 1.0 for
27
30
  continuous features. `mutual_information` is estimated on the
@@ -71,6 +74,32 @@ ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
71
74
 
72
75
  See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
73
76
 
77
+ ## Trend growth rate and goodness of fit
78
+
79
+ ```python
80
+ from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
81
+
82
+ fit = fit_huber_trend(df["age"], df["salary"])
83
+
84
+ # Rate of change of the trend itself (e.g. "salary growth per year of
85
+ # age"), not just its level
86
+ rates = trend_derivative(fit, x=[30, 40, 50])
87
+
88
+ # How well does this fit actually explain the variation in y?
89
+ goodness_of_fit(df["age"], df["salary"], degree=2)
90
+
91
+ # Don't assume a quadratic trend is always the right choice -- check
92
+ # empirically whether a higher degree captures meaningfully more
93
+ compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
94
+ ```
95
+
96
+ **Note:** x is standardized internally before building polynomial
97
+ features (both here and throughout `robustkit.core`), since raw
98
+ polynomial features become numerically unstable at higher degrees for
99
+ realistic x scales (e.g. age^5 vastly outscales age^1). This is
100
+ transparent to callers -- `predict_trend` and `trend_derivative` still
101
+ take and return values in the original x scale.
102
+
74
103
  ## Segmentation
75
104
 
76
105
  Run any `robustkit.core` analysis independently across subgroups of a
@@ -206,6 +235,127 @@ standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
206
235
  computes both across bins of a continuous x, e.g. to check whether
207
236
  dispersion (inequality) grows with age.
208
237
 
238
+ ## Loading published quantile tables (SCB / JSON-stat)
239
+
240
+ Some statistics agencies (e.g. Statistics Sweden, SCB) publish
241
+ quantiles (Q1/median/Q3) directly, with no individual-level data
242
+ available at all. `robustkit.quantiles` loads these tables generically
243
+ via JSON-stat, a standardized dimensional-data format used by SCB and
244
+ other national statistics agencies -- avoiding the fragility of
245
+ parsing metadata out of column-name strings in a wide CSV export.
246
+
247
+ ```python
248
+ from robustkit import load_scb_json_stat, plot_quantile_trend, quantile_trend_dispersion
249
+
250
+ df = load_scb_json_stat("some_scb_table.json")
251
+
252
+ # A real SCB quirk this loader does NOT try to guess automatically:
253
+ # category labels can change meaning over time (e.g. Sweden's oldest
254
+ # working-age bracket was labeled "65-66 år" through 2022 and
255
+ # "65-68 år" from 2023, following a pension-age reform). Merge such
256
+ # cases explicitly:
257
+ df = load_scb_json_stat(
258
+ "some_scb_table.json",
259
+ rename_categories={"ålder": {"65–68 år": "65–66 år"}},
260
+ )
261
+
262
+ # Once reshaped to a wide table with q1/median/q3 columns:
263
+ plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
264
+ quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
265
+ ```
266
+
267
+ This is the "quantiles are already given" case. A complementary case
268
+ -- reconstructing approximate individual-level data from aggregated
269
+ group means, for when only summary statistics (not quantiles) are
270
+ available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
271
+
272
+ See `examples/quantiles_tutorial.py` for a complete walkthrough.
273
+
274
+ ## Reconstructing individual-level data from aggregated summaries
275
+
276
+ For the complementary case -- only aggregated group summaries (n,
277
+ Q1, median, Q3) are available, not the quantile trend itself as the
278
+ final answer, and you want to run `robustkit.core` analyses as if
279
+ individual data existed:
280
+
281
+ ```python
282
+ from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
283
+
284
+ # One row per group (e.g. year), with n/q1/median/q3 columns
285
+ synthetic = expand_aggregated_table(
286
+ summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
287
+ group_cols=["year"], value_name="salary",
288
+ )
289
+
290
+ # Now usable exactly like real individual-level data:
291
+ fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
292
+ ```
293
+
294
+ Method: a lognormal distribution is calibrated (via the IQR) to match
295
+ each group's reported Q1/median/Q3, then `n` synthetic values are
296
+ drawn from it. Validated end-to-end against real published SCB salary
297
+ data: a Huber trend fitted on reconstructed pseudo-individual data
298
+ tracked the true published median trend within 2% across 12 years.
299
+
300
+ **Note on what this recovers:** because a Huber (or Tukey) fit on
301
+ right-skewed reconstructed data tracks something close to the
302
+ *median* trend it was calibrated against -- not the arithmetic mean --
303
+ this is consistent with, not a limitation of, the reconstruction
304
+ method. To target the mean instead, fit on `log(value)` and
305
+ exponentiate predictions back, which approximates the geometric mean.
306
+
307
+ Always check `check_reconstruction_quality()` before trusting a
308
+ reconstruction: real Q1/median/Q3 triples aren't always perfectly
309
+ consistent with a pure lognormal shape.
310
+
311
+ **Warning -- unbounded tail at large n:** a lognormal has no natural
312
+ upper limit, and its expected maximum grows with n. Reconstructing at
313
+ the TRUE group size from a national table (SCB salary tables can
314
+ report n in the hundreds of thousands to millions) can produce
315
+ implausibly extreme tail values -- real salaries have practical
316
+ ceilings a pure lognormal doesn't know about. This package's own
317
+ examples and tests deliberately scale n down to a few thousand for
318
+ demonstration; calibration quality (matching Q1/median/Q3) doesn't
319
+ depend on reproducing the true population size, but tail plausibility
320
+ does. No clipping is applied automatically.
321
+
322
+ ### When only a mean is available (no quantiles at all)
323
+
324
+ Some tables (e.g. SCB's age-breakdown salary tables) report only a
325
+ mean per group, with no spread information. Two deliberately separate
326
+ methods are provided, each making a different explicit assumption --
327
+ compare them rather than silently picking one:
328
+
329
+ ```python
330
+ from robustkit import (
331
+ expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
332
+ compare_reconstruction_methods,
333
+ )
334
+
335
+ # Method 1: repeat the mean n times -- zero within-group spread.
336
+ # Recovers between-group regression coefficients reasonably well
337
+ # (validated in the original technique this is based on) but
338
+ # understates individual-level variation.
339
+ flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
340
+
341
+ # Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
342
+ # report quantiles, and use it to imply an approximate spread around
343
+ # the mean. Stacks two assumptions (mean-as-median, and that the
344
+ # borrowed ratio transfers to this population) -- illustrative, not a
345
+ # substitute for genuine quantile data for this specific table.
346
+ borrowed = expand_aggregated_table_borrowed_dispersion(
347
+ df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
348
+ )
349
+
350
+ # Compare both for a single group directly:
351
+ compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
352
+ ```
353
+
354
+ Both methods are documented with their specific assumptions rather
355
+ than presented as equally valid defaults -- being explicit about which
356
+ assumption was made lets the analyst judge how much a conclusion
357
+ depends on it, rather than presenting an assumption as a measurement.
358
+
209
359
  ## Feature pairing (information)
210
360
 
211
361
  Beyond ranking single features, evaluate *pairs* of features together:
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "robustkit"
7
- version = "0.4.0"
7
+ version = "0.5.0"
8
8
  description = "Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty."
9
9
  readme = "README.md"
10
10
  license = { file = "LICENSE" }
@@ -13,11 +13,12 @@ Modules:
13
13
  robustkit.information -- mutual-information-based feature ranking
14
14
  """
15
15
 
16
- from .core.trend import fit_huber_trend, fit_tukey_trend, fit_ols_trend, predict_trend
16
+ from .core.trend import fit_huber_trend, fit_tukey_trend, fit_ols_trend, predict_trend, trend_derivative
17
17
  from .core.stability import model_stability_pct
18
18
  from .core.diagnostics import cooks_diagnostic, cook_impact
19
19
  from .core.uncertainty import bootstrap_band, bca_bootstrap_ci
20
20
  from .core.consistency import check_row_integrity, compare_row_sets
21
+ from .core.goodness_of_fit import goodness_of_fit, compare_polynomial_degrees
21
22
  from .segmentation.hierarchy import hierarchical_segment, segment_sizes
22
23
  from .segmentation.apply import apply_by_segment
23
24
  from .information.entropy import entropy
@@ -34,12 +35,21 @@ from .benchmark.robustness_map import feature_robustness_report, plot_feature_ro
34
35
  from .report.dispersion import iqr, dispersion_ratio, dispersion_by_bin
35
36
  from .report.visualize_analyst import plot_analyst_view
36
37
  from .report.visualize_publisher import plot_publisher_view
38
+ from .quantiles.io import load_scb_json_stat
39
+ from .quantiles.trend import prepare_quantile_trend, plot_quantile_trend, quantile_trend_dispersion
40
+ from .quantiles.reconstruct import (
41
+ expand_aggregated_group, expand_aggregated_table, check_reconstruction_quality,
42
+ expand_aggregated_group_flat, expand_aggregated_group_borrowed_dispersion,
43
+ expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
44
+ compare_reconstruction_methods,
45
+ )
37
46
 
38
47
  __all__ = [
39
48
  "fit_huber_trend",
40
49
  "fit_tukey_trend",
41
50
  "fit_ols_trend",
42
51
  "predict_trend",
52
+ "trend_derivative",
43
53
  "model_stability_pct",
44
54
  "cooks_diagnostic",
45
55
  "cook_impact",
@@ -47,6 +57,8 @@ __all__ = [
47
57
  "bca_bootstrap_ci",
48
58
  "check_row_integrity",
49
59
  "compare_row_sets",
60
+ "goodness_of_fit",
61
+ "compare_polynomial_degrees",
50
62
  "hierarchical_segment",
51
63
  "segment_sizes",
52
64
  "apply_by_segment",
@@ -74,6 +86,18 @@ __all__ = [
74
86
  "dispersion_by_bin",
75
87
  "plot_analyst_view",
76
88
  "plot_publisher_view",
89
+ "load_scb_json_stat",
90
+ "prepare_quantile_trend",
91
+ "plot_quantile_trend",
92
+ "quantile_trend_dispersion",
93
+ "expand_aggregated_group",
94
+ "expand_aggregated_table",
95
+ "check_reconstruction_quality",
96
+ "expand_aggregated_group_flat",
97
+ "expand_aggregated_group_borrowed_dispersion",
98
+ "expand_aggregated_table_flat",
99
+ "expand_aggregated_table_borrowed_dispersion",
100
+ "compare_reconstruction_methods",
77
101
  ]
78
102
 
79
103
  __version__ = "0.0.1"
@@ -50,8 +50,26 @@ def feature_robustness_report(df, target, features=None, degree=2,
50
50
  barely matters)
51
51
  fragile -- high stability spread AND high Cook
52
52
  impact (the least trustworthy)
53
+
54
+ Requires at least 2 features. Quadrant assignment is threshold-
55
+ based (median by default, see robustkit.classify_quadrants) --
56
+ with a single feature, that feature is trivially "at or above its
57
+ own median" on both axes, so it would always be classified
58
+ "fragile" regardless of its actual stability/impact values. This
59
+ is a property of median-based thresholding with n=1, not a
60
+ meaningful result, so it is rejected explicitly here rather than
61
+ silently returning a misleading label.
53
62
  """
54
63
  features = features or [c for c in df.columns if c != target]
64
+ if len(features) < 2:
65
+ raise ValueError(
66
+ f"feature_robustness_report requires at least 2 features to classify "
67
+ f"meaningfully (quadrant thresholds are computed across the candidate "
68
+ f"features); got {len(features)}: {features}. With a single feature, "
69
+ f"median-based thresholds are degenerate and always classify it as "
70
+ f"'fragile' regardless of its actual values."
71
+ )
72
+
55
73
  y = df[target].to_numpy(dtype=float)
56
74
 
57
75
  rows = []
@@ -14,7 +14,7 @@ dependency.
14
14
 
15
15
  import numpy as np
16
16
 
17
- from .trend import _design_matrix, fit_huber_trend, predict_trend
17
+ from .trend import _fit_design_matrix, fit_huber_trend, predict_trend
18
18
 
19
19
 
20
20
  def cooks_diagnostic(x, y, degree=2):
@@ -27,7 +27,7 @@ def cooks_diagnostic(x, y, degree=2):
27
27
  y = np.asarray(y, dtype=float)
28
28
  n = len(y)
29
29
 
30
- X, _ = _design_matrix(x, degree)
30
+ X, _, _ = _fit_design_matrix(x, degree)
31
31
  X_design = np.column_stack([np.ones(n), X])
32
32
  p = X_design.shape[1]
33
33
 
@@ -0,0 +1,77 @@
1
+ """
2
+ Measures of how well a fitted trend explains the variation in y, and a
3
+ diagnostic for choosing the polynomial degree empirically rather than
4
+ assuming a fixed degree fits every trend equally well.
5
+
6
+ Note on R² across methods: for Huber and Tukey fits, the R² computed
7
+ here is a descriptive "variance explained" measure derived from the
8
+ fit's predictions -- it is NOT the objective those methods actually
9
+ minimize (Huber/Tukey minimize a robust loss, not squared error). It
10
+ remains a useful, comparable summary of how much of y's variation the
11
+ fitted curve captures, consistently computed the same way across
12
+ methods and across polynomial degrees, which is what makes it useful
13
+ for comparison even though it isn't each method's "native" fit
14
+ statistic.
15
+ """
16
+
17
+ import numpy as np
18
+ import pandas as pd
19
+
20
+ from .trend import fit_huber_trend, fit_tukey_trend, fit_ols_trend, predict_trend
21
+
22
+ _FIT_FUNCTIONS = {
23
+ "huber": fit_huber_trend,
24
+ "tukey": fit_tukey_trend,
25
+ "ols": fit_ols_trend,
26
+ }
27
+
28
+
29
+ def goodness_of_fit(x, y, degree=2, method="huber"):
30
+ """
31
+ Fit a trend and report how well it explains the variation in y.
32
+
33
+ Returns r_squared (1 - var(residuals)/var(y)), plus RMSE and MAE
34
+ of residuals -- reported alongside R² since a single R²-style
35
+ number can hide whether errors are dominated by a few large misses
36
+ or spread evenly across observations.
37
+ """
38
+ x = np.asarray(x, dtype=float)
39
+ y = np.asarray(y, dtype=float)
40
+
41
+ fit_fn = _FIT_FUNCTIONS[method]
42
+ fit = fit_fn(x, y, degree=degree)
43
+ y_pred = predict_trend(fit, x)
44
+
45
+ residuals = y - y_pred
46
+ var_y = np.var(y)
47
+ r_squared = 1 - np.var(residuals) / var_y if var_y > 0 else 0.0
48
+
49
+ return {
50
+ "method": method,
51
+ "degree": degree,
52
+ "r_squared": float(r_squared),
53
+ "rmse": float(np.sqrt(np.mean(residuals ** 2))),
54
+ "mae": float(np.mean(np.abs(residuals))),
55
+ }
56
+
57
+
58
+ def compare_polynomial_degrees(x, y, degrees=(1, 2, 3, 4), method="huber"):
59
+ """
60
+ Fit the same (x, y) data at several polynomial degrees and report
61
+ goodness_of_fit for each, so the right degree can be chosen
62
+ empirically rather than assumed.
63
+
64
+ A quadratic trend (the package default) is a reasonable starting
65
+ point for many relationships, but not all -- some trends are more
66
+ volatile and need a higher degree to be captured honestly, while
67
+ an unnecessarily high degree risks fitting noise rather than a
68
+ real pattern. The r_squared_gain column shows how much each extra
69
+ degree actually buys you: a large gain going from degree 2 to 3
70
+ suggests the quadratic default is too restrictive for this trend;
71
+ a negligible gain suggests the extra complexity isn't earning its
72
+ keep.
73
+ """
74
+ rows = [goodness_of_fit(x, y, degree=d, method=method) for d in degrees]
75
+ df = pd.DataFrame(rows)
76
+ df["r_squared_gain"] = df["r_squared"].diff()
77
+ return df