robustkit 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {robustkit-0.4.0/robustkit.egg-info → robustkit-0.5.0}/PKG-INFO +153 -3
- {robustkit-0.4.0 → robustkit-0.5.0}/README.md +152 -2
- {robustkit-0.4.0 → robustkit-0.5.0}/pyproject.toml +1 -1
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/__init__.py +25 -1
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/benchmark/robustness_map.py +18 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/diagnostics.py +2 -2
- robustkit-0.5.0/robustkit/core/goodness_of_fit.py +77 -0
- robustkit-0.5.0/robustkit/core/trend.py +136 -0
- robustkit-0.5.0/robustkit/quantiles/io.py +104 -0
- robustkit-0.5.0/robustkit/quantiles/reconstruct.py +272 -0
- robustkit-0.5.0/robustkit/quantiles/trend.py +85 -0
- robustkit-0.5.0/robustkit/segmentation/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0/robustkit.egg-info}/PKG-INFO +153 -3
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/SOURCES.txt +10 -1
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_benchmark.py +23 -4
- robustkit-0.5.0/tests/test_quantiles_io_trend.py +126 -0
- robustkit-0.5.0/tests/test_quantiles_reconstruct.py +90 -0
- robustkit-0.5.0/tests/test_quantiles_reconstruct_mean_only.py +78 -0
- robustkit-0.5.0/tests/test_trend_extras.py +92 -0
- robustkit-0.4.0/robustkit/core/trend.py +0 -86
- {robustkit-0.4.0 → robustkit-0.5.0}/LICENSE +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/benchmark/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/benchmark/global_model.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/common/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/common/quadrants.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/consistency.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/stability.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/core/uncertainty.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/communication.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/conditional_mi.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/entropy.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/mutual_info.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/pairs.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/profile.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/quadrants.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/utils.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/information/visualization.py +0 -0
- {robustkit-0.4.0/robustkit/report → robustkit-0.5.0/robustkit/quantiles}/__init__.py +0 -0
- {robustkit-0.4.0/robustkit/segmentation → robustkit-0.5.0/robustkit/report}/__init__.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/report/dispersion.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/report/visualize_analyst.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/report/visualize_publisher.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/segmentation/apply.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit/segmentation/hierarchy.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/dependency_links.txt +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/requires.txt +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/robustkit.egg-info/top_level.txt +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/setup.cfg +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_common_quadrants.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_core.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_information.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_information_pairs.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_report.py +0 -0
- {robustkit-0.4.0 → robustkit-0.5.0}/tests/test_segmentation.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: robustkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
|
|
5
5
|
Author: Mikael Lundqvist
|
|
6
6
|
License: MIT License
|
|
@@ -60,8 +60,11 @@ consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
|
60
60
|
per-segment analysis), `robustkit.information` (mutual-information
|
|
61
61
|
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
62
62
|
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
63
|
-
Robustness Map),
|
|
64
|
-
dispersion measures)
|
|
63
|
+
Robustness Map), `robustkit.report` (analyst vs. publisher views,
|
|
64
|
+
dispersion measures), and `robustkit.quantiles` (generic JSON-stat
|
|
65
|
+
loading, published-quantile-trend visualization, and lognormal-
|
|
66
|
+
calibrated reconstruction of individual-level data from aggregated
|
|
67
|
+
summaries) are stable and tested.
|
|
65
68
|
|
|
66
69
|
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
67
70
|
continuous features. `mutual_information` is estimated on the
|
|
@@ -111,6 +114,32 @@ ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
|
|
|
111
114
|
|
|
112
115
|
See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
|
|
113
116
|
|
|
117
|
+
## Trend growth rate and goodness of fit
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
|
|
121
|
+
|
|
122
|
+
fit = fit_huber_trend(df["age"], df["salary"])
|
|
123
|
+
|
|
124
|
+
# Rate of change of the trend itself (e.g. "salary growth per year of
|
|
125
|
+
# age"), not just its level
|
|
126
|
+
rates = trend_derivative(fit, x=[30, 40, 50])
|
|
127
|
+
|
|
128
|
+
# How well does this fit actually explain the variation in y?
|
|
129
|
+
goodness_of_fit(df["age"], df["salary"], degree=2)
|
|
130
|
+
|
|
131
|
+
# Don't assume a quadratic trend is always the right choice -- check
|
|
132
|
+
# empirically whether a higher degree captures meaningfully more
|
|
133
|
+
compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
**Note:** x is standardized internally before building polynomial
|
|
137
|
+
features (both here and throughout `robustkit.core`), since raw
|
|
138
|
+
polynomial features become numerically unstable at higher degrees for
|
|
139
|
+
realistic x scales (e.g. age^5 vastly outscales age^1). This is
|
|
140
|
+
transparent to callers -- `predict_trend` and `trend_derivative` still
|
|
141
|
+
take and return values in the original x scale.
|
|
142
|
+
|
|
114
143
|
## Segmentation
|
|
115
144
|
|
|
116
145
|
Run any `robustkit.core` analysis independently across subgroups of a
|
|
@@ -246,6 +275,127 @@ standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
|
246
275
|
computes both across bins of a continuous x, e.g. to check whether
|
|
247
276
|
dispersion (inequality) grows with age.
|
|
248
277
|
|
|
278
|
+
## Loading published quantile tables (SCB / JSON-stat)
|
|
279
|
+
|
|
280
|
+
Some statistics agencies (e.g. Statistics Sweden, SCB) publish
|
|
281
|
+
quantiles (Q1/median/Q3) directly, with no individual-level data
|
|
282
|
+
available at all. `robustkit.quantiles` loads these tables generically
|
|
283
|
+
via JSON-stat, a standardized dimensional-data format used by SCB and
|
|
284
|
+
other national statistics agencies -- avoiding the fragility of
|
|
285
|
+
parsing metadata out of column-name strings in a wide CSV export.
|
|
286
|
+
|
|
287
|
+
```python
|
|
288
|
+
from robustkit import load_scb_json_stat, plot_quantile_trend, quantile_trend_dispersion
|
|
289
|
+
|
|
290
|
+
df = load_scb_json_stat("some_scb_table.json")
|
|
291
|
+
|
|
292
|
+
# A real SCB quirk this loader does NOT try to guess automatically:
|
|
293
|
+
# category labels can change meaning over time (e.g. Sweden's oldest
|
|
294
|
+
# working-age bracket was labeled "65-66 år" through 2022 and
|
|
295
|
+
# "65-68 år" from 2023, following a pension-age reform). Merge such
|
|
296
|
+
# cases explicitly:
|
|
297
|
+
df = load_scb_json_stat(
|
|
298
|
+
"some_scb_table.json",
|
|
299
|
+
rename_categories={"ålder": {"65–68 år": "65–66 år"}},
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
# Once reshaped to a wide table with q1/median/q3 columns:
|
|
303
|
+
plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
304
|
+
quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
305
|
+
```
|
|
306
|
+
|
|
307
|
+
This is the "quantiles are already given" case. A complementary case
|
|
308
|
+
-- reconstructing approximate individual-level data from aggregated
|
|
309
|
+
group means, for when only summary statistics (not quantiles) are
|
|
310
|
+
available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
|
|
311
|
+
|
|
312
|
+
See `examples/quantiles_tutorial.py` for a complete walkthrough.
|
|
313
|
+
|
|
314
|
+
## Reconstructing individual-level data from aggregated summaries
|
|
315
|
+
|
|
316
|
+
For the complementary case -- only aggregated group summaries (n,
|
|
317
|
+
Q1, median, Q3) are available, not the quantile trend itself as the
|
|
318
|
+
final answer, and you want to run `robustkit.core` analyses as if
|
|
319
|
+
individual data existed:
|
|
320
|
+
|
|
321
|
+
```python
|
|
322
|
+
from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
|
|
323
|
+
|
|
324
|
+
# One row per group (e.g. year), with n/q1/median/q3 columns
|
|
325
|
+
synthetic = expand_aggregated_table(
|
|
326
|
+
summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
|
|
327
|
+
group_cols=["year"], value_name="salary",
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
# Now usable exactly like real individual-level data:
|
|
331
|
+
fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
Method: a lognormal distribution is calibrated (via the IQR) to match
|
|
335
|
+
each group's reported Q1/median/Q3, then `n` synthetic values are
|
|
336
|
+
drawn from it. Validated end-to-end against real published SCB salary
|
|
337
|
+
data: a Huber trend fitted on reconstructed pseudo-individual data
|
|
338
|
+
tracked the true published median trend within 2% across 12 years.
|
|
339
|
+
|
|
340
|
+
**Note on what this recovers:** because a Huber (or Tukey) fit on
|
|
341
|
+
right-skewed reconstructed data tracks something close to the
|
|
342
|
+
*median* trend it was calibrated against -- not the arithmetic mean --
|
|
343
|
+
this is consistent with, not a limitation of, the reconstruction
|
|
344
|
+
method. To target the mean instead, fit on `log(value)` and
|
|
345
|
+
exponentiate predictions back, which approximates the geometric mean.
|
|
346
|
+
|
|
347
|
+
Always check `check_reconstruction_quality()` before trusting a
|
|
348
|
+
reconstruction: real Q1/median/Q3 triples aren't always perfectly
|
|
349
|
+
consistent with a pure lognormal shape.
|
|
350
|
+
|
|
351
|
+
**Warning -- unbounded tail at large n:** a lognormal has no natural
|
|
352
|
+
upper limit, and its expected maximum grows with n. Reconstructing at
|
|
353
|
+
the TRUE group size from a national table (SCB salary tables can
|
|
354
|
+
report n in the hundreds of thousands to millions) can produce
|
|
355
|
+
implausibly extreme tail values -- real salaries have practical
|
|
356
|
+
ceilings a pure lognormal doesn't know about. This package's own
|
|
357
|
+
examples and tests deliberately scale n down to a few thousand for
|
|
358
|
+
demonstration; calibration quality (matching Q1/median/Q3) doesn't
|
|
359
|
+
depend on reproducing the true population size, but tail plausibility
|
|
360
|
+
does. No clipping is applied automatically.
|
|
361
|
+
|
|
362
|
+
### When only a mean is available (no quantiles at all)
|
|
363
|
+
|
|
364
|
+
Some tables (e.g. SCB's age-breakdown salary tables) report only a
|
|
365
|
+
mean per group, with no spread information. Two deliberately separate
|
|
366
|
+
methods are provided, each making a different explicit assumption --
|
|
367
|
+
compare them rather than silently picking one:
|
|
368
|
+
|
|
369
|
+
```python
|
|
370
|
+
from robustkit import (
|
|
371
|
+
expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
|
|
372
|
+
compare_reconstruction_methods,
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
# Method 1: repeat the mean n times -- zero within-group spread.
|
|
376
|
+
# Recovers between-group regression coefficients reasonably well
|
|
377
|
+
# (validated in the original technique this is based on) but
|
|
378
|
+
# understates individual-level variation.
|
|
379
|
+
flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
|
|
380
|
+
|
|
381
|
+
# Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
|
|
382
|
+
# report quantiles, and use it to imply an approximate spread around
|
|
383
|
+
# the mean. Stacks two assumptions (mean-as-median, and that the
|
|
384
|
+
# borrowed ratio transfers to this population) -- illustrative, not a
|
|
385
|
+
# substitute for genuine quantile data for this specific table.
|
|
386
|
+
borrowed = expand_aggregated_table_borrowed_dispersion(
|
|
387
|
+
df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
|
|
388
|
+
)
|
|
389
|
+
|
|
390
|
+
# Compare both for a single group directly:
|
|
391
|
+
compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
|
|
392
|
+
```
|
|
393
|
+
|
|
394
|
+
Both methods are documented with their specific assumptions rather
|
|
395
|
+
than presented as equally valid defaults -- being explicit about which
|
|
396
|
+
assumption was made lets the analyst judge how much a conclusion
|
|
397
|
+
depends on it, rather than presenting an assumption as a measurement.
|
|
398
|
+
|
|
249
399
|
## Feature pairing (information)
|
|
250
400
|
|
|
251
401
|
Beyond ranking single features, evaluate *pairs* of features together:
|
|
@@ -20,8 +20,11 @@ consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
|
20
20
|
per-segment analysis), `robustkit.information` (mutual-information
|
|
21
21
|
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
22
22
|
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
23
|
-
Robustness Map),
|
|
24
|
-
dispersion measures)
|
|
23
|
+
Robustness Map), `robustkit.report` (analyst vs. publisher views,
|
|
24
|
+
dispersion measures), and `robustkit.quantiles` (generic JSON-stat
|
|
25
|
+
loading, published-quantile-trend visualization, and lognormal-
|
|
26
|
+
calibrated reconstruction of individual-level data from aggregated
|
|
27
|
+
summaries) are stable and tested.
|
|
25
28
|
|
|
26
29
|
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
27
30
|
continuous features. `mutual_information` is estimated on the
|
|
@@ -71,6 +74,32 @@ ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
|
|
|
71
74
|
|
|
72
75
|
See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
|
|
73
76
|
|
|
77
|
+
## Trend growth rate and goodness of fit
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
from robustkit import trend_derivative, goodness_of_fit, compare_polynomial_degrees
|
|
81
|
+
|
|
82
|
+
fit = fit_huber_trend(df["age"], df["salary"])
|
|
83
|
+
|
|
84
|
+
# Rate of change of the trend itself (e.g. "salary growth per year of
|
|
85
|
+
# age"), not just its level
|
|
86
|
+
rates = trend_derivative(fit, x=[30, 40, 50])
|
|
87
|
+
|
|
88
|
+
# How well does this fit actually explain the variation in y?
|
|
89
|
+
goodness_of_fit(df["age"], df["salary"], degree=2)
|
|
90
|
+
|
|
91
|
+
# Don't assume a quadratic trend is always the right choice -- check
|
|
92
|
+
# empirically whether a higher degree captures meaningfully more
|
|
93
|
+
compare_polynomial_degrees(df["age"], df["salary"], degrees=(1, 2, 3, 4))
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
**Note:** x is standardized internally before building polynomial
|
|
97
|
+
features (both here and throughout `robustkit.core`), since raw
|
|
98
|
+
polynomial features become numerically unstable at higher degrees for
|
|
99
|
+
realistic x scales (e.g. age^5 vastly outscales age^1). This is
|
|
100
|
+
transparent to callers -- `predict_trend` and `trend_derivative` still
|
|
101
|
+
take and return values in the original x scale.
|
|
102
|
+
|
|
74
103
|
## Segmentation
|
|
75
104
|
|
|
76
105
|
Run any `robustkit.core` analysis independently across subgroups of a
|
|
@@ -206,6 +235,127 @@ standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
|
206
235
|
computes both across bins of a continuous x, e.g. to check whether
|
|
207
236
|
dispersion (inequality) grows with age.
|
|
208
237
|
|
|
238
|
+
## Loading published quantile tables (SCB / JSON-stat)
|
|
239
|
+
|
|
240
|
+
Some statistics agencies (e.g. Statistics Sweden, SCB) publish
|
|
241
|
+
quantiles (Q1/median/Q3) directly, with no individual-level data
|
|
242
|
+
available at all. `robustkit.quantiles` loads these tables generically
|
|
243
|
+
via JSON-stat, a standardized dimensional-data format used by SCB and
|
|
244
|
+
other national statistics agencies -- avoiding the fragility of
|
|
245
|
+
parsing metadata out of column-name strings in a wide CSV export.
|
|
246
|
+
|
|
247
|
+
```python
|
|
248
|
+
from robustkit import load_scb_json_stat, plot_quantile_trend, quantile_trend_dispersion
|
|
249
|
+
|
|
250
|
+
df = load_scb_json_stat("some_scb_table.json")
|
|
251
|
+
|
|
252
|
+
# A real SCB quirk this loader does NOT try to guess automatically:
|
|
253
|
+
# category labels can change meaning over time (e.g. Sweden's oldest
|
|
254
|
+
# working-age bracket was labeled "65-66 år" through 2022 and
|
|
255
|
+
# "65-68 år" from 2023, following a pension-age reform). Merge such
|
|
256
|
+
# cases explicitly:
|
|
257
|
+
df = load_scb_json_stat(
|
|
258
|
+
"some_scb_table.json",
|
|
259
|
+
rename_categories={"ålder": {"65–68 år": "65–66 år"}},
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
# Once reshaped to a wide table with q1/median/q3 columns:
|
|
263
|
+
plot_quantile_trend(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
264
|
+
quantile_trend_dispersion(wide_df, x_col="år", q1_col="q1", median_col="median", q3_col="q3")
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
This is the "quantiles are already given" case. A complementary case
|
|
268
|
+
-- reconstructing approximate individual-level data from aggregated
|
|
269
|
+
group means, for when only summary statistics (not quantiles) are
|
|
270
|
+
available -- is planned as a follow-up (`robustkit.quantiles.reconstruct`).
|
|
271
|
+
|
|
272
|
+
See `examples/quantiles_tutorial.py` for a complete walkthrough.
|
|
273
|
+
|
|
274
|
+
## Reconstructing individual-level data from aggregated summaries
|
|
275
|
+
|
|
276
|
+
For the complementary case -- only aggregated group summaries (n,
|
|
277
|
+
Q1, median, Q3) are available, not the quantile trend itself as the
|
|
278
|
+
final answer, and you want to run `robustkit.core` analyses as if
|
|
279
|
+
individual data existed:
|
|
280
|
+
|
|
281
|
+
```python
|
|
282
|
+
from robustkit import expand_aggregated_table, check_reconstruction_quality, fit_huber_trend
|
|
283
|
+
|
|
284
|
+
# One row per group (e.g. year), with n/q1/median/q3 columns
|
|
285
|
+
synthetic = expand_aggregated_table(
|
|
286
|
+
summary_df, n_col="n", q1_col="q1", median_col="median", q3_col="q3",
|
|
287
|
+
group_cols=["year"], value_name="salary",
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
# Now usable exactly like real individual-level data:
|
|
291
|
+
fit = fit_huber_trend(synthetic["year"], synthetic["salary"])
|
|
292
|
+
```
|
|
293
|
+
|
|
294
|
+
Method: a lognormal distribution is calibrated (via the IQR) to match
|
|
295
|
+
each group's reported Q1/median/Q3, then `n` synthetic values are
|
|
296
|
+
drawn from it. Validated end-to-end against real published SCB salary
|
|
297
|
+
data: a Huber trend fitted on reconstructed pseudo-individual data
|
|
298
|
+
tracked the true published median trend within 2% across 12 years.
|
|
299
|
+
|
|
300
|
+
**Note on what this recovers:** because a Huber (or Tukey) fit on
|
|
301
|
+
right-skewed reconstructed data tracks something close to the
|
|
302
|
+
*median* trend it was calibrated against -- not the arithmetic mean --
|
|
303
|
+
this is consistent with, not a limitation of, the reconstruction
|
|
304
|
+
method. To target the mean instead, fit on `log(value)` and
|
|
305
|
+
exponentiate predictions back, which approximates the geometric mean.
|
|
306
|
+
|
|
307
|
+
Always check `check_reconstruction_quality()` before trusting a
|
|
308
|
+
reconstruction: real Q1/median/Q3 triples aren't always perfectly
|
|
309
|
+
consistent with a pure lognormal shape.
|
|
310
|
+
|
|
311
|
+
**Warning -- unbounded tail at large n:** a lognormal has no natural
|
|
312
|
+
upper limit, and its expected maximum grows with n. Reconstructing at
|
|
313
|
+
the TRUE group size from a national table (SCB salary tables can
|
|
314
|
+
report n in the hundreds of thousands to millions) can produce
|
|
315
|
+
implausibly extreme tail values -- real salaries have practical
|
|
316
|
+
ceilings a pure lognormal doesn't know about. This package's own
|
|
317
|
+
examples and tests deliberately scale n down to a few thousand for
|
|
318
|
+
demonstration; calibration quality (matching Q1/median/Q3) doesn't
|
|
319
|
+
depend on reproducing the true population size, but tail plausibility
|
|
320
|
+
does. No clipping is applied automatically.
|
|
321
|
+
|
|
322
|
+
### When only a mean is available (no quantiles at all)
|
|
323
|
+
|
|
324
|
+
Some tables (e.g. SCB's age-breakdown salary tables) report only a
|
|
325
|
+
mean per group, with no spread information. Two deliberately separate
|
|
326
|
+
methods are provided, each making a different explicit assumption --
|
|
327
|
+
compare them rather than silently picking one:
|
|
328
|
+
|
|
329
|
+
```python
|
|
330
|
+
from robustkit import (
|
|
331
|
+
expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
|
|
332
|
+
compare_reconstruction_methods,
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
# Method 1: repeat the mean n times -- zero within-group spread.
|
|
336
|
+
# Recovers between-group regression coefficients reasonably well
|
|
337
|
+
# (validated in the original technique this is based on) but
|
|
338
|
+
# understates individual-level variation.
|
|
339
|
+
flat = expand_aggregated_table_flat(df, n_col="n", mean_col="mean_salary", group_cols=["age"])
|
|
340
|
+
|
|
341
|
+
# Method 2: borrow a dispersion_ratio from a DIFFERENT table that does
|
|
342
|
+
# report quantiles, and use it to imply an approximate spread around
|
|
343
|
+
# the mean. Stacks two assumptions (mean-as-median, and that the
|
|
344
|
+
# borrowed ratio transfers to this population) -- illustrative, not a
|
|
345
|
+
# substitute for genuine quantile data for this specific table.
|
|
346
|
+
borrowed = expand_aggregated_table_borrowed_dispersion(
|
|
347
|
+
df, n_col="n", mean_col="mean_salary", dispersion_ratio=0.45, group_cols=["age"],
|
|
348
|
+
)
|
|
349
|
+
|
|
350
|
+
# Compare both for a single group directly:
|
|
351
|
+
compare_reconstruction_methods(n=2000, mean=52200, dispersion_ratio=0.45)
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
Both methods are documented with their specific assumptions rather
|
|
355
|
+
than presented as equally valid defaults -- being explicit about which
|
|
356
|
+
assumption was made lets the analyst judge how much a conclusion
|
|
357
|
+
depends on it, rather than presenting an assumption as a measurement.
|
|
358
|
+
|
|
209
359
|
## Feature pairing (information)
|
|
210
360
|
|
|
211
361
|
Beyond ranking single features, evaluate *pairs* of features together:
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "robustkit"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.5.0"
|
|
8
8
|
description = "Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { file = "LICENSE" }
|
|
@@ -13,11 +13,12 @@ Modules:
|
|
|
13
13
|
robustkit.information -- mutual-information-based feature ranking
|
|
14
14
|
"""
|
|
15
15
|
|
|
16
|
-
from .core.trend import fit_huber_trend, fit_tukey_trend, fit_ols_trend, predict_trend
|
|
16
|
+
from .core.trend import fit_huber_trend, fit_tukey_trend, fit_ols_trend, predict_trend, trend_derivative
|
|
17
17
|
from .core.stability import model_stability_pct
|
|
18
18
|
from .core.diagnostics import cooks_diagnostic, cook_impact
|
|
19
19
|
from .core.uncertainty import bootstrap_band, bca_bootstrap_ci
|
|
20
20
|
from .core.consistency import check_row_integrity, compare_row_sets
|
|
21
|
+
from .core.goodness_of_fit import goodness_of_fit, compare_polynomial_degrees
|
|
21
22
|
from .segmentation.hierarchy import hierarchical_segment, segment_sizes
|
|
22
23
|
from .segmentation.apply import apply_by_segment
|
|
23
24
|
from .information.entropy import entropy
|
|
@@ -34,12 +35,21 @@ from .benchmark.robustness_map import feature_robustness_report, plot_feature_ro
|
|
|
34
35
|
from .report.dispersion import iqr, dispersion_ratio, dispersion_by_bin
|
|
35
36
|
from .report.visualize_analyst import plot_analyst_view
|
|
36
37
|
from .report.visualize_publisher import plot_publisher_view
|
|
38
|
+
from .quantiles.io import load_scb_json_stat
|
|
39
|
+
from .quantiles.trend import prepare_quantile_trend, plot_quantile_trend, quantile_trend_dispersion
|
|
40
|
+
from .quantiles.reconstruct import (
|
|
41
|
+
expand_aggregated_group, expand_aggregated_table, check_reconstruction_quality,
|
|
42
|
+
expand_aggregated_group_flat, expand_aggregated_group_borrowed_dispersion,
|
|
43
|
+
expand_aggregated_table_flat, expand_aggregated_table_borrowed_dispersion,
|
|
44
|
+
compare_reconstruction_methods,
|
|
45
|
+
)
|
|
37
46
|
|
|
38
47
|
__all__ = [
|
|
39
48
|
"fit_huber_trend",
|
|
40
49
|
"fit_tukey_trend",
|
|
41
50
|
"fit_ols_trend",
|
|
42
51
|
"predict_trend",
|
|
52
|
+
"trend_derivative",
|
|
43
53
|
"model_stability_pct",
|
|
44
54
|
"cooks_diagnostic",
|
|
45
55
|
"cook_impact",
|
|
@@ -47,6 +57,8 @@ __all__ = [
|
|
|
47
57
|
"bca_bootstrap_ci",
|
|
48
58
|
"check_row_integrity",
|
|
49
59
|
"compare_row_sets",
|
|
60
|
+
"goodness_of_fit",
|
|
61
|
+
"compare_polynomial_degrees",
|
|
50
62
|
"hierarchical_segment",
|
|
51
63
|
"segment_sizes",
|
|
52
64
|
"apply_by_segment",
|
|
@@ -74,6 +86,18 @@ __all__ = [
|
|
|
74
86
|
"dispersion_by_bin",
|
|
75
87
|
"plot_analyst_view",
|
|
76
88
|
"plot_publisher_view",
|
|
89
|
+
"load_scb_json_stat",
|
|
90
|
+
"prepare_quantile_trend",
|
|
91
|
+
"plot_quantile_trend",
|
|
92
|
+
"quantile_trend_dispersion",
|
|
93
|
+
"expand_aggregated_group",
|
|
94
|
+
"expand_aggregated_table",
|
|
95
|
+
"check_reconstruction_quality",
|
|
96
|
+
"expand_aggregated_group_flat",
|
|
97
|
+
"expand_aggregated_group_borrowed_dispersion",
|
|
98
|
+
"expand_aggregated_table_flat",
|
|
99
|
+
"expand_aggregated_table_borrowed_dispersion",
|
|
100
|
+
"compare_reconstruction_methods",
|
|
77
101
|
]
|
|
78
102
|
|
|
79
103
|
__version__ = "0.0.1"
|
|
@@ -50,8 +50,26 @@ def feature_robustness_report(df, target, features=None, degree=2,
|
|
|
50
50
|
barely matters)
|
|
51
51
|
fragile -- high stability spread AND high Cook
|
|
52
52
|
impact (the least trustworthy)
|
|
53
|
+
|
|
54
|
+
Requires at least 2 features. Quadrant assignment is threshold-
|
|
55
|
+
based (median by default, see robustkit.classify_quadrants) --
|
|
56
|
+
with a single feature, that feature is trivially "at or above its
|
|
57
|
+
own median" on both axes, so it would always be classified
|
|
58
|
+
"fragile" regardless of its actual stability/impact values. This
|
|
59
|
+
is a property of median-based thresholding with n=1, not a
|
|
60
|
+
meaningful result, so it is rejected explicitly here rather than
|
|
61
|
+
silently returning a misleading label.
|
|
53
62
|
"""
|
|
54
63
|
features = features or [c for c in df.columns if c != target]
|
|
64
|
+
if len(features) < 2:
|
|
65
|
+
raise ValueError(
|
|
66
|
+
f"feature_robustness_report requires at least 2 features to classify "
|
|
67
|
+
f"meaningfully (quadrant thresholds are computed across the candidate "
|
|
68
|
+
f"features); got {len(features)}: {features}. With a single feature, "
|
|
69
|
+
f"median-based thresholds are degenerate and always classify it as "
|
|
70
|
+
f"'fragile' regardless of its actual values."
|
|
71
|
+
)
|
|
72
|
+
|
|
55
73
|
y = df[target].to_numpy(dtype=float)
|
|
56
74
|
|
|
57
75
|
rows = []
|
|
@@ -14,7 +14,7 @@ dependency.
|
|
|
14
14
|
|
|
15
15
|
import numpy as np
|
|
16
16
|
|
|
17
|
-
from .trend import
|
|
17
|
+
from .trend import _fit_design_matrix, fit_huber_trend, predict_trend
|
|
18
18
|
|
|
19
19
|
|
|
20
20
|
def cooks_diagnostic(x, y, degree=2):
|
|
@@ -27,7 +27,7 @@ def cooks_diagnostic(x, y, degree=2):
|
|
|
27
27
|
y = np.asarray(y, dtype=float)
|
|
28
28
|
n = len(y)
|
|
29
29
|
|
|
30
|
-
X, _ =
|
|
30
|
+
X, _, _ = _fit_design_matrix(x, degree)
|
|
31
31
|
X_design = np.column_stack([np.ones(n), X])
|
|
32
32
|
p = X_design.shape[1]
|
|
33
33
|
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Measures of how well a fitted trend explains the variation in y, and a
|
|
3
|
+
diagnostic for choosing the polynomial degree empirically rather than
|
|
4
|
+
assuming a fixed degree fits every trend equally well.
|
|
5
|
+
|
|
6
|
+
Note on R² across methods: for Huber and Tukey fits, the R² computed
|
|
7
|
+
here is a descriptive "variance explained" measure derived from the
|
|
8
|
+
fit's predictions -- it is NOT the objective those methods actually
|
|
9
|
+
minimize (Huber/Tukey minimize a robust loss, not squared error). It
|
|
10
|
+
remains a useful, comparable summary of how much of y's variation the
|
|
11
|
+
fitted curve captures, consistently computed the same way across
|
|
12
|
+
methods and across polynomial degrees, which is what makes it useful
|
|
13
|
+
for comparison even though it isn't each method's "native" fit
|
|
14
|
+
statistic.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import pandas as pd
|
|
19
|
+
|
|
20
|
+
from .trend import fit_huber_trend, fit_tukey_trend, fit_ols_trend, predict_trend
|
|
21
|
+
|
|
22
|
+
_FIT_FUNCTIONS = {
|
|
23
|
+
"huber": fit_huber_trend,
|
|
24
|
+
"tukey": fit_tukey_trend,
|
|
25
|
+
"ols": fit_ols_trend,
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def goodness_of_fit(x, y, degree=2, method="huber"):
|
|
30
|
+
"""
|
|
31
|
+
Fit a trend and report how well it explains the variation in y.
|
|
32
|
+
|
|
33
|
+
Returns r_squared (1 - var(residuals)/var(y)), plus RMSE and MAE
|
|
34
|
+
of residuals -- reported alongside R² since a single R²-style
|
|
35
|
+
number can hide whether errors are dominated by a few large misses
|
|
36
|
+
or spread evenly across observations.
|
|
37
|
+
"""
|
|
38
|
+
x = np.asarray(x, dtype=float)
|
|
39
|
+
y = np.asarray(y, dtype=float)
|
|
40
|
+
|
|
41
|
+
fit_fn = _FIT_FUNCTIONS[method]
|
|
42
|
+
fit = fit_fn(x, y, degree=degree)
|
|
43
|
+
y_pred = predict_trend(fit, x)
|
|
44
|
+
|
|
45
|
+
residuals = y - y_pred
|
|
46
|
+
var_y = np.var(y)
|
|
47
|
+
r_squared = 1 - np.var(residuals) / var_y if var_y > 0 else 0.0
|
|
48
|
+
|
|
49
|
+
return {
|
|
50
|
+
"method": method,
|
|
51
|
+
"degree": degree,
|
|
52
|
+
"r_squared": float(r_squared),
|
|
53
|
+
"rmse": float(np.sqrt(np.mean(residuals ** 2))),
|
|
54
|
+
"mae": float(np.mean(np.abs(residuals))),
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def compare_polynomial_degrees(x, y, degrees=(1, 2, 3, 4), method="huber"):
|
|
59
|
+
"""
|
|
60
|
+
Fit the same (x, y) data at several polynomial degrees and report
|
|
61
|
+
goodness_of_fit for each, so the right degree can be chosen
|
|
62
|
+
empirically rather than assumed.
|
|
63
|
+
|
|
64
|
+
A quadratic trend (the package default) is a reasonable starting
|
|
65
|
+
point for many relationships, but not all -- some trends are more
|
|
66
|
+
volatile and need a higher degree to be captured honestly, while
|
|
67
|
+
an unnecessarily high degree risks fitting noise rather than a
|
|
68
|
+
real pattern. The r_squared_gain column shows how much each extra
|
|
69
|
+
degree actually buys you: a large gain going from degree 2 to 3
|
|
70
|
+
suggests the quadratic default is too restrictive for this trend;
|
|
71
|
+
a negligible gain suggests the extra complexity isn't earning its
|
|
72
|
+
keep.
|
|
73
|
+
"""
|
|
74
|
+
rows = [goodness_of_fit(x, y, degree=d, method=method) for d in degrees]
|
|
75
|
+
df = pd.DataFrame(rows)
|
|
76
|
+
df["r_squared_gain"] = df["r_squared"].diff()
|
|
77
|
+
return df
|