robustkit 0.0.1__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {robustkit-0.0.1/robustkit.egg-info → robustkit-0.4.0}/PKG-INFO +132 -5
- robustkit-0.4.0/README.md +256 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/pyproject.toml +1 -1
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/__init__.py +25 -0
- robustkit-0.4.0/robustkit/benchmark/global_model.py +74 -0
- robustkit-0.4.0/robustkit/benchmark/robustness_map.py +142 -0
- robustkit-0.4.0/robustkit/common/quadrants.py +57 -0
- robustkit-0.4.0/robustkit/information/__init__.py +0 -0
- robustkit-0.4.0/robustkit/information/communication.py +143 -0
- robustkit-0.4.0/robustkit/information/conditional_mi.py +50 -0
- robustkit-0.4.0/robustkit/information/mutual_info.py +172 -0
- robustkit-0.4.0/robustkit/information/pairs.py +88 -0
- robustkit-0.4.0/robustkit/information/quadrants.py +56 -0
- robustkit-0.4.0/robustkit/information/utils.py +23 -0
- robustkit-0.4.0/robustkit/report/__init__.py +0 -0
- robustkit-0.4.0/robustkit/report/dispersion.py +68 -0
- robustkit-0.4.0/robustkit/report/visualize_analyst.py +52 -0
- robustkit-0.4.0/robustkit/report/visualize_publisher.py +57 -0
- robustkit-0.4.0/robustkit/segmentation/__init__.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0/robustkit.egg-info}/PKG-INFO +132 -5
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit.egg-info/SOURCES.txt +17 -0
- robustkit-0.4.0/tests/test_benchmark.py +99 -0
- robustkit-0.4.0/tests/test_common_quadrants.py +30 -0
- robustkit-0.4.0/tests/test_information_pairs.py +122 -0
- robustkit-0.4.0/tests/test_report.py +93 -0
- robustkit-0.0.1/README.md +0 -129
- robustkit-0.0.1/robustkit/information/mutual_info.py +0 -87
- robustkit-0.0.1/robustkit/information/quadrants.py +0 -67
- {robustkit-0.0.1 → robustkit-0.4.0}/LICENSE +0 -0
- {robustkit-0.0.1/robustkit/core → robustkit-0.4.0/robustkit/benchmark}/__init__.py +0 -0
- {robustkit-0.0.1/robustkit/information → robustkit-0.4.0/robustkit/common}/__init__.py +0 -0
- {robustkit-0.0.1/robustkit/segmentation → robustkit-0.4.0/robustkit/core}/__init__.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/core/consistency.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/core/diagnostics.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/core/stability.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/core/trend.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/core/uncertainty.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/information/entropy.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/information/profile.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/information/visualization.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/segmentation/apply.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit/segmentation/hierarchy.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit.egg-info/dependency_links.txt +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit.egg-info/requires.txt +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/robustkit.egg-info/top_level.txt +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/setup.cfg +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/tests/test_core.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/tests/test_information.py +0 -0
- {robustkit-0.0.1 → robustkit-0.4.0}/tests/test_segmentation.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: robustkit
|
|
3
|
-
Version: 0.0
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty.
|
|
5
5
|
Author: Mikael Lundqvist
|
|
6
6
|
License: MIT License
|
|
@@ -57,10 +57,21 @@ observations, and get honest, bias-corrected uncertainty estimates.
|
|
|
57
57
|
|
|
58
58
|
`robustkit.core` (trend fitting, stability, diagnostics, uncertainty,
|
|
59
59
|
consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
60
|
-
per-segment analysis),
|
|
61
|
-
feature ranking
|
|
62
|
-
|
|
63
|
-
|
|
60
|
+
per-segment analysis), `robustkit.information` (mutual-information
|
|
61
|
+
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
62
|
+
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
63
|
+
Robustness Map), and `robustkit.report` (analyst vs. publisher views,
|
|
64
|
+
dispersion measures) are stable and tested.
|
|
65
|
+
|
|
66
|
+
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
67
|
+
continuous features. `mutual_information` is estimated on the
|
|
68
|
+
full-resolution continuous values, while `entropy_bits` is computed on
|
|
69
|
+
a binned version of the same feature (since `entropy()` expects
|
|
70
|
+
categorical input). Binning discards information, so `entropy_bits` is
|
|
71
|
+
a lower bound on the feature's true entropy -- an efficiency above 1.0
|
|
72
|
+
signals that the feature carries more usable information than a coarse
|
|
73
|
+
categorical summary of it would capture. This is expected behavior,
|
|
74
|
+
not a bug.
|
|
64
75
|
|
|
65
76
|
## Installation
|
|
66
77
|
|
|
@@ -153,6 +164,122 @@ quadrant label alone.
|
|
|
153
164
|
|
|
154
165
|
See `examples/information_tutorial.py` for a complete walkthrough.
|
|
155
166
|
|
|
167
|
+
## Benchmarking against a global trend
|
|
168
|
+
|
|
169
|
+
Compare each segment's observed outcome against what a single global
|
|
170
|
+
robust trend predicts, with bootstrap uncertainty on the difference --
|
|
171
|
+
answers "which groups deviate from the overall trend, and by how
|
|
172
|
+
much?" rather than "how does the trend look overall?":
|
|
173
|
+
|
|
174
|
+
```python
|
|
175
|
+
from robustkit import segment_position_report
|
|
176
|
+
|
|
177
|
+
report = segment_position_report(
|
|
178
|
+
df, segment_col="department", x_col="age", y_col="salary",
|
|
179
|
+
)
|
|
180
|
+
# segment n observed_median expected_median difference ci_lower ci_upper
|
|
181
|
+
# Finance 176 48339.70 47799.82 539.88 202.15 1031.01
|
|
182
|
+
# HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36
|
|
183
|
+
# IT 250 47226.50 47126.91 99.59 -117.56 510.81
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
A segment's confidence interval crossing zero means no clear deviation
|
|
187
|
+
from the benchmark; HR and Finance above don't cross zero, IT does.
|
|
188
|
+
|
|
189
|
+
## Robustness Map
|
|
190
|
+
|
|
191
|
+
Classify features by how much a conclusion about their relationship
|
|
192
|
+
with the target depends on (a) fitting method choice and (b) specific
|
|
193
|
+
influential observations -- two genuinely different failure modes that
|
|
194
|
+
a single diagnostic can miss:
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
from robustkit import feature_robustness_report, plot_feature_robustness
|
|
198
|
+
|
|
199
|
+
report = feature_robustness_report(df, target="value")
|
|
200
|
+
# feature stability_pct cook_impact_pct quadrant
|
|
201
|
+
# CRIM 8.9 17.1 fragile
|
|
202
|
+
# AGE 16.8 15.5 fragile
|
|
203
|
+
# RM 4.9 0.1 robust
|
|
204
|
+
# TAX 22.2 1.4 structural_sensitivity
|
|
205
|
+
|
|
206
|
+
plot_feature_robustness(report=report)
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
Four quadrants: **robust** (low spread, low impact), **structural
|
|
210
|
+
sensitivity** (sensitive to fitting method, not to specific points),
|
|
211
|
+
**data sensitive** (a few points drive the conclusion, method choice
|
|
212
|
+
barely matters), **fragile** (both -- least trustworthy).
|
|
213
|
+
|
|
214
|
+
`quadrant_report`/`plot_feature_space` (information) and
|
|
215
|
+
`feature_robustness_report`/`plot_feature_robustness` (benchmark) both
|
|
216
|
+
route through the same shared classifier, `robustkit.classify_quadrants`
|
|
217
|
+
-- any future quadrant-based analysis in this package will too.
|
|
218
|
+
|
|
219
|
+
## Analyst view vs. publisher view
|
|
220
|
+
|
|
221
|
+
Two visualizations that look superficially similar but answer
|
|
222
|
+
genuinely different questions:
|
|
223
|
+
|
|
224
|
+
```python
|
|
225
|
+
from robustkit import plot_analyst_view, plot_publisher_view, dispersion_ratio, iqr
|
|
226
|
+
|
|
227
|
+
# "How confident are we in the trend estimate?" -- a bootstrap
|
|
228
|
+
# confidence band that SHRINKS as sample size grows.
|
|
229
|
+
plot_analyst_view(df["age"], df["salary"])
|
|
230
|
+
|
|
231
|
+
# "How spread out are actual values in the population?" -- a median +
|
|
232
|
+
# IQR band that does NOT shrink with more data, since it reflects
|
|
233
|
+
# real dispersion, not estimation uncertainty. show_points defaults to
|
|
234
|
+
# False, since this view is meant for publishing potentially sensitive
|
|
235
|
+
# data (e.g. individual salaries) without exposing raw points.
|
|
236
|
+
plot_publisher_view(df["age"], df["salary"])
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
This distinction matters in practice: with 20x more data (same
|
|
240
|
+
underlying distribution), the analyst view's confidence band roughly
|
|
241
|
+
halves in width, while the publisher view's IQR band stays essentially
|
|
242
|
+
unchanged -- confirmed by the package's own test suite.
|
|
243
|
+
|
|
244
|
+
`dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
|
|
245
|
+
standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
246
|
+
computes both across bins of a continuous x, e.g. to check whether
|
|
247
|
+
dispersion (inequality) grows with age.
|
|
248
|
+
|
|
249
|
+
## Feature pairing (information)
|
|
250
|
+
|
|
251
|
+
Beyond ranking single features, evaluate *pairs* of features together:
|
|
252
|
+
how redundant are they with each other, and does knowing one reveal
|
|
253
|
+
additional predictive value in the other (synergy, e.g. an interaction
|
|
254
|
+
effect)?
|
|
255
|
+
|
|
256
|
+
```python
|
|
257
|
+
from robustkit import (
|
|
258
|
+
conditional_mutual_information, communication_score,
|
|
259
|
+
rank_by_communication, pair_redundancy, pair_synergy,
|
|
260
|
+
rank_communicative_pairs,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
# How communicable is a single feature -- not just predictive, but
|
|
264
|
+
# suitable for a clear chart/table (adequate group sizes, homogeneous
|
|
265
|
+
# groups, few enough categories to show at once)?
|
|
266
|
+
comm_ranking = rank_by_communication(df, target="value")
|
|
267
|
+
|
|
268
|
+
# How much does region's relevance to the target change once
|
|
269
|
+
# department is already known?
|
|
270
|
+
synergy = pair_synergy(df, feature_1="department", feature_2="region", target="value")
|
|
271
|
+
|
|
272
|
+
# Rank every candidate pair by combined relevance, penalizing
|
|
273
|
+
# redundant pairs and rewarding genuine synergy
|
|
274
|
+
pairs = rank_communicative_pairs(df, target="value")
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
All mutual-information-based quantities in this module (`rank_features`,
|
|
278
|
+
`conditional_mutual_information`, `pair_redundancy`, `pair_synergy`,
|
|
279
|
+
`communication_score`) are expressed in **bits**, consistent with
|
|
280
|
+
`entropy()` -- internally, scikit-learn's MI estimators return nats
|
|
281
|
+
and are converted before being used anywhere in this package.
|
|
282
|
+
|
|
156
283
|
## Design principles
|
|
157
284
|
|
|
158
285
|
- **One continuous x, one continuous y** at the core. This keeps every
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
# robustkit
|
|
2
|
+
|
|
3
|
+
> ⚠️ **Under active development.** This is an early placeholder release
|
|
4
|
+
> to claim the package name on PyPI. The API is incomplete and may
|
|
5
|
+
> change without notice. Not yet recommended for production use.
|
|
6
|
+
|
|
7
|
+
Practical tools for robust analysis of a single continuous relationship:
|
|
8
|
+
y as a function of one continuous x.
|
|
9
|
+
|
|
10
|
+
The guiding idea: **a conclusion that survives multiple fitting methods
|
|
11
|
+
is more trustworthy than one that only holds under a single model.**
|
|
12
|
+
`robustkit` makes it easy to compare Huber, Tukey biweight, and OLS
|
|
13
|
+
fits side by side, identify and quantify the influence of individual
|
|
14
|
+
observations, and get honest, bias-corrected uncertainty estimates.
|
|
15
|
+
|
|
16
|
+
## Status
|
|
17
|
+
|
|
18
|
+
`robustkit.core` (trend fitting, stability, diagnostics, uncertainty,
|
|
19
|
+
consistency checks), `robustkit.segmentation` (hierarchical grouping,
|
|
20
|
+
per-segment analysis), `robustkit.information` (mutual-information
|
|
21
|
+
feature ranking, quadrant classification, pairwise redundancy/synergy
|
|
22
|
+
scoring), `robustkit.benchmark` (global-trend segment comparison,
|
|
23
|
+
Robustness Map), and `robustkit.report` (analyst vs. publisher views,
|
|
24
|
+
dispersion measures) are stable and tested.
|
|
25
|
+
|
|
26
|
+
**Note on `information_efficiency`:** values can exceed 1.0 for
|
|
27
|
+
continuous features. `mutual_information` is estimated on the
|
|
28
|
+
full-resolution continuous values, while `entropy_bits` is computed on
|
|
29
|
+
a binned version of the same feature (since `entropy()` expects
|
|
30
|
+
categorical input). Binning discards information, so `entropy_bits` is
|
|
31
|
+
a lower bound on the feature's true entropy -- an efficiency above 1.0
|
|
32
|
+
signals that the feature carries more usable information than a coarse
|
|
33
|
+
categorical summary of it would capture. This is expected behavior,
|
|
34
|
+
not a bug.
|
|
35
|
+
|
|
36
|
+
## Installation
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
git clone https://github.com/<your-username>/robustkit.git
|
|
40
|
+
cd robustkit
|
|
41
|
+
pip install -e ".[dev]"
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Quickstart
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
import numpy as np
|
|
48
|
+
from robustkit import (
|
|
49
|
+
fit_huber_trend, fit_tukey_trend, predict_trend,
|
|
50
|
+
model_stability_pct, cooks_diagnostic, cook_impact,
|
|
51
|
+
bootstrap_band, bca_bootstrap_ci,
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
# x: a single continuous predictor, y: a single continuous outcome
|
|
55
|
+
x = np.random.default_rng(0).uniform(20, 60, 200)
|
|
56
|
+
y = 1000 + 50 * x - 0.4 * x**2 + np.random.default_rng(1).normal(0, 500, 200)
|
|
57
|
+
|
|
58
|
+
fit = fit_huber_trend(x, y, degree=2)
|
|
59
|
+
y_pred = predict_trend(fit, x_new=[30, 40, 50])
|
|
60
|
+
|
|
61
|
+
stability = model_stability_pct(x, y)
|
|
62
|
+
print("Median % spread between Huber/Tukey/OLS:", stability["median_pct_diff"])
|
|
63
|
+
|
|
64
|
+
diag = cooks_diagnostic(x, y)
|
|
65
|
+
impact = cook_impact(x, y, diag["flagged_indices"])
|
|
66
|
+
print("Median % change in curve if flagged points removed:", impact["median_pct_change"])
|
|
67
|
+
|
|
68
|
+
band = bootstrap_band(x, y)
|
|
69
|
+
ci = bca_bootstrap_ci(x, y, statistic_fn=lambda x_, y_: np.median(y_))
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
See `examples/quickstart_tutorial.py` for a complete, runnable walkthrough.
|
|
73
|
+
|
|
74
|
+
## Segmentation
|
|
75
|
+
|
|
76
|
+
Run any `robustkit.core` analysis independently across subgroups of a
|
|
77
|
+
larger dataset, with automatic fallback to coarser groupings when a
|
|
78
|
+
finer one is too small to analyze reliably:
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from robustkit import hierarchical_segment, apply_by_segment, model_stability_pct
|
|
82
|
+
|
|
83
|
+
hierarchy = [["department", "level", "status"], ["level", "status"], ["status"]]
|
|
84
|
+
segmented = hierarchical_segment(df, hierarchy, min_size=20)
|
|
85
|
+
|
|
86
|
+
report = apply_by_segment(
|
|
87
|
+
segmented, segment_col="segment_id", x_col="age", y_col="value",
|
|
88
|
+
analysis_fn=model_stability_pct,
|
|
89
|
+
)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
`apply_by_segment` works with any function shaped like
|
|
93
|
+
`analysis_fn(x, y, **kwargs) -> dict` -- built-in ones
|
|
94
|
+
(`model_stability_pct`, `cook_impact`, `bca_bootstrap_ci`, ...) or your
|
|
95
|
+
own. Only scalar values in the returned dict end up in the report
|
|
96
|
+
table; segments below `min_points` are skipped rather than causing an
|
|
97
|
+
error.
|
|
98
|
+
|
|
99
|
+
## Feature ranking (information)
|
|
100
|
+
|
|
101
|
+
Rank features by mutual information with a target, normalized by each
|
|
102
|
+
feature's own entropy, and classify them into four quadrants:
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
from robustkit import rank_features, quadrant_report, plot_feature_space
|
|
106
|
+
|
|
107
|
+
ranking = rank_features(df, target="value")
|
|
108
|
+
report = quadrant_report(df, target="value") # adds a `quadrant` column
|
|
109
|
+
plot_feature_space(df, target="value") # same quadrants, visualized
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
`quadrant_report` and `plot_feature_space` always agree on quadrant
|
|
113
|
+
assignment -- both route through the same thresholding logic.
|
|
114
|
+
|
|
115
|
+
**Caveat:** default thresholds are the *median* mutual information /
|
|
116
|
+
efficiency across the ranked features. With only a handful of
|
|
117
|
+
features, this can put a genuinely weak feature in the same "high"
|
|
118
|
+
half as a strong one, since roughly half of any list sits above its
|
|
119
|
+
own median regardless of how large the actual gap is. Median
|
|
120
|
+
thresholding becomes meaningful with a reasonably large feature set;
|
|
121
|
+
for a handful of candidates, read the raw `mutual_information` /
|
|
122
|
+
`information_efficiency` values directly rather than relying on the
|
|
123
|
+
quadrant label alone.
|
|
124
|
+
|
|
125
|
+
See `examples/information_tutorial.py` for a complete walkthrough.
|
|
126
|
+
|
|
127
|
+
## Benchmarking against a global trend
|
|
128
|
+
|
|
129
|
+
Compare each segment's observed outcome against what a single global
|
|
130
|
+
robust trend predicts, with bootstrap uncertainty on the difference --
|
|
131
|
+
answers "which groups deviate from the overall trend, and by how
|
|
132
|
+
much?" rather than "how does the trend look overall?":
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
from robustkit import segment_position_report
|
|
136
|
+
|
|
137
|
+
report = segment_position_report(
|
|
138
|
+
df, segment_col="department", x_col="age", y_col="salary",
|
|
139
|
+
)
|
|
140
|
+
# segment n observed_median expected_median difference ci_lower ci_upper
|
|
141
|
+
# Finance 176 48339.70 47799.82 539.88 202.15 1031.01
|
|
142
|
+
# HR 174 45718.84 46647.39 -928.55 -1293.26 -580.36
|
|
143
|
+
# IT 250 47226.50 47126.91 99.59 -117.56 510.81
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
A segment's confidence interval crossing zero means no clear deviation
|
|
147
|
+
from the benchmark; HR and Finance above don't cross zero, IT does.
|
|
148
|
+
|
|
149
|
+
## Robustness Map
|
|
150
|
+
|
|
151
|
+
Classify features by how much a conclusion about their relationship
|
|
152
|
+
with the target depends on (a) fitting method choice and (b) specific
|
|
153
|
+
influential observations -- two genuinely different failure modes that
|
|
154
|
+
a single diagnostic can miss:
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
from robustkit import feature_robustness_report, plot_feature_robustness
|
|
158
|
+
|
|
159
|
+
report = feature_robustness_report(df, target="value")
|
|
160
|
+
# feature stability_pct cook_impact_pct quadrant
|
|
161
|
+
# CRIM 8.9 17.1 fragile
|
|
162
|
+
# AGE 16.8 15.5 fragile
|
|
163
|
+
# RM 4.9 0.1 robust
|
|
164
|
+
# TAX 22.2 1.4 structural_sensitivity
|
|
165
|
+
|
|
166
|
+
plot_feature_robustness(report=report)
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Four quadrants: **robust** (low spread, low impact), **structural
|
|
170
|
+
sensitivity** (sensitive to fitting method, not to specific points),
|
|
171
|
+
**data sensitive** (a few points drive the conclusion, method choice
|
|
172
|
+
barely matters), **fragile** (both -- least trustworthy).
|
|
173
|
+
|
|
174
|
+
`quadrant_report`/`plot_feature_space` (information) and
|
|
175
|
+
`feature_robustness_report`/`plot_feature_robustness` (benchmark) both
|
|
176
|
+
route through the same shared classifier, `robustkit.classify_quadrants`
|
|
177
|
+
-- any future quadrant-based analysis in this package will too.
|
|
178
|
+
|
|
179
|
+
## Analyst view vs. publisher view
|
|
180
|
+
|
|
181
|
+
Two visualizations that look superficially similar but answer
|
|
182
|
+
genuinely different questions:
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
from robustkit import plot_analyst_view, plot_publisher_view, dispersion_ratio, iqr
|
|
186
|
+
|
|
187
|
+
# "How confident are we in the trend estimate?" -- a bootstrap
|
|
188
|
+
# confidence band that SHRINKS as sample size grows.
|
|
189
|
+
plot_analyst_view(df["age"], df["salary"])
|
|
190
|
+
|
|
191
|
+
# "How spread out are actual values in the population?" -- a median +
|
|
192
|
+
# IQR band that does NOT shrink with more data, since it reflects
|
|
193
|
+
# real dispersion, not estimation uncertainty. show_points defaults to
|
|
194
|
+
# False, since this view is meant for publishing potentially sensitive
|
|
195
|
+
# data (e.g. individual salaries) without exposing raw points.
|
|
196
|
+
plot_publisher_view(df["age"], df["salary"])
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
This distinction matters in practice: with 20x more data (same
|
|
200
|
+
underlying distribution), the analyst view's confidence band roughly
|
|
201
|
+
halves in width, while the publisher view's IQR band stays essentially
|
|
202
|
+
unchanged -- confirmed by the package's own test suite.
|
|
203
|
+
|
|
204
|
+
`dispersion_ratio(y)` -- (Q3-Q1)/median -- and `iqr(y)` are available
|
|
205
|
+
standalone for tabular reporting; `dispersion_by_bin(x, y, n_bins=10)`
|
|
206
|
+
computes both across bins of a continuous x, e.g. to check whether
|
|
207
|
+
dispersion (inequality) grows with age.
|
|
208
|
+
|
|
209
|
+
## Feature pairing (information)
|
|
210
|
+
|
|
211
|
+
Beyond ranking single features, evaluate *pairs* of features together:
|
|
212
|
+
how redundant are they with each other, and does knowing one reveal
|
|
213
|
+
additional predictive value in the other (synergy, e.g. an interaction
|
|
214
|
+
effect)?
|
|
215
|
+
|
|
216
|
+
```python
|
|
217
|
+
from robustkit import (
|
|
218
|
+
conditional_mutual_information, communication_score,
|
|
219
|
+
rank_by_communication, pair_redundancy, pair_synergy,
|
|
220
|
+
rank_communicative_pairs,
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
# How communicable is a single feature -- not just predictive, but
|
|
224
|
+
# suitable for a clear chart/table (adequate group sizes, homogeneous
|
|
225
|
+
# groups, few enough categories to show at once)?
|
|
226
|
+
comm_ranking = rank_by_communication(df, target="value")
|
|
227
|
+
|
|
228
|
+
# How much does region's relevance to the target change once
|
|
229
|
+
# department is already known?
|
|
230
|
+
synergy = pair_synergy(df, feature_1="department", feature_2="region", target="value")
|
|
231
|
+
|
|
232
|
+
# Rank every candidate pair by combined relevance, penalizing
|
|
233
|
+
# redundant pairs and rewarding genuine synergy
|
|
234
|
+
pairs = rank_communicative_pairs(df, target="value")
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
All mutual-information-based quantities in this module (`rank_features`,
|
|
238
|
+
`conditional_mutual_information`, `pair_redundancy`, `pair_synergy`,
|
|
239
|
+
`communication_score`) are expressed in **bits**, consistent with
|
|
240
|
+
`entropy()` -- internally, scikit-learn's MI estimators return nats
|
|
241
|
+
and are converted before being used anywhere in this package.
|
|
242
|
+
|
|
243
|
+
## Design principles
|
|
244
|
+
|
|
245
|
+
- **One continuous x, one continuous y** at the core. This keeps every
|
|
246
|
+
function's output visually and numerically interpretable (a curve
|
|
247
|
+
you can plot, a band you can read).
|
|
248
|
+
- **Diagnosis and action are separate steps.** `cooks_diagnostic`
|
|
249
|
+
flags candidates; `cook_impact` tells you whether removing them
|
|
250
|
+
actually changes anything.
|
|
251
|
+
- **OLS is a reference point, not the enemy.** Comparing robust fits
|
|
252
|
+
against OLS is how you know whether robustness mattered at all.
|
|
253
|
+
|
|
254
|
+
## License
|
|
255
|
+
|
|
256
|
+
MIT -- see [LICENSE](LICENSE).
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "robustkit"
|
|
7
|
-
version = "0.0
|
|
7
|
+
version = "0.4.0"
|
|
8
8
|
description = "Practical tools for robust analysis of a single continuous relationship: trend fitting, stability checks, influence diagnostics, and bootstrap uncertainty."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { file = "LICENSE" }
|
|
@@ -25,6 +25,15 @@ from .information.mutual_info import rank_features, information_efficiency
|
|
|
25
25
|
from .information.quadrants import quadrant_report
|
|
26
26
|
from .information.visualization import plot_feature_space, feature_map
|
|
27
27
|
from .information.profile import profile, print_profile
|
|
28
|
+
from .information.conditional_mi import conditional_mutual_information
|
|
29
|
+
from .information.communication import communication_score, rank_by_communication
|
|
30
|
+
from .information.pairs import pair_redundancy, pair_synergy, rank_communicative_pairs
|
|
31
|
+
from .common.quadrants import classify_quadrants
|
|
32
|
+
from .benchmark.global_model import fit_huber_benchmark, segment_position_report
|
|
33
|
+
from .benchmark.robustness_map import feature_robustness_report, plot_feature_robustness
|
|
34
|
+
from .report.dispersion import iqr, dispersion_ratio, dispersion_by_bin
|
|
35
|
+
from .report.visualize_analyst import plot_analyst_view
|
|
36
|
+
from .report.visualize_publisher import plot_publisher_view
|
|
28
37
|
|
|
29
38
|
__all__ = [
|
|
30
39
|
"fit_huber_trend",
|
|
@@ -49,6 +58,22 @@ __all__ = [
|
|
|
49
58
|
"feature_map",
|
|
50
59
|
"profile",
|
|
51
60
|
"print_profile",
|
|
61
|
+
"conditional_mutual_information",
|
|
62
|
+
"communication_score",
|
|
63
|
+
"rank_by_communication",
|
|
64
|
+
"pair_redundancy",
|
|
65
|
+
"pair_synergy",
|
|
66
|
+
"rank_communicative_pairs",
|
|
67
|
+
"classify_quadrants",
|
|
68
|
+
"fit_huber_benchmark",
|
|
69
|
+
"segment_position_report",
|
|
70
|
+
"feature_robustness_report",
|
|
71
|
+
"plot_feature_robustness",
|
|
72
|
+
"iqr",
|
|
73
|
+
"dispersion_ratio",
|
|
74
|
+
"dispersion_by_bin",
|
|
75
|
+
"plot_analyst_view",
|
|
76
|
+
"plot_publisher_view",
|
|
52
77
|
]
|
|
53
78
|
|
|
54
79
|
__version__ = "0.0.1"
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Compare each segment's observed outcome against what a single global
|
|
3
|
+
robust trend would predict, with bootstrap uncertainty on the
|
|
4
|
+
difference.
|
|
5
|
+
|
|
6
|
+
This answers a different question than robustkit.core.stability
|
|
7
|
+
(which asks "how does the trend look overall, across fitting
|
|
8
|
+
methods?"): here the question is "which groups deviate from the
|
|
9
|
+
overall robust trend, once we account for their x values, and how
|
|
10
|
+
confident are we in that deviation?"
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
from ..core.trend import fit_huber_trend, predict_trend
|
|
17
|
+
from ..core.uncertainty import bca_bootstrap_ci
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def fit_huber_benchmark(x, y, degree=2):
|
|
21
|
+
"""
|
|
22
|
+
Fit a single global Huber trend intended to serve as the
|
|
23
|
+
reference/benchmark that segments will be compared against. This
|
|
24
|
+
is just fit_huber_trend under a clearer name for this use case --
|
|
25
|
+
fit it once on the FULL population, not on any one segment.
|
|
26
|
+
"""
|
|
27
|
+
return fit_huber_trend(x, y, degree=degree)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def segment_position_report(df, segment_col, x_col, y_col, benchmark_fit=None,
|
|
31
|
+
degree=2, n_boot=500, ci=95, seed=0):
|
|
32
|
+
"""
|
|
33
|
+
For each segment, compare its observed median outcome against what
|
|
34
|
+
the global benchmark trend predicts at each row's x, with a BCa
|
|
35
|
+
bootstrap confidence interval on the (observed - expected)
|
|
36
|
+
difference.
|
|
37
|
+
|
|
38
|
+
benchmark_fit: a fit dict from fit_huber_benchmark, ideally fitted
|
|
39
|
+
on the full dataset (df should then be the same full dataset
|
|
40
|
+
this was fitted on, not a pre-filtered subset -- otherwise the
|
|
41
|
+
benchmark isn't a genuine "overall" reference for the segments
|
|
42
|
+
being compared). If omitted, one is fitted here on all of df.
|
|
43
|
+
|
|
44
|
+
Returns one row per segment: n, observed_median, expected_median,
|
|
45
|
+
difference (bootstrap point estimate), and the CI bounds.
|
|
46
|
+
"""
|
|
47
|
+
if benchmark_fit is None:
|
|
48
|
+
benchmark_fit = fit_huber_benchmark(
|
|
49
|
+
df[x_col].to_numpy(dtype=float), df[y_col].to_numpy(dtype=float), degree=degree,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
rows = []
|
|
53
|
+
for segment_value, group in df.groupby(segment_col, observed=True):
|
|
54
|
+
x = group[x_col].to_numpy(dtype=float)
|
|
55
|
+
y = group[y_col].to_numpy(dtype=float)
|
|
56
|
+
expected = predict_trend(benchmark_fit, x)
|
|
57
|
+
|
|
58
|
+
def difference_stat(x_, y_, _fit=benchmark_fit):
|
|
59
|
+
exp = predict_trend(_fit, x_)
|
|
60
|
+
return float(np.median(y_ - exp))
|
|
61
|
+
|
|
62
|
+
ci_result = bca_bootstrap_ci(x, y, difference_stat, n_boot=n_boot, ci=ci, seed=seed)
|
|
63
|
+
|
|
64
|
+
rows.append({
|
|
65
|
+
"segment": segment_value,
|
|
66
|
+
"n": len(group),
|
|
67
|
+
"observed_median": float(np.median(y)),
|
|
68
|
+
"expected_median": float(np.median(expected)),
|
|
69
|
+
"difference": ci_result["estimate"],
|
|
70
|
+
"ci_lower": ci_result["lower"],
|
|
71
|
+
"ci_upper": ci_result["upper"],
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
return pd.DataFrame(rows).sort_values("segment").reset_index(drop=True)
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The "Robustness Map": classify features by how much a conclusion about
|
|
3
|
+
their relationship with the target depends on (a) which robust fitting
|
|
4
|
+
method is used, and (b) which specific observations are included.
|
|
5
|
+
|
|
6
|
+
Motivated by a concrete finding: two features can have a similar
|
|
7
|
+
number of Cook's-distance-flagged observations while having wildly
|
|
8
|
+
different actual influence on the fitted trend (cook_impact). Model
|
|
9
|
+
stability and Cook-impact are answering genuinely different questions,
|
|
10
|
+
and a feature's position on both axes together says more about how
|
|
11
|
+
much to trust a conclusion involving it than either axis alone.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
from ..core.stability import model_stability_pct
|
|
17
|
+
from ..core.diagnostics import cooks_diagnostic, cook_impact
|
|
18
|
+
from ..common.quadrants import classify_quadrants
|
|
19
|
+
|
|
20
|
+
ROBUSTNESS_LABELS = {
|
|
21
|
+
"high_high": "fragile", # unstable AND data-driven -- least trustworthy
|
|
22
|
+
"high_y_only": "data_sensitive", # stable across methods, but driven by a few points
|
|
23
|
+
"high_x_only": "structural_sensitivity", # sensitive to method choice, not to specific points
|
|
24
|
+
"low_low": "robust", # stable across methods AND not driven by outliers
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
_ROBUSTNESS_DISPLAY = {
|
|
28
|
+
"robust": "Robust",
|
|
29
|
+
"structural_sensitivity": "Structural Sensitivity",
|
|
30
|
+
"data_sensitive": "Data Sensitive",
|
|
31
|
+
"fragile": "Fragile",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def feature_robustness_report(df, target, features=None, degree=2,
|
|
36
|
+
stability_threshold="median", impact_threshold="median"):
|
|
37
|
+
"""
|
|
38
|
+
For each candidate feature (used as x against target as y),
|
|
39
|
+
compute model_stability_pct and cook_impact, and classify the
|
|
40
|
+
feature into one of four robustness quadrants:
|
|
41
|
+
|
|
42
|
+
robust -- low stability spread, low Cook impact
|
|
43
|
+
structural_sensitivity -- high stability spread, low Cook impact
|
|
44
|
+
(the conclusion depends on which
|
|
45
|
+
fitting method you pick, not on
|
|
46
|
+
specific data points)
|
|
47
|
+
data_sensitive -- low stability spread, high Cook impact
|
|
48
|
+
(a handful of observations drive
|
|
49
|
+
the conclusion, but method choice
|
|
50
|
+
barely matters)
|
|
51
|
+
fragile -- high stability spread AND high Cook
|
|
52
|
+
impact (the least trustworthy)
|
|
53
|
+
"""
|
|
54
|
+
features = features or [c for c in df.columns if c != target]
|
|
55
|
+
y = df[target].to_numpy(dtype=float)
|
|
56
|
+
|
|
57
|
+
rows = []
|
|
58
|
+
for feature in features:
|
|
59
|
+
x = df[feature].to_numpy(dtype=float)
|
|
60
|
+
|
|
61
|
+
stability = model_stability_pct(x, y, degree=degree)
|
|
62
|
+
diag = cooks_diagnostic(x, y, degree=degree)
|
|
63
|
+
|
|
64
|
+
if len(diag["flagged_indices"]) > 0:
|
|
65
|
+
impact = cook_impact(x, y, diag["flagged_indices"], degree=degree)
|
|
66
|
+
cook_impact_pct = impact["median_pct_change"]
|
|
67
|
+
else:
|
|
68
|
+
cook_impact_pct = 0.0
|
|
69
|
+
|
|
70
|
+
rows.append({
|
|
71
|
+
"feature": feature,
|
|
72
|
+
"stability_pct": stability["median_pct_diff"],
|
|
73
|
+
"cook_impact_pct": cook_impact_pct,
|
|
74
|
+
"n_flagged": len(diag["flagged_indices"]),
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
report = pd.DataFrame(rows)
|
|
78
|
+
classified = classify_quadrants(
|
|
79
|
+
report, x_col="stability_pct", y_col="cook_impact_pct",
|
|
80
|
+
x_threshold=stability_threshold, y_threshold=impact_threshold,
|
|
81
|
+
labels=ROBUSTNESS_LABELS,
|
|
82
|
+
)
|
|
83
|
+
return classified.sort_values("cook_impact_pct", ascending=False).reset_index(drop=True)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def plot_feature_robustness(df=None, target=None, report=None, annotate=True, figsize=(10, 7), ax=None):
|
|
87
|
+
"""
|
|
88
|
+
Scatter plot of features in robustness space: x = model stability
|
|
89
|
+
spread (%), y = Cook-impact (%), colored by robustness quadrant.
|
|
90
|
+
|
|
91
|
+
Provide either a precomputed `report` (ideally the output of
|
|
92
|
+
feature_robustness_report, so quadrant labels are already
|
|
93
|
+
attached) or a `df`/`target` pair to compute everything internally.
|
|
94
|
+
Always routes quadrant assignment through the same logic as
|
|
95
|
+
feature_robustness_report, so the plot and the table can never
|
|
96
|
+
disagree.
|
|
97
|
+
"""
|
|
98
|
+
import matplotlib.pyplot as plt
|
|
99
|
+
from matplotlib.lines import Line2D
|
|
100
|
+
|
|
101
|
+
if report is None or "quadrant" not in report.columns:
|
|
102
|
+
report = feature_robustness_report(df, target=target)
|
|
103
|
+
|
|
104
|
+
colors_map = {
|
|
105
|
+
"robust": "green",
|
|
106
|
+
"structural_sensitivity": "steelblue",
|
|
107
|
+
"data_sensitive": "darkorange",
|
|
108
|
+
"fragile": "red",
|
|
109
|
+
}
|
|
110
|
+
colors = report["quadrant"].map(colors_map)
|
|
111
|
+
|
|
112
|
+
created_fig = ax is None
|
|
113
|
+
if created_fig:
|
|
114
|
+
fig, ax = plt.subplots(figsize=figsize)
|
|
115
|
+
|
|
116
|
+
ax.scatter(
|
|
117
|
+
report["stability_pct"], report["cook_impact_pct"],
|
|
118
|
+
c=colors, s=80, edgecolors="black", linewidths=0.5, alpha=0.8,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
if annotate:
|
|
122
|
+
for _, row in report.iterrows():
|
|
123
|
+
ax.annotate(
|
|
124
|
+
row["feature"], (row["stability_pct"], row["cook_impact_pct"]),
|
|
125
|
+
fontsize=8, xytext=(4, 4), textcoords="offset points",
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
legend_elements = [
|
|
129
|
+
Line2D([0], [0], marker="o", color="w", label=_ROBUSTNESS_DISPLAY[quadrant],
|
|
130
|
+
markerfacecolor=color, markersize=10)
|
|
131
|
+
for quadrant, color in colors_map.items()
|
|
132
|
+
]
|
|
133
|
+
ax.legend(handles=legend_elements, loc="best")
|
|
134
|
+
ax.set_xlabel("Model Stability Spread (%)")
|
|
135
|
+
ax.set_ylabel("Cook Impact (%)")
|
|
136
|
+
ax.set_title("Feature Robustness Map")
|
|
137
|
+
ax.grid(True, alpha=0.3)
|
|
138
|
+
|
|
139
|
+
if created_fig:
|
|
140
|
+
fig.tight_layout()
|
|
141
|
+
|
|
142
|
+
return report
|