dfshrink 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dfshrink-0.1.0/LICENSE +21 -0
- dfshrink-0.1.0/PKG-INFO +279 -0
- dfshrink-0.1.0/README.md +244 -0
- dfshrink-0.1.0/pyproject.toml +79 -0
- dfshrink-0.1.0/pyproject.toml.orig +82 -0
- dfshrink-0.1.0/src/dfshrink/__init__.py +31 -0
- dfshrink-0.1.0/src/dfshrink/_render.py +198 -0
- dfshrink-0.1.0/src/dfshrink/columns.py +240 -0
- dfshrink-0.1.0/src/dfshrink/ext/__init__.py +20 -0
- dfshrink-0.1.0/src/dfshrink/ext/_adapter.py +168 -0
- dfshrink-0.1.0/src/dfshrink/ext/dataframely.py +114 -0
- dfshrink-0.1.0/src/dfshrink/ext/pandera.py +133 -0
- dfshrink-0.1.0/src/dfshrink/ext/patito.py +113 -0
- dfshrink-0.1.0/src/dfshrink/ext/pytest.py +51 -0
- dfshrink-0.1.0/src/dfshrink/failure.py +95 -0
- dfshrink-0.1.0/src/dfshrink/py.typed +0 -0
- dfshrink-0.1.0/src/dfshrink/shrink.py +333 -0
- dfshrink-0.1.0/src/dfshrink/values.py +385 -0
dfshrink-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Tomas Perez Alvarez
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
dfshrink-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dfshrink
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Turn a failing Polars DataFrame into a minimal reproducible example
|
|
5
|
+
Keywords: polars,dataframe,data-contract,delta-debugging,ddmin,minimal-repro,testing,validation
|
|
6
|
+
Author: Tomas Perez Alvarez
|
|
7
|
+
Author-email: Tomas Perez Alvarez <tomasperezalvarez@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
17
|
+
Classifier: Topic :: Software Development :: Testing
|
|
18
|
+
Requires-Dist: polars>=1.44.2
|
|
19
|
+
Requires-Dist: dataframely>=3 ; extra == 'dataframely'
|
|
20
|
+
Requires-Dist: pandera>=0.22 ; extra == 'pandera'
|
|
21
|
+
Requires-Dist: patito>=0.8.6 ; extra == 'patito'
|
|
22
|
+
Requires-Dist: dataframely>=3 ; extra == 'validators'
|
|
23
|
+
Requires-Dist: pandera>=0.22 ; extra == 'validators'
|
|
24
|
+
Requires-Dist: patito>=0.8.6 ; extra == 'validators'
|
|
25
|
+
Requires-Python: >=3.12
|
|
26
|
+
Project-URL: Homepage, https://github.com/Tomperez98/dfshrink
|
|
27
|
+
Project-URL: Repository, https://github.com/Tomperez98/dfshrink
|
|
28
|
+
Project-URL: Issues, https://github.com/Tomperez98/dfshrink/issues
|
|
29
|
+
Project-URL: Changelog, https://github.com/Tomperez98/dfshrink/blob/main/CHANGELOG.md
|
|
30
|
+
Provides-Extra: dataframely
|
|
31
|
+
Provides-Extra: pandera
|
|
32
|
+
Provides-Extra: patito
|
|
33
|
+
Provides-Extra: validators
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
|
|
36
|
+
# dfshrink
|
|
37
|
+
|
|
38
|
+
**Your data contract failed. Get the smallest frame that still breaks it — and a regression test.**
|
|
39
|
+
|
|
40
|
+
A validator tells you the frame is bad; it can't tell you *which* rows. `shrink_rows`
|
|
41
|
+
runs that same predicate over smaller and smaller subsets and returns the minimal
|
|
42
|
+
frame that still fails:
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
import polars as pl
|
|
46
|
+
from dfshrink import shrink_rows
|
|
47
|
+
|
|
48
|
+
df = pl.DataFrame(
|
|
49
|
+
{
|
|
50
|
+
"order_id": [1, 1, 1, 2, 2, 3],
|
|
51
|
+
"amount": [10, 20, 30, 5, 7, 9],
|
|
52
|
+
"total": [50, 50, 50, 12, 12, 9],
|
|
53
|
+
}
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def bug(df: pl.DataFrame) -> bool:
|
|
58
|
+
"""Each order's line items must sum to its declared total."""
|
|
59
|
+
bad = (
|
|
60
|
+
df.group_by("order_id")
|
|
61
|
+
.agg(pl.col("amount").sum().alias("lines"), pl.col("total").first())
|
|
62
|
+
.filter(pl.col("lines") != pl.col("total"))
|
|
63
|
+
)
|
|
64
|
+
return bad.height > 0
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
repro = shrink_rows(df, bug)
|
|
68
|
+
print(repro)
|
|
69
|
+
print(repro.frame)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)
|
|
74
|
+
shape: (1, 3)
|
|
75
|
+
┌──────────┬────────┬───────┐
|
|
76
|
+
│ order_id ┆ amount ┆ total │
|
|
77
|
+
│ --- ┆ --- ┆ --- │
|
|
78
|
+
│ i64 ┆ i64 ┆ i64 │
|
|
79
|
+
╞══════════╪════════╪═══════╡
|
|
80
|
+
│ 1 ┆ 10 ┆ 50 │
|
|
81
|
+
└──────────┴────────┴───────┘
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Six rows in, one row out — in four calls to `bug`. The rule is an aggregate, so
|
|
85
|
+
there is no per-row mask to `.filter()` on; "does this subset still fail?" is the
|
|
86
|
+
only question, and shrinking answers it. Paste the row into a regression test with
|
|
87
|
+
`repro.to_code()`, or drop `repro.to_markdown()` into the incident ticket.
|
|
88
|
+
|
|
89
|
+
That's the data engineer's loop: **a contract check fails → the smallest still-bad
|
|
90
|
+
frame and the broken rule → a committed test and a fix** — without hand-slicing a
|
|
91
|
+
warehouse extract.
|
|
92
|
+
|
|
93
|
+
## What it is
|
|
94
|
+
|
|
95
|
+
dfshrink is the triage step between "the contract check failed" and "here is the
|
|
96
|
+
fix." Point it at the validator you already run — dataframely, pandera, patito, or
|
|
97
|
+
a plain `DataFrame -> bool` — and it returns the smallest frame that still fails,
|
|
98
|
+
plus the rule and column that broke.
|
|
99
|
+
|
|
100
|
+
`shrink_rows(frame, fails, *, max_evals=10_000) -> Repro | None` runs delta
|
|
101
|
+
debugging (`ddmin`) over a Polars frame's rows and returns the smallest subset on
|
|
102
|
+
which `fails` is still `True` — a minimal repro for a data bug, the Python/Polars
|
|
103
|
+
analog of R's `minex::reduce_rows`.
|
|
104
|
+
|
|
105
|
+
## Install
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
pip install dfshrink # or: uv add dfshrink
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Requires Python 3.12+ and Polars. The base package imports only Polars; each
|
|
112
|
+
validator adapter is an extra (`pip install "dfshrink[dataframely]"`,
|
|
113
|
+
`dfshrink[pandera]`, `dfshrink[patito]`).
|
|
114
|
+
|
|
115
|
+
Developing dfshrink itself? From a checkout, `mise install && mise run setup`.
|
|
116
|
+
|
|
117
|
+
**If you can write the failing rule as a per-row mask, use `.filter()` — it's
|
|
118
|
+
simpler.** Shrinking is for when you can't: a validator that returns a single bit
|
|
119
|
+
(`is_valid`, `validate`, patito), an aggregate, or a cross-row interaction. There
|
|
120
|
+
is no mask then; there is only "does this subset still fail?".
|
|
121
|
+
|
|
122
|
+
## Reading the result
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
repro = shrink_rows(df, bug)
|
|
126
|
+
|
|
127
|
+
if repro is None:
|
|
128
|
+
print("the frame does not fail") # expected failure -> None
|
|
129
|
+
else:
|
|
130
|
+
print(repro.frame) # the minimal failing rows
|
|
131
|
+
print(repro.removed_rows) # rows dropped from the input
|
|
132
|
+
print(repro.minimality_proven) # True => removing any one row makes it pass
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
`fails` is the seam — a lambda, a test assertion, or a schema adapter (next
|
|
136
|
+
section). Keep it pure: shrinking re-runs it many times, so it must be
|
|
137
|
+
deterministic.
|
|
138
|
+
|
|
139
|
+
### Paste it into a test or a ticket
|
|
140
|
+
|
|
141
|
+
A `Repro` renders itself for wherever the repro is going:
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
repro.to_code() # a pl.DataFrame({...}, schema={...}) constructor
|
|
145
|
+
repro.to_markdown() # a markdown table, dtypes in the headers
|
|
146
|
+
str(repro) # 'Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)'
|
|
147
|
+
repro.as_frame() # the replayed frame, to re-run your predicate on
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
`to_code()` round-trips: `eval(repro.to_code())`, with only `polars as pl` in
|
|
151
|
+
scope, rebuilds an equal frame — so the constructor goes straight into a test.
|
|
152
|
+
A dtype that cannot be rendered without loss (a column of `pl.Object`, say)
|
|
153
|
+
raises `TypeError` rather than emitting code that quietly builds a different
|
|
154
|
+
frame.
|
|
155
|
+
|
|
156
|
+
## Shrink against dataframely, pandera, or patito
|
|
157
|
+
|
|
158
|
+
A schema tells you *that* a rule broke — not which rows did it:
|
|
159
|
+
|
|
160
|
+
```text
|
|
161
|
+
dataframely.exc.ValidationError: 1 rules failed validation:
|
|
162
|
+
* Column 'amount' failed validation for 1 rules:
|
|
163
|
+
- 'min' failed for 1 rows
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
Hand the schema to `shrink_rows` instead, and it returns the row that did it.
|
|
167
|
+
You already wrote the validator, so there's nothing to re-express:
|
|
168
|
+
|
|
169
|
+
```python
|
|
170
|
+
import dataframely as dy
|
|
171
|
+
from dfshrink.ext.dataframely import shrink_rows
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
class HouseSchema(dy.Schema):
|
|
175
|
+
amount = dy.Int64(nullable=False, min=0)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
repro = shrink_rows(df, HouseSchema) # inverts HouseSchema.is_valid(df)
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Same shape for the others — `dfshrink.ext.pandera` wraps `validate(df)`-raises,
|
|
182
|
+
`dfshrink.ext.patito` wraps `Model.validate(df)`:
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
import pandera.polars as pa
|
|
186
|
+
from dfshrink.ext.pandera import shrink_rows
|
|
187
|
+
|
|
188
|
+
schema = pa.DataFrameSchema({"amount": pa.Column(int, pa.Check.gt(0))})
|
|
189
|
+
repro = shrink_rows(df, schema)
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
import patito as pt
|
|
194
|
+
from dfshrink.ext.patito import shrink_rows
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class House(pt.Model):
|
|
198
|
+
amount: int = pt.Field(ge=0)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
repro = shrink_rows(df, House)
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
Each module also exposes `as_predicate` for the raw `DataFrame -> bool`. The
|
|
205
|
+
frame must already match the schema's columns and dtypes — shrinking only removes
|
|
206
|
+
rows, so a structural mismatch is the caller's bug. Runnable versions:
|
|
207
|
+
[`examples/dataframely_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/dataframely_schema.py),
|
|
208
|
+
[`examples/pandera_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/pandera_schema.py),
|
|
209
|
+
[`examples/patito_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/patito_schema.py), and the
|
|
210
|
+
dependency-free [`examples/predicate.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/predicate.py).
|
|
211
|
+
|
|
212
|
+
## Fast: ~log₂(n) predicate calls
|
|
213
|
+
|
|
214
|
+
The only cost that scales with your data is your predicate. `ddmin` narrows by
|
|
215
|
+
chunks instead of removing one row at a time (which is `O(n²)` calls), so a bad
|
|
216
|
+
row is isolated in ~log₂(n) calls:
|
|
217
|
+
|
|
218
|
+
| Frame | Bad rows | Result | Predicate calls |
|
|
219
|
+
|---|---|---|---|
|
|
220
|
+
| 20,000 rows | 1 | 1 row | 16–29 |
|
|
221
|
+
| 6 rows | 1 | 1 row | 4 |
|
|
222
|
+
|
|
223
|
+
Call counts depend on where the bad rows sit; the range is measured. Candidates
|
|
224
|
+
are selected with `DataFrame.slice`, so copying the frame is not the cost: with a
|
|
225
|
+
cheap predicate, 20,000 rows shrink to one in ~0.25 ms. Cap the predicate calls
|
|
226
|
+
with `max_evals` (default 10,000) —
|
|
227
|
+
[performance in depth →](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md).
|
|
228
|
+
|
|
229
|
+
## Who it's for — and when not to use it
|
|
230
|
+
|
|
231
|
+
**It's for data engineers running Polars pipelines with a pass/fail validator** —
|
|
232
|
+
a data contract, a schema check, or a cross-row invariant (totals, uniqueness,
|
|
233
|
+
ordering) you assert in a test, in CI, or during an incident. The payoff is
|
|
234
|
+
turning a failed check into the one row and rule to fix, a regression test to
|
|
235
|
+
commit, and a ticket the upstream owner can act on.
|
|
236
|
+
|
|
237
|
+
It is deliberately narrow:
|
|
238
|
+
|
|
239
|
+
- **Polars only.** pandas and SQL/warehouse data aren't supported today. If the
|
|
240
|
+
bad data lives in a warehouse, materialize the frame first.
|
|
241
|
+
- **Not a validator.** It doesn't define or check rules; it runs *after* one
|
|
242
|
+
fails, using your predicate as the oracle.
|
|
243
|
+
- **Not production monitoring.** It's a developer/CI debugging tool, not a
|
|
244
|
+
data-observability service.
|
|
245
|
+
- **No pipeline attribution.** It won't tell you *which step* (a join, a cast)
|
|
246
|
+
introduced the bad rows — only which rows.
|
|
247
|
+
- **Columns are schema-aware.** It minimizes rows and numeric values; dropping
|
|
248
|
+
columns needs the adapter's explainer (`diagnose(..., columns=True)`), because a
|
|
249
|
+
black-box predicate cannot tell a real failure from a frame that went
|
|
250
|
+
structurally invalid.
|
|
251
|
+
- **Needs a pure, deterministic predicate.** Nondeterminism makes shrinking
|
|
252
|
+
meaningless.
|
|
253
|
+
|
|
254
|
+
## Learn more
|
|
255
|
+
|
|
256
|
+
The README is the front door; each guide owns one task in depth:
|
|
257
|
+
|
|
258
|
+
- **[Diagnose the failure](https://github.com/Tomperez98/dfshrink/blob/main/docs/diagnose.md)** — get the rule and column the validator flagged, not just the rows.
|
|
259
|
+
- **[Fail with a repro in CI](https://github.com/Tomperez98/dfshrink/blob/main/docs/ci.md)** — `assert_valid` turns a failed job into a minimal repro.
|
|
260
|
+
- **[Move the value to the boundary](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-values.md)** — shrink a bad cell to the last value that still fails.
|
|
261
|
+
- **[Drop the columns the rule does not need](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-columns.md)** — schema-aware column reduction.
|
|
262
|
+
- **[Performance in depth](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md)** — call counts, the worst case, and `max_evals`.
|
|
263
|
+
- **[The contract](https://github.com/Tomperez98/dfshrink/blob/main/docs/contract.md)** — panic/value semantics, minimality guarantees, and the algorithm.
|
|
264
|
+
|
|
265
|
+
## Development
|
|
266
|
+
|
|
267
|
+
[mise](https://mise.jdx.dev/) drives every task, so a laptop and CI run the same
|
|
268
|
+
commands:
|
|
269
|
+
|
|
270
|
+
```bash
|
|
271
|
+
mise install # pinned Python and uv
|
|
272
|
+
mise run setup # every dependency, including the validator extras
|
|
273
|
+
mise run ci # the merge gate: format, lint, type, hygiene, tests, examples, build, smoke
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
While iterating, run one tier: `mise run test`, `mise run lint`, `mise run
|
|
277
|
+
type`. Every diagnostic is an error. See [CONTRIBUTING.md](CONTRIBUTING.md) for
|
|
278
|
+
the pull-request checklist, and [RELEASING.md](RELEASING.md) for how a release
|
|
279
|
+
is cut.
|
dfshrink-0.1.0/README.md
ADDED
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
# dfshrink
|
|
2
|
+
|
|
3
|
+
**Your data contract failed. Get the smallest frame that still breaks it — and a regression test.**
|
|
4
|
+
|
|
5
|
+
A validator tells you the frame is bad; it can't tell you *which* rows. `shrink_rows`
|
|
6
|
+
runs that same predicate over smaller and smaller subsets and returns the minimal
|
|
7
|
+
frame that still fails:
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
import polars as pl
|
|
11
|
+
from dfshrink import shrink_rows
|
|
12
|
+
|
|
13
|
+
df = pl.DataFrame(
|
|
14
|
+
{
|
|
15
|
+
"order_id": [1, 1, 1, 2, 2, 3],
|
|
16
|
+
"amount": [10, 20, 30, 5, 7, 9],
|
|
17
|
+
"total": [50, 50, 50, 12, 12, 9],
|
|
18
|
+
}
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def bug(df: pl.DataFrame) -> bool:
|
|
23
|
+
"""Each order's line items must sum to its declared total."""
|
|
24
|
+
bad = (
|
|
25
|
+
df.group_by("order_id")
|
|
26
|
+
.agg(pl.col("amount").sum().alias("lines"), pl.col("total").first())
|
|
27
|
+
.filter(pl.col("lines") != pl.col("total"))
|
|
28
|
+
)
|
|
29
|
+
return bad.height > 0
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
repro = shrink_rows(df, bug)
|
|
33
|
+
print(repro)
|
|
34
|
+
print(repro.frame)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
```
|
|
38
|
+
Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)
|
|
39
|
+
shape: (1, 3)
|
|
40
|
+
┌──────────┬────────┬───────┐
|
|
41
|
+
│ order_id ┆ amount ┆ total │
|
|
42
|
+
│ --- ┆ --- ┆ --- │
|
|
43
|
+
│ i64 ┆ i64 ┆ i64 │
|
|
44
|
+
╞══════════╪════════╪═══════╡
|
|
45
|
+
│ 1 ┆ 10 ┆ 50 │
|
|
46
|
+
└──────────┴────────┴───────┘
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Six rows in, one row out — in four calls to `bug`. The rule is an aggregate, so
|
|
50
|
+
there is no per-row mask to `.filter()` on; "does this subset still fail?" is the
|
|
51
|
+
only question, and shrinking answers it. Paste the row into a regression test with
|
|
52
|
+
`repro.to_code()`, or drop `repro.to_markdown()` into the incident ticket.
|
|
53
|
+
|
|
54
|
+
That's the data engineer's loop: **a contract check fails → the smallest still-bad
|
|
55
|
+
frame and the broken rule → a committed test and a fix** — without hand-slicing a
|
|
56
|
+
warehouse extract.
|
|
57
|
+
|
|
58
|
+
## What it is
|
|
59
|
+
|
|
60
|
+
dfshrink is the triage step between "the contract check failed" and "here is the
|
|
61
|
+
fix." Point it at the validator you already run — dataframely, pandera, patito, or
|
|
62
|
+
a plain `DataFrame -> bool` — and it returns the smallest frame that still fails,
|
|
63
|
+
plus the rule and column that broke.
|
|
64
|
+
|
|
65
|
+
`shrink_rows(frame, fails, *, max_evals=10_000) -> Repro | None` runs delta
|
|
66
|
+
debugging (`ddmin`) over a Polars frame's rows and returns the smallest subset on
|
|
67
|
+
which `fails` is still `True` — a minimal repro for a data bug, the Python/Polars
|
|
68
|
+
analog of R's `minex::reduce_rows`.
|
|
69
|
+
|
|
70
|
+
## Install
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
pip install dfshrink # or: uv add dfshrink
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Requires Python 3.12+ and Polars. The base package imports only Polars; each
|
|
77
|
+
validator adapter is an extra (`pip install "dfshrink[dataframely]"`,
|
|
78
|
+
`dfshrink[pandera]`, `dfshrink[patito]`).
|
|
79
|
+
|
|
80
|
+
Developing dfshrink itself? From a checkout, `mise install && mise run setup`.
|
|
81
|
+
|
|
82
|
+
**If you can write the failing rule as a per-row mask, use `.filter()` — it's
|
|
83
|
+
simpler.** Shrinking is for when you can't: a validator that returns a single bit
|
|
84
|
+
(`is_valid`, `validate`, patito), an aggregate, or a cross-row interaction. There
|
|
85
|
+
is no mask then; there is only "does this subset still fail?".
|
|
86
|
+
|
|
87
|
+
## Reading the result
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
repro = shrink_rows(df, bug)
|
|
91
|
+
|
|
92
|
+
if repro is None:
|
|
93
|
+
print("the frame does not fail") # expected failure -> None
|
|
94
|
+
else:
|
|
95
|
+
print(repro.frame) # the minimal failing rows
|
|
96
|
+
print(repro.removed_rows) # rows dropped from the input
|
|
97
|
+
print(repro.minimality_proven) # True => removing any one row makes it pass
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
`fails` is the seam — a lambda, a test assertion, or a schema adapter (next
|
|
101
|
+
section). Keep it pure: shrinking re-runs it many times, so it must be
|
|
102
|
+
deterministic.
|
|
103
|
+
|
|
104
|
+
### Paste it into a test or a ticket
|
|
105
|
+
|
|
106
|
+
A `Repro` renders itself for wherever the repro is going:
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
repro.to_code() # a pl.DataFrame({...}, schema={...}) constructor
|
|
110
|
+
repro.to_markdown() # a markdown table, dtypes in the headers
|
|
111
|
+
str(repro) # 'Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)'
|
|
112
|
+
repro.as_frame() # the replayed frame, to re-run your predicate on
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
`to_code()` round-trips: `eval(repro.to_code())`, with only `polars as pl` in
|
|
116
|
+
scope, rebuilds an equal frame — so the constructor goes straight into a test.
|
|
117
|
+
A dtype that cannot be rendered without loss (a column of `pl.Object`, say)
|
|
118
|
+
raises `TypeError` rather than emitting code that quietly builds a different
|
|
119
|
+
frame.
|
|
120
|
+
|
|
121
|
+
## Shrink against dataframely, pandera, or patito
|
|
122
|
+
|
|
123
|
+
A schema tells you *that* a rule broke — not which rows did it:
|
|
124
|
+
|
|
125
|
+
```text
|
|
126
|
+
dataframely.exc.ValidationError: 1 rules failed validation:
|
|
127
|
+
* Column 'amount' failed validation for 1 rules:
|
|
128
|
+
- 'min' failed for 1 rows
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Hand the schema to `shrink_rows` instead, and it returns the row that did it.
|
|
132
|
+
You already wrote the validator, so there's nothing to re-express:
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
import dataframely as dy
|
|
136
|
+
from dfshrink.ext.dataframely import shrink_rows
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class HouseSchema(dy.Schema):
|
|
140
|
+
amount = dy.Int64(nullable=False, min=0)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
repro = shrink_rows(df, HouseSchema) # inverts HouseSchema.is_valid(df)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Same shape for the others — `dfshrink.ext.pandera` wraps `validate(df)`-raises,
|
|
147
|
+
`dfshrink.ext.patito` wraps `Model.validate(df)`:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
import pandera.polars as pa
|
|
151
|
+
from dfshrink.ext.pandera import shrink_rows
|
|
152
|
+
|
|
153
|
+
schema = pa.DataFrameSchema({"amount": pa.Column(int, pa.Check.gt(0))})
|
|
154
|
+
repro = shrink_rows(df, schema)
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
```python
|
|
158
|
+
import patito as pt
|
|
159
|
+
from dfshrink.ext.patito import shrink_rows
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class House(pt.Model):
|
|
163
|
+
amount: int = pt.Field(ge=0)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
repro = shrink_rows(df, House)
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Each module also exposes `as_predicate` for the raw `DataFrame -> bool`. The
|
|
170
|
+
frame must already match the schema's columns and dtypes — shrinking only removes
|
|
171
|
+
rows, so a structural mismatch is the caller's bug. Runnable versions:
|
|
172
|
+
[`examples/dataframely_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/dataframely_schema.py),
|
|
173
|
+
[`examples/pandera_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/pandera_schema.py),
|
|
174
|
+
[`examples/patito_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/patito_schema.py), and the
|
|
175
|
+
dependency-free [`examples/predicate.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/predicate.py).
|
|
176
|
+
|
|
177
|
+
## Fast: ~log₂(n) predicate calls
|
|
178
|
+
|
|
179
|
+
The only cost that scales with your data is your predicate. `ddmin` narrows by
|
|
180
|
+
chunks instead of removing one row at a time (which is `O(n²)` calls), so a bad
|
|
181
|
+
row is isolated in ~log₂(n) calls:
|
|
182
|
+
|
|
183
|
+
| Frame | Bad rows | Result | Predicate calls |
|
|
184
|
+
|---|---|---|---|
|
|
185
|
+
| 20,000 rows | 1 | 1 row | 16–29 |
|
|
186
|
+
| 6 rows | 1 | 1 row | 4 |
|
|
187
|
+
|
|
188
|
+
Call counts depend on where the bad rows sit; the range is measured. Candidates
|
|
189
|
+
are selected with `DataFrame.slice`, so copying the frame is not the cost: with a
|
|
190
|
+
cheap predicate, 20,000 rows shrink to one in ~0.25 ms. Cap the predicate calls
|
|
191
|
+
with `max_evals` (default 10,000) —
|
|
192
|
+
[performance in depth →](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md).
|
|
193
|
+
|
|
194
|
+
## Who it's for — and when not to use it
|
|
195
|
+
|
|
196
|
+
**It's for data engineers running Polars pipelines with a pass/fail validator** —
|
|
197
|
+
a data contract, a schema check, or a cross-row invariant (totals, uniqueness,
|
|
198
|
+
ordering) you assert in a test, in CI, or during an incident. The payoff is
|
|
199
|
+
turning a failed check into the one row and rule to fix, a regression test to
|
|
200
|
+
commit, and a ticket the upstream owner can act on.
|
|
201
|
+
|
|
202
|
+
It is deliberately narrow:
|
|
203
|
+
|
|
204
|
+
- **Polars only.** pandas and SQL/warehouse data aren't supported today. If the
|
|
205
|
+
bad data lives in a warehouse, materialize the frame first.
|
|
206
|
+
- **Not a validator.** It doesn't define or check rules; it runs *after* one
|
|
207
|
+
fails, using your predicate as the oracle.
|
|
208
|
+
- **Not production monitoring.** It's a developer/CI debugging tool, not a
|
|
209
|
+
data-observability service.
|
|
210
|
+
- **No pipeline attribution.** It won't tell you *which step* (a join, a cast)
|
|
211
|
+
introduced the bad rows — only which rows.
|
|
212
|
+
- **Columns are schema-aware.** It minimizes rows and numeric values; dropping
|
|
213
|
+
columns needs the adapter's explainer (`diagnose(..., columns=True)`), because a
|
|
214
|
+
black-box predicate cannot tell a real failure from a frame that went
|
|
215
|
+
structurally invalid.
|
|
216
|
+
- **Needs a pure, deterministic predicate.** Nondeterminism makes shrinking
|
|
217
|
+
meaningless.
|
|
218
|
+
|
|
219
|
+
## Learn more
|
|
220
|
+
|
|
221
|
+
The README is the front door; each guide owns one task in depth:
|
|
222
|
+
|
|
223
|
+
- **[Diagnose the failure](https://github.com/Tomperez98/dfshrink/blob/main/docs/diagnose.md)** — get the rule and column the validator flagged, not just the rows.
|
|
224
|
+
- **[Fail with a repro in CI](https://github.com/Tomperez98/dfshrink/blob/main/docs/ci.md)** — `assert_valid` turns a failed job into a minimal repro.
|
|
225
|
+
- **[Move the value to the boundary](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-values.md)** — shrink a bad cell to the last value that still fails.
|
|
226
|
+
- **[Drop the columns the rule does not need](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-columns.md)** — schema-aware column reduction.
|
|
227
|
+
- **[Performance in depth](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md)** — call counts, the worst case, and `max_evals`.
|
|
228
|
+
- **[The contract](https://github.com/Tomperez98/dfshrink/blob/main/docs/contract.md)** — panic/value semantics, minimality guarantees, and the algorithm.
|
|
229
|
+
|
|
230
|
+
## Development
|
|
231
|
+
|
|
232
|
+
[mise](https://mise.jdx.dev/) drives every task, so a laptop and CI run the same
|
|
233
|
+
commands:
|
|
234
|
+
|
|
235
|
+
```bash
|
|
236
|
+
mise install # pinned Python and uv
|
|
237
|
+
mise run setup # every dependency, including the validator extras
|
|
238
|
+
mise run ci # the merge gate: format, lint, type, hygiene, tests, examples, build, smoke
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
While iterating, run one tier: `mise run test`, `mise run lint`, `mise run
|
|
242
|
+
type`. Every diagnostic is an error. See [CONTRIBUTING.md](CONTRIBUTING.md) for
|
|
243
|
+
the pull-request checklist, and [RELEASING.md](RELEASING.md) for how a release
|
|
244
|
+
is cut.
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "dfshrink"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Turn a failing Polars DataFrame into a minimal reproducible example"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = ["LICENSE"]
|
|
8
|
+
requires-python = ">=3.12"
|
|
9
|
+
keywords = [
|
|
10
|
+
"polars",
|
|
11
|
+
"dataframe",
|
|
12
|
+
"data-contract",
|
|
13
|
+
"delta-debugging",
|
|
14
|
+
"ddmin",
|
|
15
|
+
"minimal-repro",
|
|
16
|
+
"testing",
|
|
17
|
+
"validation",
|
|
18
|
+
]
|
|
19
|
+
classifiers = [
|
|
20
|
+
"Development Status :: 3 - Alpha",
|
|
21
|
+
"Intended Audience :: Developers",
|
|
22
|
+
"Operating System :: OS Independent",
|
|
23
|
+
"Programming Language :: Python :: 3",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
27
|
+
"Topic :: Software Development :: Testing",
|
|
28
|
+
]
|
|
29
|
+
dependencies = ["polars>=1.44.2"]
|
|
30
|
+
|
|
31
|
+
[[project.authors]]
|
|
32
|
+
name = "Tomas Perez Alvarez"
|
|
33
|
+
email = "tomasperezalvarez@gmail.com"
|
|
34
|
+
|
|
35
|
+
[project.urls]
|
|
36
|
+
Homepage = "https://github.com/Tomperez98/dfshrink"
|
|
37
|
+
Repository = "https://github.com/Tomperez98/dfshrink"
|
|
38
|
+
Issues = "https://github.com/Tomperez98/dfshrink/issues"
|
|
39
|
+
Changelog = "https://github.com/Tomperez98/dfshrink/blob/main/CHANGELOG.md"
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
dataframely = ["dataframely>=3"]
|
|
43
|
+
pandera = ["pandera>=0.22"]
|
|
44
|
+
patito = ["patito>=0.8.6"]
|
|
45
|
+
validators = [
|
|
46
|
+
"dataframely>=3",
|
|
47
|
+
"pandera>=0.22",
|
|
48
|
+
"patito>=0.8.6",
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
[build-system]
|
|
52
|
+
requires = ["uv_build>=0.12.17,<0.13.0"]
|
|
53
|
+
build-backend = "uv_build"
|
|
54
|
+
|
|
55
|
+
[dependency-groups]
|
|
56
|
+
dev = [
|
|
57
|
+
"hypothesis>=6.168.2",
|
|
58
|
+
"pytest>=9.1.1",
|
|
59
|
+
"pytest-cov>=7.1.0",
|
|
60
|
+
"ruff>=0.16.9",
|
|
61
|
+
"ty>=0.0.84",
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
[tool.pytest.ini_options]
|
|
65
|
+
testpaths = ["tests"]
|
|
66
|
+
addopts = "--cov=src --cov-report=term-missing --cov-report=term"
|
|
67
|
+
markers = ["slow: long-running check; runs on main and on a schedule (MONITOR.md), not on every pull request"]
|
|
68
|
+
|
|
69
|
+
[tool.uv]
|
|
70
|
+
required-version = ">=0.12.17,<0.13"
|
|
71
|
+
|
|
72
|
+
[tool.ty.src]
|
|
73
|
+
include = [
|
|
74
|
+
"src",
|
|
75
|
+
"tests",
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
[tool.ty.rules]
|
|
79
|
+
all = "error"
|