dfshrink 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dfshrink-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Tomas Perez Alvarez
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,279 @@
1
+ Metadata-Version: 2.4
2
+ Name: dfshrink
3
+ Version: 0.1.0
4
+ Summary: Turn a failing Polars DataFrame into a minimal reproducible example
5
+ Keywords: polars,dataframe,data-contract,delta-debugging,ddmin,minimal-repro,testing,validation
6
+ Author: Tomas Perez Alvarez
7
+ Author-email: Tomas Perez Alvarez <tomasperezalvarez@gmail.com>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Software Development :: Quality Assurance
17
+ Classifier: Topic :: Software Development :: Testing
18
+ Requires-Dist: polars>=1.44.2
19
+ Requires-Dist: dataframely>=3 ; extra == 'dataframely'
20
+ Requires-Dist: pandera>=0.22 ; extra == 'pandera'
21
+ Requires-Dist: patito>=0.8.6 ; extra == 'patito'
22
+ Requires-Dist: dataframely>=3 ; extra == 'validators'
23
+ Requires-Dist: pandera>=0.22 ; extra == 'validators'
24
+ Requires-Dist: patito>=0.8.6 ; extra == 'validators'
25
+ Requires-Python: >=3.12
26
+ Project-URL: Homepage, https://github.com/Tomperez98/dfshrink
27
+ Project-URL: Repository, https://github.com/Tomperez98/dfshrink
28
+ Project-URL: Issues, https://github.com/Tomperez98/dfshrink/issues
29
+ Project-URL: Changelog, https://github.com/Tomperez98/dfshrink/blob/main/CHANGELOG.md
30
+ Provides-Extra: dataframely
31
+ Provides-Extra: pandera
32
+ Provides-Extra: patito
33
+ Provides-Extra: validators
34
+ Description-Content-Type: text/markdown
35
+
36
+ # dfshrink
37
+
38
+ **Your data contract failed. Get the smallest frame that still breaks it — and a regression test.**
39
+
40
+ A validator tells you the frame is bad; it can't tell you *which* rows. `shrink_rows`
41
+ runs that same predicate over smaller and smaller subsets and returns the minimal
42
+ frame that still fails:
43
+
44
+ ```python
45
+ import polars as pl
46
+ from dfshrink import shrink_rows
47
+
48
+ df = pl.DataFrame(
49
+ {
50
+ "order_id": [1, 1, 1, 2, 2, 3],
51
+ "amount": [10, 20, 30, 5, 7, 9],
52
+ "total": [50, 50, 50, 12, 12, 9],
53
+ }
54
+ )
55
+
56
+
57
+ def bug(df: pl.DataFrame) -> bool:
58
+ """Each order's line items must sum to its declared total."""
59
+ bad = (
60
+ df.group_by("order_id")
61
+ .agg(pl.col("amount").sum().alias("lines"), pl.col("total").first())
62
+ .filter(pl.col("lines") != pl.col("total"))
63
+ )
64
+ return bad.height > 0
65
+
66
+
67
+ repro = shrink_rows(df, bug)
68
+ print(repro)
69
+ print(repro.frame)
70
+ ```
71
+
72
+ ```
73
+ Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)
74
+ shape: (1, 3)
75
+ ┌──────────┬────────┬───────┐
76
+ │ order_id ┆ amount ┆ total │
77
+ │ --- ┆ --- ┆ --- │
78
+ │ i64 ┆ i64 ┆ i64 │
79
+ ╞══════════╪════════╪═══════╡
80
+ │ 1 ┆ 10 ┆ 50 │
81
+ └──────────┴────────┴───────┘
82
+ ```
83
+
84
+ Six rows in, one row out — in four calls to `bug`. The rule is an aggregate, so
85
+ there is no per-row mask to `.filter()` on; "does this subset still fail?" is the
86
+ only question, and shrinking answers it. Paste the row into a regression test with
87
+ `repro.to_code()`, or drop `repro.to_markdown()` into the incident ticket.
88
+
89
+ That's the data engineer's loop: **a contract check fails → the smallest still-bad
90
+ frame and the broken rule → a committed test and a fix** — without hand-slicing a
91
+ warehouse extract.
92
+
93
+ ## What it is
94
+
95
+ dfshrink is the triage step between "the contract check failed" and "here is the
96
+ fix." Point it at the validator you already run — dataframely, pandera, patito, or
97
+ a plain `DataFrame -> bool` — and it returns the smallest frame that still fails,
98
+ plus the rule and column that broke.
99
+
100
+ `shrink_rows(frame, fails, *, max_evals=10_000) -> Repro | None` runs delta
101
+ debugging (`ddmin`) over a Polars frame's rows and returns the smallest subset on
102
+ which `fails` is still `True` — a minimal repro for a data bug, the Python/Polars
103
+ analog of R's `minex::reduce_rows`.
104
+
105
+ ## Install
106
+
107
+ ```bash
108
+ pip install dfshrink # or: uv add dfshrink
109
+ ```
110
+
111
+ Requires Python 3.12+ and Polars. The base package imports only Polars; each
112
+ validator adapter is an extra (`pip install "dfshrink[dataframely]"`,
113
+ `dfshrink[pandera]`, `dfshrink[patito]`).
114
+
115
+ Developing dfshrink itself? From a checkout, `mise install && mise run setup`.
116
+
117
+ **If you can write the failing rule as a per-row mask, use `.filter()` — it's
118
+ simpler.** Shrinking is for when you can't: a validator that returns a single bit
119
+ (`is_valid`, `validate`, patito), an aggregate, or a cross-row interaction. There
120
+ is no mask then; there is only "does this subset still fail?".
121
+
122
+ ## Reading the result
123
+
124
+ ```python
125
+ repro = shrink_rows(df, bug)
126
+
127
+ if repro is None:
128
+ print("the frame does not fail") # expected failure -> None
129
+ else:
130
+ print(repro.frame) # the minimal failing rows
131
+ print(repro.removed_rows) # rows dropped from the input
132
+ print(repro.minimality_proven) # True => removing any one row makes it pass
133
+ ```
134
+
135
+ `fails` is the seam — a lambda, a test assertion, or a schema adapter (next
136
+ section). Keep it pure: shrinking re-runs it many times, so it must be
137
+ deterministic.
138
+
139
+ ### Paste it into a test or a ticket
140
+
141
+ A `Repro` renders itself for wherever the repro is going:
142
+
143
+ ```python
144
+ repro.to_code() # a pl.DataFrame({...}, schema={...}) constructor
145
+ repro.to_markdown() # a markdown table, dtypes in the headers
146
+ str(repro) # 'Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)'
147
+ repro.as_frame() # the replayed frame, to re-run your predicate on
148
+ ```
149
+
150
+ `to_code()` round-trips: `eval(repro.to_code())`, with only `polars as pl` in
151
+ scope, rebuilds an equal frame — so the constructor goes straight into a test.
152
+ A dtype that cannot be rendered without loss (a column of `pl.Object`, say)
153
+ raises `TypeError` rather than emitting code that quietly builds a different
154
+ frame.
155
+
156
+ ## Shrink against dataframely, pandera, or patito
157
+
158
+ A schema tells you *that* a rule broke — not which rows did it:
159
+
160
+ ```text
161
+ dataframely.exc.ValidationError: 1 rules failed validation:
162
+ * Column 'amount' failed validation for 1 rules:
163
+ - 'min' failed for 1 rows
164
+ ```
165
+
166
+ Hand the schema to `shrink_rows` instead, and it returns the row that did it.
167
+ You already wrote the validator, so there's nothing to re-express:
168
+
169
+ ```python
170
+ import dataframely as dy
171
+ from dfshrink.ext.dataframely import shrink_rows
172
+
173
+
174
+ class HouseSchema(dy.Schema):
175
+ amount = dy.Int64(nullable=False, min=0)
176
+
177
+
178
+ repro = shrink_rows(df, HouseSchema) # inverts HouseSchema.is_valid(df)
179
+ ```
180
+
181
+ Same shape for the others — `dfshrink.ext.pandera` wraps `validate(df)`-raises,
182
+ `dfshrink.ext.patito` wraps `Model.validate(df)`:
183
+
184
+ ```python
185
+ import pandera.polars as pa
186
+ from dfshrink.ext.pandera import shrink_rows
187
+
188
+ schema = pa.DataFrameSchema({"amount": pa.Column(int, pa.Check.gt(0))})
189
+ repro = shrink_rows(df, schema)
190
+ ```
191
+
192
+ ```python
193
+ import patito as pt
194
+ from dfshrink.ext.patito import shrink_rows
195
+
196
+
197
+ class House(pt.Model):
198
+ amount: int = pt.Field(ge=0)
199
+
200
+
201
+ repro = shrink_rows(df, House)
202
+ ```
203
+
204
+ Each module also exposes `as_predicate` for the raw `DataFrame -> bool`. The
205
+ frame must already match the schema's columns and dtypes — shrinking only removes
206
+ rows, so a structural mismatch is the caller's bug. Runnable versions:
207
+ [`examples/dataframely_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/dataframely_schema.py),
208
+ [`examples/pandera_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/pandera_schema.py),
209
+ [`examples/patito_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/patito_schema.py), and the
210
+ dependency-free [`examples/predicate.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/predicate.py).
211
+
212
+ ## Fast: ~log₂(n) predicate calls
213
+
214
+ The only cost that scales with your data is your predicate. `ddmin` narrows by
215
+ chunks instead of removing one row at a time (which is `O(n²)` calls), so a bad
216
+ row is isolated in ~log₂(n) calls:
217
+
218
+ | Frame | Bad rows | Result | Predicate calls |
219
+ |---|---|---|---|
220
+ | 20,000 rows | 1 | 1 row | 16–29 |
221
+ | 6 rows | 1 | 1 row | 4 |
222
+
223
+ Call counts depend on where the bad rows sit; the range is measured. Candidates
224
+ are selected with `DataFrame.slice`, so copying the frame is not the cost: with a
225
+ cheap predicate, 20,000 rows shrink to one in ~0.25 ms. Cap the predicate calls
226
+ with `max_evals` (default 10,000) —
227
+ [performance in depth →](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md).
228
+
229
+ ## Who it's for — and when not to use it
230
+
231
+ **It's for data engineers running Polars pipelines with a pass/fail validator** —
232
+ a data contract, a schema check, or a cross-row invariant (totals, uniqueness,
233
+ ordering) you assert in a test, in CI, or during an incident. The payoff is
234
+ turning a failed check into the one row and rule to fix, a regression test to
235
+ commit, and a ticket the upstream owner can act on.
236
+
237
+ It is deliberately narrow:
238
+
239
+ - **Polars only.** pandas and SQL/warehouse data aren't supported today. If the
240
+ bad data lives in a warehouse, materialize the frame first.
241
+ - **Not a validator.** It doesn't define or check rules; it runs *after* one
242
+ fails, using your predicate as the oracle.
243
+ - **Not production monitoring.** It's a developer/CI debugging tool, not a
244
+ data-observability service.
245
+ - **No pipeline attribution.** It won't tell you *which step* (a join, a cast)
246
+ introduced the bad rows — only which rows.
247
+ - **Columns are schema-aware.** It minimizes rows and numeric values; dropping
248
+ columns needs the adapter's explainer (`diagnose(..., columns=True)`), because a
249
+ black-box predicate cannot tell a real failure from a frame that went
250
+ structurally invalid.
251
+ - **Needs a pure, deterministic predicate.** Nondeterminism makes shrinking
252
+ meaningless.
253
+
254
+ ## Learn more
255
+
256
+ The README is the front door; each guide owns one task in depth:
257
+
258
+ - **[Diagnose the failure](https://github.com/Tomperez98/dfshrink/blob/main/docs/diagnose.md)** — get the rule and column the validator flagged, not just the rows.
259
+ - **[Fail with a repro in CI](https://github.com/Tomperez98/dfshrink/blob/main/docs/ci.md)** — `assert_valid` turns a failed job into a minimal repro.
260
+ - **[Move the value to the boundary](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-values.md)** — shrink a bad cell to the last value that still fails.
261
+ - **[Drop the columns the rule does not need](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-columns.md)** — schema-aware column reduction.
262
+ - **[Performance in depth](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md)** — call counts, the worst case, and `max_evals`.
263
+ - **[The contract](https://github.com/Tomperez98/dfshrink/blob/main/docs/contract.md)** — panic/value semantics, minimality guarantees, and the algorithm.
264
+
265
+ ## Development
266
+
267
+ [mise](https://mise.jdx.dev/) drives every task, so a laptop and CI run the same
268
+ commands:
269
+
270
+ ```bash
271
+ mise install # pinned Python and uv
272
+ mise run setup # every dependency, including the validator extras
273
+ mise run ci # the merge gate: format, lint, type, hygiene, tests, examples, build, smoke
274
+ ```
275
+
276
+ While iterating, run one tier: `mise run test`, `mise run lint`, `mise run
277
+ type`. Every diagnostic is an error. See [CONTRIBUTING.md](CONTRIBUTING.md) for
278
+ the pull-request checklist, and [RELEASING.md](RELEASING.md) for how a release
279
+ is cut.
@@ -0,0 +1,244 @@
1
+ # dfshrink
2
+
3
+ **Your data contract failed. Get the smallest frame that still breaks it — and a regression test.**
4
+
5
+ A validator tells you the frame is bad; it can't tell you *which* rows. `shrink_rows`
6
+ runs that same predicate over smaller and smaller subsets and returns the minimal
7
+ frame that still fails:
8
+
9
+ ```python
10
+ import polars as pl
11
+ from dfshrink import shrink_rows
12
+
13
+ df = pl.DataFrame(
14
+ {
15
+ "order_id": [1, 1, 1, 2, 2, 3],
16
+ "amount": [10, 20, 30, 5, 7, 9],
17
+ "total": [50, 50, 50, 12, 12, 9],
18
+ }
19
+ )
20
+
21
+
22
+ def bug(df: pl.DataFrame) -> bool:
23
+ """Each order's line items must sum to its declared total."""
24
+ bad = (
25
+ df.group_by("order_id")
26
+ .agg(pl.col("amount").sum().alias("lines"), pl.col("total").first())
27
+ .filter(pl.col("lines") != pl.col("total"))
28
+ )
29
+ return bad.height > 0
30
+
31
+
32
+ repro = shrink_rows(df, bug)
33
+ print(repro)
34
+ print(repro.frame)
35
+ ```
36
+
37
+ ```
38
+ Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)
39
+ shape: (1, 3)
40
+ ┌──────────┬────────┬───────┐
41
+ │ order_id ┆ amount ┆ total │
42
+ │ --- ┆ --- ┆ --- │
43
+ │ i64 ┆ i64 ┆ i64 │
44
+ ╞══════════╪════════╪═══════╡
45
+ │ 1 ┆ 10 ┆ 50 │
46
+ └──────────┴────────┴───────┘
47
+ ```
48
+
49
+ Six rows in, one row out — in four calls to `bug`. The rule is an aggregate, so
50
+ there is no per-row mask to `.filter()` on; "does this subset still fail?" is the
51
+ only question, and shrinking answers it. Paste the row into a regression test with
52
+ `repro.to_code()`, or drop `repro.to_markdown()` into the incident ticket.
53
+
54
+ That's the data engineer's loop: **a contract check fails → the smallest still-bad
55
+ frame and the broken rule → a committed test and a fix** — without hand-slicing a
56
+ warehouse extract.
57
+
58
+ ## What it is
59
+
60
+ dfshrink is the triage step between "the contract check failed" and "here is the
61
+ fix." Point it at the validator you already run — dataframely, pandera, patito, or
62
+ a plain `DataFrame -> bool` — and it returns the smallest frame that still fails,
63
+ plus the rule and column that broke.
64
+
65
+ `shrink_rows(frame, fails, *, max_evals=10_000) -> Repro | None` runs delta
66
+ debugging (`ddmin`) over a Polars frame's rows and returns the smallest subset on
67
+ which `fails` is still `True` — a minimal repro for a data bug, the Python/Polars
68
+ analog of R's `minex::reduce_rows`.
69
+
70
+ ## Install
71
+
72
+ ```bash
73
+ pip install dfshrink # or: uv add dfshrink
74
+ ```
75
+
76
+ Requires Python 3.12+ and Polars. The base package imports only Polars; each
77
+ validator adapter is an extra (`pip install "dfshrink[dataframely]"`,
78
+ `dfshrink[pandera]`, `dfshrink[patito]`).
79
+
80
+ Developing dfshrink itself? From a checkout, `mise install && mise run setup`.
81
+
82
+ **If you can write the failing rule as a per-row mask, use `.filter()` — it's
83
+ simpler.** Shrinking is for when you can't: a validator that returns a single bit
84
+ (`is_valid`, `validate`, patito), an aggregate, or a cross-row interaction. There
85
+ is no mask then; there is only "does this subset still fail?".
86
+
87
+ ## Reading the result
88
+
89
+ ```python
90
+ repro = shrink_rows(df, bug)
91
+
92
+ if repro is None:
93
+ print("the frame does not fail") # expected failure -> None
94
+ else:
95
+ print(repro.frame) # the minimal failing rows
96
+ print(repro.removed_rows) # rows dropped from the input
97
+ print(repro.minimality_proven) # True => removing any one row makes it pass
98
+ ```
99
+
100
+ `fails` is the seam — a lambda, a test assertion, or a schema adapter (next
101
+ section). Keep it pure: shrinking re-runs it many times, so it must be
102
+ deterministic.
103
+
104
+ ### Paste it into a test or a ticket
105
+
106
+ A `Repro` renders itself for wherever the repro is going:
107
+
108
+ ```python
109
+ repro.to_code() # a pl.DataFrame({...}, schema={...}) constructor
110
+ repro.to_markdown() # a markdown table, dtypes in the headers
111
+ str(repro) # 'Repro(1 row, removed 5 of 6, 4 predicate calls, minimality proven)'
112
+ repro.as_frame() # the replayed frame, to re-run your predicate on
113
+ ```
114
+
115
+ `to_code()` round-trips: `eval(repro.to_code())`, with only `polars as pl` in
116
+ scope, rebuilds an equal frame — so the constructor goes straight into a test.
117
+ A dtype that cannot be rendered without loss (a column of `pl.Object`, say)
118
+ raises `TypeError` rather than emitting code that quietly builds a different
119
+ frame.
120
+
121
+ ## Shrink against dataframely, pandera, or patito
122
+
123
+ A schema tells you *that* a rule broke — not which rows did it:
124
+
125
+ ```text
126
+ dataframely.exc.ValidationError: 1 rules failed validation:
127
+ * Column 'amount' failed validation for 1 rules:
128
+ - 'min' failed for 1 rows
129
+ ```
130
+
131
+ Hand the schema to `shrink_rows` instead, and it returns the row that did it.
132
+ You already wrote the validator, so there's nothing to re-express:
133
+
134
+ ```python
135
+ import dataframely as dy
136
+ from dfshrink.ext.dataframely import shrink_rows
137
+
138
+
139
+ class HouseSchema(dy.Schema):
140
+ amount = dy.Int64(nullable=False, min=0)
141
+
142
+
143
+ repro = shrink_rows(df, HouseSchema) # inverts HouseSchema.is_valid(df)
144
+ ```
145
+
146
+ Same shape for the others — `dfshrink.ext.pandera` wraps `validate(df)`-raises,
147
+ `dfshrink.ext.patito` wraps `Model.validate(df)`:
148
+
149
+ ```python
150
+ import pandera.polars as pa
151
+ from dfshrink.ext.pandera import shrink_rows
152
+
153
+ schema = pa.DataFrameSchema({"amount": pa.Column(int, pa.Check.gt(0))})
154
+ repro = shrink_rows(df, schema)
155
+ ```
156
+
157
+ ```python
158
+ import patito as pt
159
+ from dfshrink.ext.patito import shrink_rows
160
+
161
+
162
+ class House(pt.Model):
163
+ amount: int = pt.Field(ge=0)
164
+
165
+
166
+ repro = shrink_rows(df, House)
167
+ ```
168
+
169
+ Each module also exposes `as_predicate` for the raw `DataFrame -> bool`. The
170
+ frame must already match the schema's columns and dtypes — shrinking only removes
171
+ rows, so a structural mismatch is the caller's bug. Runnable versions:
172
+ [`examples/dataframely_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/dataframely_schema.py),
173
+ [`examples/pandera_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/pandera_schema.py),
174
+ [`examples/patito_schema.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/patito_schema.py), and the
175
+ dependency-free [`examples/predicate.py`](https://github.com/Tomperez98/dfshrink/blob/main/examples/predicate.py).
176
+
177
+ ## Fast: ~log₂(n) predicate calls
178
+
179
+ The only cost that scales with your data is your predicate. `ddmin` narrows by
180
+ chunks instead of removing one row at a time (which is `O(n²)` calls), so a bad
181
+ row is isolated in ~log₂(n) calls:
182
+
183
+ | Frame | Bad rows | Result | Predicate calls |
184
+ |---|---|---|---|
185
+ | 20,000 rows | 1 | 1 row | 16–29 |
186
+ | 6 rows | 1 | 1 row | 4 |
187
+
188
+ Call counts depend on where the bad rows sit; the range is measured. Candidates
189
+ are selected with `DataFrame.slice`, so copying the frame is not the cost: with a
190
+ cheap predicate, 20,000 rows shrink to one in ~0.25 ms. Cap the predicate calls
191
+ with `max_evals` (default 10,000) —
192
+ [performance in depth →](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md).
193
+
194
+ ## Who it's for — and when not to use it
195
+
196
+ **It's for data engineers running Polars pipelines with a pass/fail validator** —
197
+ a data contract, a schema check, or a cross-row invariant (totals, uniqueness,
198
+ ordering) you assert in a test, in CI, or during an incident. The payoff is
199
+ turning a failed check into the one row and rule to fix, a regression test to
200
+ commit, and a ticket the upstream owner can act on.
201
+
202
+ It is deliberately narrow:
203
+
204
+ - **Polars only.** pandas and SQL/warehouse data aren't supported today. If the
205
+ bad data lives in a warehouse, materialize the frame first.
206
+ - **Not a validator.** It doesn't define or check rules; it runs *after* one
207
+ fails, using your predicate as the oracle.
208
+ - **Not production monitoring.** It's a developer/CI debugging tool, not a
209
+ data-observability service.
210
+ - **No pipeline attribution.** It won't tell you *which step* (a join, a cast)
211
+ introduced the bad rows — only which rows.
212
+ - **Columns are schema-aware.** It minimizes rows and numeric values; dropping
213
+ columns needs the adapter's explainer (`diagnose(..., columns=True)`), because a
214
+ black-box predicate cannot tell a real failure from a frame that went
215
+ structurally invalid.
216
+ - **Needs a pure, deterministic predicate.** Nondeterminism makes shrinking
217
+ meaningless.
218
+
219
+ ## Learn more
220
+
221
+ The README is the front door; each guide owns one task in depth:
222
+
223
+ - **[Diagnose the failure](https://github.com/Tomperez98/dfshrink/blob/main/docs/diagnose.md)** — get the rule and column the validator flagged, not just the rows.
224
+ - **[Fail with a repro in CI](https://github.com/Tomperez98/dfshrink/blob/main/docs/ci.md)** — `assert_valid` turns a failed job into a minimal repro.
225
+ - **[Move the value to the boundary](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-values.md)** — shrink a bad cell to the last value that still fails.
226
+ - **[Drop the columns the rule does not need](https://github.com/Tomperez98/dfshrink/blob/main/docs/minimize-columns.md)** — schema-aware column reduction.
227
+ - **[Performance in depth](https://github.com/Tomperez98/dfshrink/blob/main/docs/performance.md)** — call counts, the worst case, and `max_evals`.
228
+ - **[The contract](https://github.com/Tomperez98/dfshrink/blob/main/docs/contract.md)** — panic/value semantics, minimality guarantees, and the algorithm.
229
+
230
+ ## Development
231
+
232
+ [mise](https://mise.jdx.dev/) drives every task, so a laptop and CI run the same
233
+ commands:
234
+
235
+ ```bash
236
+ mise install # pinned Python and uv
237
+ mise run setup # every dependency, including the validator extras
238
+ mise run ci # the merge gate: format, lint, type, hygiene, tests, examples, build, smoke
239
+ ```
240
+
241
+ While iterating, run one tier: `mise run test`, `mise run lint`, `mise run
242
+ type`. Every diagnostic is an error. See [CONTRIBUTING.md](CONTRIBUTING.md) for
243
+ the pull-request checklist, and [RELEASING.md](RELEASING.md) for how a release
244
+ is cut.
@@ -0,0 +1,79 @@
1
+ [project]
2
+ name = "dfshrink"
3
+ version = "0.1.0"
4
+ description = "Turn a failing Polars DataFrame into a minimal reproducible example"
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ requires-python = ">=3.12"
9
+ keywords = [
10
+ "polars",
11
+ "dataframe",
12
+ "data-contract",
13
+ "delta-debugging",
14
+ "ddmin",
15
+ "minimal-repro",
16
+ "testing",
17
+ "validation",
18
+ ]
19
+ classifiers = [
20
+ "Development Status :: 3 - Alpha",
21
+ "Intended Audience :: Developers",
22
+ "Operating System :: OS Independent",
23
+ "Programming Language :: Python :: 3",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Topic :: Software Development :: Quality Assurance",
27
+ "Topic :: Software Development :: Testing",
28
+ ]
29
+ dependencies = ["polars>=1.44.2"]
30
+
31
+ [[project.authors]]
32
+ name = "Tomas Perez Alvarez"
33
+ email = "tomasperezalvarez@gmail.com"
34
+
35
+ [project.urls]
36
+ Homepage = "https://github.com/Tomperez98/dfshrink"
37
+ Repository = "https://github.com/Tomperez98/dfshrink"
38
+ Issues = "https://github.com/Tomperez98/dfshrink/issues"
39
+ Changelog = "https://github.com/Tomperez98/dfshrink/blob/main/CHANGELOG.md"
40
+
41
+ [project.optional-dependencies]
42
+ dataframely = ["dataframely>=3"]
43
+ pandera = ["pandera>=0.22"]
44
+ patito = ["patito>=0.8.6"]
45
+ validators = [
46
+ "dataframely>=3",
47
+ "pandera>=0.22",
48
+ "patito>=0.8.6",
49
+ ]
50
+
51
+ [build-system]
52
+ requires = ["uv_build>=0.12.17,<0.13.0"]
53
+ build-backend = "uv_build"
54
+
55
+ [dependency-groups]
56
+ dev = [
57
+ "hypothesis>=6.168.2",
58
+ "pytest>=9.1.1",
59
+ "pytest-cov>=7.1.0",
60
+ "ruff>=0.16.9",
61
+ "ty>=0.0.84",
62
+ ]
63
+
64
+ [tool.pytest.ini_options]
65
+ testpaths = ["tests"]
66
+ addopts = "--cov=src --cov-report=term-missing --cov-report=term"
67
+ markers = ["slow: long-running check; runs on main and on a schedule (MONITOR.md), not on every pull request"]
68
+
69
+ [tool.uv]
70
+ required-version = ">=0.12.17,<0.13"
71
+
72
+ [tool.ty.src]
73
+ include = [
74
+ "src",
75
+ "tests",
76
+ ]
77
+
78
+ [tool.ty.rules]
79
+ all = "error"