dfshrink 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dfshrink/__init__.py +31 -0
- dfshrink/_render.py +198 -0
- dfshrink/columns.py +240 -0
- dfshrink/ext/__init__.py +20 -0
- dfshrink/ext/_adapter.py +168 -0
- dfshrink/ext/dataframely.py +114 -0
- dfshrink/ext/pandera.py +133 -0
- dfshrink/ext/patito.py +113 -0
- dfshrink/ext/pytest.py +51 -0
- dfshrink/failure.py +95 -0
- dfshrink/py.typed +0 -0
- dfshrink/shrink.py +333 -0
- dfshrink/values.py +385 -0
- dfshrink-0.1.0.dist-info/METADATA +279 -0
- dfshrink-0.1.0.dist-info/RECORD +17 -0
- dfshrink-0.1.0.dist-info/WHEEL +4 -0
- dfshrink-0.1.0.dist-info/licenses/LICENSE +21 -0
dfshrink/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""dfshrink -- fail-fast data tooling for Polars."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
6
|
+
|
|
7
|
+
from dfshrink.columns import ColumnReduction, minimize_columns
|
|
8
|
+
from dfshrink.failure import Diagnosis, Explainer, Failure
|
|
9
|
+
from dfshrink.shrink import DEFAULT_MAX_EVALS, FailPredicate, Repro, shrink_rows
|
|
10
|
+
from dfshrink.values import Direction, ValueReduction, direction_for_rule, minimize_values
|
|
11
|
+
|
|
12
|
+
try:
|
|
13
|
+
__version__ = version("dfshrink")
|
|
14
|
+
except PackageNotFoundError: # pragma: no cover - only when run from a source tree
|
|
15
|
+
__version__ = "0.0.0+unknown"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"DEFAULT_MAX_EVALS",
|
|
19
|
+
"ColumnReduction",
|
|
20
|
+
"Diagnosis",
|
|
21
|
+
"Direction",
|
|
22
|
+
"Explainer",
|
|
23
|
+
"FailPredicate",
|
|
24
|
+
"Failure",
|
|
25
|
+
"Repro",
|
|
26
|
+
"ValueReduction",
|
|
27
|
+
"direction_for_rule",
|
|
28
|
+
"minimize_columns",
|
|
29
|
+
"minimize_values",
|
|
30
|
+
"shrink_rows",
|
|
31
|
+
]
|
dfshrink/_render.py
ADDED
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
"""Render a repro as pasteable code, a markdown table, or a failure line.
|
|
2
|
+
|
|
3
|
+
:meth:`dfshrink.Repro.to_code` and :meth:`dfshrink.Repro.to_markdown` are the
|
|
4
|
+
public entry points; this module holds the formatting so :mod:`dfshrink.shrink`
|
|
5
|
+
stays about the algorithm. :func:`render_diagnosis_markdown` folds the
|
|
6
|
+
:class:`~dfshrink.Failure` into the same report, so a ticket reads *why* a frame
|
|
7
|
+
failed and shows the row that does it.
|
|
8
|
+
|
|
9
|
+
Rendering is **lossless or loud**: :func:`render_code` rebuilds the frame from
|
|
10
|
+
the data and schema it is about to print and panics when the rebuild does not
|
|
11
|
+
equal the original. A dtype whose values cannot be written as Python literals
|
|
12
|
+
(a column of :class:`polars.Object`, say) raises instead of emitting code that
|
|
13
|
+
would silently build a different frame.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import datetime
|
|
19
|
+
import math
|
|
20
|
+
from decimal import Decimal
|
|
21
|
+
from typing import TYPE_CHECKING, Any
|
|
22
|
+
|
|
23
|
+
import polars as pl
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING:
|
|
26
|
+
from collections.abc import Sequence
|
|
27
|
+
|
|
28
|
+
from dfshrink.failure import Diagnosis, Failure
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def render_code(frame: pl.DataFrame) -> str:
|
|
32
|
+
"""Return a pasteable ``pl.DataFrame(...)`` constructor for ``frame``.
|
|
33
|
+
|
|
34
|
+
Postcondition: ``eval(render_code(frame))`` -- with only ``polars as pl`` in
|
|
35
|
+
scope -- is a :class:`polars.DataFrame` equal to ``frame``. A dtype that
|
|
36
|
+
cannot be rendered without loss raises ``TypeError`` rather than emitting
|
|
37
|
+
code that builds a different frame.
|
|
38
|
+
"""
|
|
39
|
+
data: dict[str, list[Any]] = {}
|
|
40
|
+
dtypes: dict[str, pl.DataType] = {}
|
|
41
|
+
values: list[str] = []
|
|
42
|
+
schema: list[str] = []
|
|
43
|
+
for name, series in zip(frame.columns, frame.iter_columns(), strict=True):
|
|
44
|
+
ready = _ready_values(series)
|
|
45
|
+
try:
|
|
46
|
+
rendered = _render_values(ready)
|
|
47
|
+
except TypeError as exc:
|
|
48
|
+
msg = f"cannot render column {name!r} as code: {exc}"
|
|
49
|
+
raise TypeError(msg) from exc
|
|
50
|
+
data[name] = ready
|
|
51
|
+
dtypes[name] = series.dtype
|
|
52
|
+
values.append(f"{name!r}: {rendered}")
|
|
53
|
+
schema.append(f"{name!r}: {_render_dtype(series.dtype)}")
|
|
54
|
+
|
|
55
|
+
_verify_round_trip(frame, data, dtypes)
|
|
56
|
+
return f"pl.DataFrame(\n {{{', '.join(values)}}},\n schema={{{', '.join(schema)}}},\n)"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def render_markdown(frame: pl.DataFrame) -> str:
|
|
60
|
+
"""Return ``frame`` as a GitHub-flavoured markdown table.
|
|
61
|
+
|
|
62
|
+
Each header carries the column's dtype (``amount (Int64)``), so the table
|
|
63
|
+
documents the schema as well as the values.
|
|
64
|
+
"""
|
|
65
|
+
headers = [f"{name} ({dtype})" for name, dtype in frame.schema.items()]
|
|
66
|
+
lines = [
|
|
67
|
+
f"| {' | '.join(headers)} |",
|
|
68
|
+
f"| {' | '.join('---' for _ in headers)} |",
|
|
69
|
+
*(f"| {' | '.join(_render_cell(value) for value in row)} |" for row in frame.iter_rows()),
|
|
70
|
+
]
|
|
71
|
+
return "\n".join(lines)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def describe_failure(failure: Failure | None) -> str:
|
|
75
|
+
"""A short phrase naming the failing rule and column, for a summary line."""
|
|
76
|
+
if failure is None:
|
|
77
|
+
return "no failure metadata"
|
|
78
|
+
rule, column = failure.rule, failure.column
|
|
79
|
+
if column is not None and rule is not None:
|
|
80
|
+
return f"column {column!r} fails rule {rule!r}"
|
|
81
|
+
if column is not None:
|
|
82
|
+
return f"column {column!r} fails"
|
|
83
|
+
if rule is not None:
|
|
84
|
+
return f"rule {rule!r} fails"
|
|
85
|
+
return "validation failed"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def render_diagnosis_markdown(diagnosis: Diagnosis) -> str:
|
|
89
|
+
"""A ticket-ready report: why the frame failed, then the minimal repro."""
|
|
90
|
+
failure = diagnosis.failure
|
|
91
|
+
lines = [f"Validation failed: {describe_failure(failure)}."]
|
|
92
|
+
if failure is not None and failure.message:
|
|
93
|
+
lines += ["", *(f"> {line}" for line in failure.message.splitlines())]
|
|
94
|
+
lines += ["", "Minimal repro:", "", render_markdown(diagnosis.repro.frame)]
|
|
95
|
+
return "\n".join(lines)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _ready_values(series: pl.Series) -> list[Any]:
|
|
99
|
+
"""The series' values, as the literals :func:`pl.DataFrame` rebuilds from.
|
|
100
|
+
|
|
101
|
+
This is the same data the emitted code carries, so the round-trip check in
|
|
102
|
+
:func:`_verify_round_trip` exercises exactly what ``to_code`` prints.
|
|
103
|
+
Datetime and Duration columns become integer counts in the dtype's own
|
|
104
|
+
unit, which is exact at every unit -- Python datetimes only carry
|
|
105
|
+
microseconds, so a nanosecond column would otherwise lose precision.
|
|
106
|
+
"""
|
|
107
|
+
if isinstance(series.dtype, (pl.Datetime, pl.Duration)):
|
|
108
|
+
return series.cast(pl.Int64).to_list()
|
|
109
|
+
return [_to_literal(value, series.dtype) for value in series.to_list()]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _to_literal(value: Any, dtype: Any) -> Any:
|
|
113
|
+
"""Rewrite one value into the literal form ``dtype`` rebuilds it from."""
|
|
114
|
+
if value is None:
|
|
115
|
+
return None
|
|
116
|
+
if isinstance(dtype, (pl.List, pl.Array)):
|
|
117
|
+
return [_to_literal(item, dtype.inner) for item in value]
|
|
118
|
+
if isinstance(dtype, pl.Struct):
|
|
119
|
+
fields = {field.name: field.dtype for field in dtype.fields}
|
|
120
|
+
return {key: _to_literal(item, fields.get(key)) for key, item in value.items()}
|
|
121
|
+
if isinstance(dtype, (pl.Date, pl.Time)) or isinstance(value, (datetime.date, datetime.time)):
|
|
122
|
+
return value.isoformat()
|
|
123
|
+
if isinstance(dtype, pl.Decimal) or isinstance(value, Decimal):
|
|
124
|
+
return str(value)
|
|
125
|
+
if isinstance(value, datetime.timedelta):
|
|
126
|
+
return value / datetime.timedelta(microseconds=1)
|
|
127
|
+
return value
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _render_values(values: Sequence[Any]) -> str:
|
|
131
|
+
return f"[{', '.join(_render_value(value) for value in values)}]"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _render_value(value: Any) -> str:
|
|
135
|
+
"""A Python literal for a value already reduced by :func:`_to_literal`."""
|
|
136
|
+
if value is None or isinstance(value, (bool, int, str, bytes)):
|
|
137
|
+
return repr(value)
|
|
138
|
+
if isinstance(value, float):
|
|
139
|
+
if math.isnan(value):
|
|
140
|
+
return "float('nan')"
|
|
141
|
+
if math.isinf(value):
|
|
142
|
+
return "float('inf')" if value > 0 else "float('-inf')"
|
|
143
|
+
return repr(value)
|
|
144
|
+
if isinstance(value, (list, tuple)):
|
|
145
|
+
return f"[{', '.join(_render_value(item) for item in value)}]"
|
|
146
|
+
if isinstance(value, dict):
|
|
147
|
+
items = ", ".join(f"{_render_value(k)}: {_render_value(v)}" for k, v in value.items())
|
|
148
|
+
return f"{{{items}}}"
|
|
149
|
+
msg = f"cannot render value of type {type(value).__name__!r} as code"
|
|
150
|
+
raise TypeError(msg)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _render_dtype(dtype: Any) -> str:
|
|
154
|
+
"""A ``pl.<Dtype>(...)`` expression that reconstructs ``dtype`` exactly."""
|
|
155
|
+
if isinstance(dtype, pl.List):
|
|
156
|
+
return f"pl.List({_render_dtype(dtype.inner)})"
|
|
157
|
+
if isinstance(dtype, pl.Array):
|
|
158
|
+
return f"pl.Array({_render_dtype(dtype.inner)}, {dtype.size})"
|
|
159
|
+
if isinstance(dtype, pl.Struct):
|
|
160
|
+
fields = ", ".join(
|
|
161
|
+
f"{field.name!r}: {_render_dtype(field.dtype)}" for field in dtype.fields
|
|
162
|
+
)
|
|
163
|
+
return f"pl.Struct({{{fields}}})"
|
|
164
|
+
if isinstance(dtype, pl.Datetime):
|
|
165
|
+
if dtype.time_zone is None:
|
|
166
|
+
return f"pl.Datetime(time_unit={dtype.time_unit!r})"
|
|
167
|
+
return f"pl.Datetime(time_unit={dtype.time_unit!r}, time_zone={dtype.time_zone!r})"
|
|
168
|
+
if isinstance(dtype, pl.Duration):
|
|
169
|
+
return f"pl.Duration(time_unit={dtype.time_unit!r})"
|
|
170
|
+
if isinstance(dtype, pl.Decimal):
|
|
171
|
+
return f"pl.Decimal(precision={dtype.precision}, scale={dtype.scale})"
|
|
172
|
+
if isinstance(dtype, pl.Enum):
|
|
173
|
+
return f"pl.Enum({dtype.categories.to_list()!r})"
|
|
174
|
+
return f"pl.{dtype.base_type()}"
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _render_cell(value: Any) -> str:
|
|
178
|
+
if value is None:
|
|
179
|
+
return "null"
|
|
180
|
+
if isinstance(value, float) and (math.isnan(value) or math.isinf(value)):
|
|
181
|
+
return _render_value(value)
|
|
182
|
+
text = value if isinstance(value, str) else str(value)
|
|
183
|
+
return text.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "<br>")
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _verify_round_trip(
|
|
187
|
+
frame: pl.DataFrame,
|
|
188
|
+
data: dict[str, list[Any]],
|
|
189
|
+
dtypes: dict[str, pl.DataType],
|
|
190
|
+
) -> None:
|
|
191
|
+
"""Postcondition check: the data and schema we print rebuild ``frame``."""
|
|
192
|
+
msg = f"to_code cannot render this frame without loss; columns: {dict(frame.schema)}"
|
|
193
|
+
try:
|
|
194
|
+
rebuilt = pl.DataFrame(data, schema=dtypes)
|
|
195
|
+
except Exception as exc: # one vocabulary for any rebuild failure
|
|
196
|
+
raise TypeError(msg) from exc
|
|
197
|
+
if not rebuilt.equals(frame):
|
|
198
|
+
raise TypeError(msg)
|
dfshrink/columns.py
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""Drop the columns a failing frame does not need to keep failing.
|
|
2
|
+
|
|
3
|
+
:func:`dfshrink.shrink_rows` minimizes *rows*; :func:`dfshrink.minimize_values`
|
|
4
|
+
moves *values*; :func:`minimize_columns` removes whole *columns*. The result is
|
|
5
|
+
the same failure on the narrowest frame: a schema that flags ``amount`` on a
|
|
6
|
+
frame carrying a debug column reduces to the columns the rule actually reads.
|
|
7
|
+
|
|
8
|
+
Column reduction is **schema-aware**, because a black-box ``DataFrame -> bool``
|
|
9
|
+
predicate cannot tell "still fails for the original reason" from "now fails
|
|
10
|
+
because a column went missing". So the caller passes the adapter's *explainer*
|
|
11
|
+
(``as_failure(schema)``); a candidate column subset is accepted only while the
|
|
12
|
+
explainer still names the same rule and column. That is the contract: the
|
|
13
|
+
frame must keep failing **for the same reason**, not merely keep failing.
|
|
14
|
+
|
|
15
|
+
The "frame must already match the schema's columns and dtypes" precondition
|
|
16
|
+
relaxes here: the reducer may **drop** columns, but never changes a kept
|
|
17
|
+
column's dtype. ``minimality_proven`` spans two dimensions and both are
|
|
18
|
+
reported: :attr:`ColumnReduction.proven` is the column dimension, and
|
|
19
|
+
``repro.minimality_proven`` is the row dimension, re-checked against the
|
|
20
|
+
preserved reason so a column drop cannot make an inherited flag a lie.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass, replace
|
|
26
|
+
from typing import TYPE_CHECKING, override
|
|
27
|
+
|
|
28
|
+
from dfshrink.failure import Diagnosis, Explainer, Failure
|
|
29
|
+
from dfshrink.shrink import DEFAULT_MAX_EVALS, Repro, _split, _without_row
|
|
30
|
+
|
|
31
|
+
if TYPE_CHECKING:
|
|
32
|
+
from collections.abc import Sequence
|
|
33
|
+
|
|
34
|
+
import polars as pl
|
|
35
|
+
|
|
36
|
+
__all__ = ["ColumnReduction", "minimize_columns"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class _Exhausted:
|
|
40
|
+
"""Sentinel: the explainer budget is spent, so the reason is unknown."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
_EXHAUSTED = _Exhausted()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True, slots=True, eq=False)
|
|
47
|
+
class ColumnReduction:
|
|
48
|
+
"""A repro with the columns no failing rule needed removed.
|
|
49
|
+
|
|
50
|
+
``eq=False`` for the same reason as :class:`dfshrink.Repro`: a
|
|
51
|
+
:class:`polars.DataFrame` field would otherwise compare element-wise.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
repro: Repro
|
|
55
|
+
"""The column-reduced repro. Its frame still fails for ``failure``'s reason,
|
|
56
|
+
with the kept columns' dtypes and row order unchanged. Its
|
|
57
|
+
``minimality_proven`` is the *row* dimension re-checked for that reason --
|
|
58
|
+
not for the whole schema, which the reduced frame may no longer satisfy."""
|
|
59
|
+
|
|
60
|
+
failure: Failure
|
|
61
|
+
"""The failure preserved across the reduction -- the reason the reduced
|
|
62
|
+
frame still fails."""
|
|
63
|
+
|
|
64
|
+
dropped_columns: tuple[str, ...]
|
|
65
|
+
"""The removed column names, in original frame order."""
|
|
66
|
+
|
|
67
|
+
proven: bool
|
|
68
|
+
"""``True`` iff the kept column set is 1-minimal for the preserved reason:
|
|
69
|
+
dropping any single remaining column changes or removes that reason."""
|
|
70
|
+
|
|
71
|
+
predicate_calls: int
|
|
72
|
+
"""Explainer calls the column search spent (row shrinking did its own)."""
|
|
73
|
+
|
|
74
|
+
@override
|
|
75
|
+
def __repr__(self) -> str:
|
|
76
|
+
"""A compact repr: dropped count, width, and the proof flag."""
|
|
77
|
+
state = "proven" if self.proven else "not proven"
|
|
78
|
+
return (
|
|
79
|
+
f"ColumnReduction(dropped={len(self.dropped_columns)}, "
|
|
80
|
+
f"columns_left={self.repro.frame.width}, {state}, calls={self.predicate_calls})"
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def minimize_columns(
|
|
85
|
+
source: Repro | Diagnosis,
|
|
86
|
+
explain: Explainer,
|
|
87
|
+
*,
|
|
88
|
+
max_evals: int = DEFAULT_MAX_EVALS,
|
|
89
|
+
) -> ColumnReduction:
|
|
90
|
+
"""Remove every column the failing frame does not need to keep failing.
|
|
91
|
+
|
|
92
|
+
``explain`` is the adapter's failure explainer (``as_failure(schema)``):
|
|
93
|
+
``DataFrame -> Failure | None``. ``source`` may be a
|
|
94
|
+
:class:`~dfshrink.Diagnosis` or a bare :class:`~dfshrink.Repro`; its rows are
|
|
95
|
+
kept and its frame defines the candidate columns.
|
|
96
|
+
|
|
97
|
+
Contract:
|
|
98
|
+
|
|
99
|
+
* Preconditions -- caller's bug, so panic: ``max_evals >= 1``; a source
|
|
100
|
+
frame that fails; a failure the explainer can *name* (a structural
|
|
101
|
+
failure with no rule is refused, because nothing can be preserved).
|
|
102
|
+
* Postcondition: ``repro.frame`` is a column-subset of the input, keeps the
|
|
103
|
+
input's rows and every kept column's dtype, and still fails for the
|
|
104
|
+
preserved reason.
|
|
105
|
+
* Expected failures, returned as values: an exhausted search budget returns
|
|
106
|
+
the frame untouched with ``proven=False``; it never guesses.
|
|
107
|
+
|
|
108
|
+
The search is ddmin over columns, so it reaches 1-minimality in roughly
|
|
109
|
+
``O(k log k)`` explainer calls for ``k`` columns. A column is removed only
|
|
110
|
+
while ``explain`` still reports the *same* rule and column; a drop that
|
|
111
|
+
merely breaks the frame some other way (a missing required column) is
|
|
112
|
+
rejected. The column search gets its own ``max_evals`` budget; naming the
|
|
113
|
+
source reason is setup and always fits (``max_evals >= 1``).
|
|
114
|
+
"""
|
|
115
|
+
if max_evals < 1:
|
|
116
|
+
msg = f"max_evals must be >= 1, got {max_evals}"
|
|
117
|
+
raise ValueError(msg)
|
|
118
|
+
|
|
119
|
+
repro = source.repro if isinstance(source, Diagnosis) else source
|
|
120
|
+
frame = repro.frame
|
|
121
|
+
|
|
122
|
+
tracker = _ReasonTracker(frame, explain, max_evals)
|
|
123
|
+
baseline = tracker.explain(frame)
|
|
124
|
+
assert not isinstance(baseline, _Exhausted), "max_evals >= 1 leaves room for the baseline"
|
|
125
|
+
if baseline is None:
|
|
126
|
+
msg = "minimize_columns needs a failing frame; the source repro passes"
|
|
127
|
+
raise ValueError(msg)
|
|
128
|
+
if baseline.rule is None:
|
|
129
|
+
msg = (
|
|
130
|
+
"minimize_columns needs a named failing rule; the validator reported a "
|
|
131
|
+
"structural failure (missing column or wrong dtype) with no rule to preserve"
|
|
132
|
+
)
|
|
133
|
+
raise ValueError(msg)
|
|
134
|
+
|
|
135
|
+
current = tuple(range(frame.width))
|
|
136
|
+
tracker.last_true = current
|
|
137
|
+
if frame.width >= 2:
|
|
138
|
+
n = 2
|
|
139
|
+
while len(current) >= 2 and not tracker.exhausted:
|
|
140
|
+
reduced, n = _reduce_once(tracker, current, n, baseline)
|
|
141
|
+
if reduced is not None:
|
|
142
|
+
current = reduced
|
|
143
|
+
continue
|
|
144
|
+
if n >= len(current):
|
|
145
|
+
break
|
|
146
|
+
n = min(len(current), 2 * n)
|
|
147
|
+
assert current == tracker.last_true, "postcondition: reduced columns must still fail"
|
|
148
|
+
|
|
149
|
+
reduced_frame = frame.select([frame.columns[i] for i in current])
|
|
150
|
+
rows_proven = repro.minimality_proven and _rows_minimal(reduced_frame, tracker, baseline)
|
|
151
|
+
dropped = tuple(name for i, name in enumerate(frame.columns) if i not in set(current))
|
|
152
|
+
return ColumnReduction(
|
|
153
|
+
repro=replace(repro, frame=reduced_frame, minimality_proven=rows_proven),
|
|
154
|
+
failure=baseline,
|
|
155
|
+
dropped_columns=dropped,
|
|
156
|
+
proven=frame.width < 2 or not tracker.exhausted,
|
|
157
|
+
predicate_calls=tracker.calls,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _reduce_once(
|
|
162
|
+
tracker: _ReasonTracker,
|
|
163
|
+
current: tuple[int, ...],
|
|
164
|
+
n: int,
|
|
165
|
+
reason: Failure,
|
|
166
|
+
) -> tuple[tuple[int, ...] | None, int]:
|
|
167
|
+
"""One ddmin step over columns: prefer a same-reason chunk, then a complement.
|
|
168
|
+
|
|
169
|
+
Returns the reduced index tuple (or ``None`` when neither a chunk nor a
|
|
170
|
+
complement keeps the reason) and the chunk count to use next.
|
|
171
|
+
"""
|
|
172
|
+
chunks = _split(current, n)
|
|
173
|
+
for _, _, chunk in chunks:
|
|
174
|
+
if tracker.holds_on(chunk, reason) is True:
|
|
175
|
+
return chunk, 2
|
|
176
|
+
for start, stop, _ in chunks:
|
|
177
|
+
complement = current[:start] + current[stop:]
|
|
178
|
+
if complement and tracker.holds_on(complement, reason) is True:
|
|
179
|
+
return complement, max(n - 1, 2)
|
|
180
|
+
return None, n
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _rows_minimal(frame: pl.DataFrame, tracker: _ReasonTracker, reason: Failure) -> bool:
|
|
184
|
+
"""Re-check 1-minimality along rows against the preserved reason.
|
|
185
|
+
|
|
186
|
+
Mirrors :func:`dfshrink.shrink_rows`: a one-row frame is trivially minimal
|
|
187
|
+
(the empty candidate is never tested), and a row that can be dropped while
|
|
188
|
+
the reason survives means the inherited flag is no longer true. Returns
|
|
189
|
+
``False`` when the budget runs out before a row is cleared.
|
|
190
|
+
"""
|
|
191
|
+
if frame.height < 2:
|
|
192
|
+
return True
|
|
193
|
+
for row in range(frame.height):
|
|
194
|
+
if tracker.same_reason(_without_row(frame, row), reason) is not False:
|
|
195
|
+
return False
|
|
196
|
+
return True
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class _ReasonTracker:
|
|
200
|
+
"""Budgeted explainer that answers "same failing reason?" over column subsets.
|
|
201
|
+
|
|
202
|
+
Three-state: :meth:`same_reason` returns ``True`` (same rule and column),
|
|
203
|
+
``False`` (a different or absent failure), or ``None`` once the budget is
|
|
204
|
+
spent. ``None`` is never read as either side -- the search stops and the
|
|
205
|
+
result is reported unproven.
|
|
206
|
+
"""
|
|
207
|
+
|
|
208
|
+
def __init__(self, frame: pl.DataFrame, explain: Explainer, max_evals: int) -> None:
|
|
209
|
+
self._frame = frame
|
|
210
|
+
self._explain = explain
|
|
211
|
+
self.max_evals = max_evals
|
|
212
|
+
self.calls = 0
|
|
213
|
+
self.last_true: tuple[int, ...] | None = None
|
|
214
|
+
|
|
215
|
+
@property
|
|
216
|
+
def exhausted(self) -> bool:
|
|
217
|
+
"""``True`` once no further explainer call may be made."""
|
|
218
|
+
return self.calls >= self.max_evals
|
|
219
|
+
|
|
220
|
+
def explain(self, frame: pl.DataFrame) -> Failure | _Exhausted | None:
|
|
221
|
+
"""Explain ``frame``, or return :data:`_EXHAUSTED` when the budget is spent."""
|
|
222
|
+
if self.exhausted:
|
|
223
|
+
return _EXHAUSTED
|
|
224
|
+
self.calls += 1
|
|
225
|
+
return self._explain(frame)
|
|
226
|
+
|
|
227
|
+
def same_reason(self, frame: pl.DataFrame, reason: Failure) -> bool | None:
|
|
228
|
+
"""Whether ``frame`` still fails for ``reason`` (same rule and column)."""
|
|
229
|
+
found = self.explain(frame)
|
|
230
|
+
if isinstance(found, _Exhausted):
|
|
231
|
+
return None
|
|
232
|
+
return found is not None and found.rule == reason.rule and found.column == reason.column
|
|
233
|
+
|
|
234
|
+
def holds_on(self, columns: Sequence[int], reason: Failure) -> bool | None:
|
|
235
|
+
"""Whether selecting ``columns`` still fails for ``reason``."""
|
|
236
|
+
candidate = self._frame.select([self._frame.columns[i] for i in columns])
|
|
237
|
+
outcome = self.same_reason(candidate, reason)
|
|
238
|
+
if outcome is True:
|
|
239
|
+
self.last_true = tuple(columns)
|
|
240
|
+
return outcome
|
dfshrink/ext/__init__.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Per-library adapters that turn validators into failure predicates and explainers.
|
|
2
|
+
|
|
3
|
+
``dfshrink`` core knows nothing about validators: :func:`dfshrink.shrink_rows`
|
|
4
|
+
takes a plain ``DataFrame -> bool`` predicate. Each module here owns the
|
|
5
|
+
adapter for one real library, so importing the module is what pulls in that
|
|
6
|
+
library -- the base ``dfshrink`` package never does.
|
|
7
|
+
|
|
8
|
+
* :mod:`dfshrink.ext.dataframely` -- ``is_valid(df) -> bool`` schemas.
|
|
9
|
+
* :mod:`dfshrink.ext.pandera` -- ``validate(df)``-raises schemas.
|
|
10
|
+
* :mod:`dfshrink.ext.patito` -- ``Model.validate(df)``-raises models.
|
|
11
|
+
|
|
12
|
+
Use them as::
|
|
13
|
+
|
|
14
|
+
from dfshrink.ext.dataframely import shrink_rows, diagnose
|
|
15
|
+
repro = shrink_rows(df, HouseSchema)
|
|
16
|
+
found = diagnose(df, HouseSchema) # repro + why it failed
|
|
17
|
+
|
|
18
|
+
Each module also exposes :func:`as_predicate` for the raw predicate and
|
|
19
|
+
:func:`as_failure` for the raw ``DataFrame -> Failure | None`` explainer.
|
|
20
|
+
"""
|
dfshrink/ext/_adapter.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Shared plumbing for the per-library adapters in :mod:`dfshrink.ext`.
|
|
2
|
+
|
|
3
|
+
Every adapter module turns one validator's failure signal into a
|
|
4
|
+
``DataFrame -> bool`` predicate and hands it to :func:`dfshrink.shrink_rows`.
|
|
5
|
+
The predicate is library-specific; the shrink call is not. Adapters build
|
|
6
|
+
their public ``shrink_rows`` from :func:`make_shrink_rows`, so the wrapper --
|
|
7
|
+
including the keyword-only ``max_evals`` and its default -- is written once and
|
|
8
|
+
cannot drift between libraries. The same adapters build ``diagnose`` from
|
|
9
|
+
:func:`make_diagnose`: it explains *why* the frame fails (rule, column, invalid
|
|
10
|
+
rows) and shrinks -- using the invalid rows as the starting point when the
|
|
11
|
+
validator reports them, the full frame otherwise.
|
|
12
|
+
|
|
13
|
+
This module imports no validator library; the adapters own those imports.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from typing import TYPE_CHECKING, Protocol
|
|
19
|
+
|
|
20
|
+
from dfshrink.columns import minimize_columns
|
|
21
|
+
from dfshrink.failure import Diagnosis, Explainer, Failure
|
|
22
|
+
from dfshrink.shrink import (
|
|
23
|
+
DEFAULT_MAX_EVALS,
|
|
24
|
+
FailPredicate,
|
|
25
|
+
Repro,
|
|
26
|
+
shrink_rows as _shrink_rows,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from collections.abc import Callable
|
|
31
|
+
|
|
32
|
+
import polars as pl
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"DEFAULT_MAX_EVALS",
|
|
36
|
+
"Diagnose",
|
|
37
|
+
"Diagnosis",
|
|
38
|
+
"Explainer",
|
|
39
|
+
"FailPredicate",
|
|
40
|
+
"Failure",
|
|
41
|
+
"Repro",
|
|
42
|
+
"ShrinkRows",
|
|
43
|
+
"make_diagnose",
|
|
44
|
+
"make_shrink_rows",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ShrinkRows[SchemaT](Protocol):
|
|
49
|
+
"""A schema-aware :func:`dfshrink.shrink_rows` bound to one library's predicate."""
|
|
50
|
+
|
|
51
|
+
def __call__(
|
|
52
|
+
self,
|
|
53
|
+
frame: pl.DataFrame,
|
|
54
|
+
schema: SchemaT,
|
|
55
|
+
*,
|
|
56
|
+
max_evals: int = DEFAULT_MAX_EVALS,
|
|
57
|
+
) -> Repro | None: ...
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class Diagnose[SchemaT](Protocol):
|
|
61
|
+
"""A schema-aware :func:`dfshrink.ext.<lib>.diagnose` bound to one library."""
|
|
62
|
+
|
|
63
|
+
def __call__(
|
|
64
|
+
self,
|
|
65
|
+
frame: pl.DataFrame,
|
|
66
|
+
schema: SchemaT,
|
|
67
|
+
*,
|
|
68
|
+
max_evals: int = DEFAULT_MAX_EVALS,
|
|
69
|
+
columns: bool = False,
|
|
70
|
+
) -> Diagnosis | None: ...
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def make_shrink_rows[SchemaT](
|
|
74
|
+
as_predicate: Callable[[SchemaT], FailPredicate],
|
|
75
|
+
) -> ShrinkRows[SchemaT]:
|
|
76
|
+
"""Build the ``shrink_rows(frame, schema, *, max_evals)`` wrapper for an adapter.
|
|
77
|
+
|
|
78
|
+
``as_predicate`` maps a library schema to the failure predicate
|
|
79
|
+
:func:`dfshrink.shrink_rows` minimizes; the returned callable re-applies it
|
|
80
|
+
on each invocation. See :func:`dfshrink.shrink_rows` for the full contract.
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
def shrink_rows(
|
|
84
|
+
frame: pl.DataFrame,
|
|
85
|
+
schema: SchemaT,
|
|
86
|
+
*,
|
|
87
|
+
max_evals: int = DEFAULT_MAX_EVALS,
|
|
88
|
+
) -> Repro | None:
|
|
89
|
+
"""Shrink ``frame`` to a minimal row subset that still fails ``schema``.
|
|
90
|
+
|
|
91
|
+
A thin wrapper over :func:`dfshrink.shrink_rows`; see there for the full
|
|
92
|
+
contract.
|
|
93
|
+
"""
|
|
94
|
+
return _shrink_rows(frame, as_predicate(schema), max_evals=max_evals)
|
|
95
|
+
|
|
96
|
+
return shrink_rows
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def make_diagnose[SchemaT](
|
|
100
|
+
as_failure: Callable[[SchemaT], Explainer],
|
|
101
|
+
as_predicate: Callable[[SchemaT], FailPredicate],
|
|
102
|
+
) -> Diagnose[SchemaT]:
|
|
103
|
+
"""Build the ``diagnose(frame, schema, *, max_evals)`` wrapper for an adapter.
|
|
104
|
+
|
|
105
|
+
``as_failure`` maps a schema to its failure explainer (``DataFrame ->
|
|
106
|
+
Failure | None``); ``as_predicate`` maps it to the failure predicate
|
|
107
|
+
:func:`dfshrink.shrink_rows` minimizes. ``diagnose`` explains first and
|
|
108
|
+
shrinks second: when the explainer reports the invalid rows, shrinking
|
|
109
|
+
starts there instead of over the whole frame, so the validator's own
|
|
110
|
+
failure signal -- not black-box ddmin -- pinpoints the repro.
|
|
111
|
+
|
|
112
|
+
Contract (same shape as :func:`dfshrink.shrink_rows`):
|
|
113
|
+
|
|
114
|
+
* Preconditions -- caller's bug, so panic: an empty ``frame`` or
|
|
115
|
+
``max_evals < 1`` raises ``ValueError``.
|
|
116
|
+
* Expected failure, returned as a value: a non-failing frame returns
|
|
117
|
+
``None``; otherwise a :class:`Diagnosis` whose ``repro`` still fails,
|
|
118
|
+
has >= 1 row, preserves order, and is 1-minimal when
|
|
119
|
+
``minimality_proven`` is ``True``.
|
|
120
|
+
* A validator that reports a failure the predicate does not reproduce is
|
|
121
|
+
an adapter bug and panics.
|
|
122
|
+
* ``columns=True`` removes columns no failing rule needs (see
|
|
123
|
+
:func:`dfshrink.minimize_columns`), preserving the diagnosed failure; the
|
|
124
|
+
column search gets its own ``max_evals`` budget.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
def diagnose(
|
|
128
|
+
frame: pl.DataFrame,
|
|
129
|
+
schema: SchemaT,
|
|
130
|
+
*,
|
|
131
|
+
max_evals: int = DEFAULT_MAX_EVALS,
|
|
132
|
+
columns: bool = False,
|
|
133
|
+
) -> Diagnosis | None:
|
|
134
|
+
"""Explain and shrink ``frame`` against ``schema``.
|
|
135
|
+
|
|
136
|
+
Returns ``None`` when the frame passes, else a :class:`Diagnosis` with
|
|
137
|
+
the minimal failing repro and the reason it fails (when the validator
|
|
138
|
+
exposes one). See :func:`dfshrink.shrink_rows` for the repro contract.
|
|
139
|
+
|
|
140
|
+
With ``columns=True`` the repro is additionally reduced along columns
|
|
141
|
+
via :func:`dfshrink.minimize_columns`, keeping the same failing rule; the
|
|
142
|
+
column search spends its own ``max_evals`` budget after the row search.
|
|
143
|
+
"""
|
|
144
|
+
if frame.height < 1:
|
|
145
|
+
msg = f"frame must have at least one row, got {frame.height}"
|
|
146
|
+
raise ValueError(msg)
|
|
147
|
+
if max_evals < 1:
|
|
148
|
+
msg = f"max_evals must be >= 1, got {max_evals}"
|
|
149
|
+
raise ValueError(msg)
|
|
150
|
+
|
|
151
|
+
explainer = as_failure(schema)
|
|
152
|
+
failure = explainer(frame)
|
|
153
|
+
if failure is None:
|
|
154
|
+
return None
|
|
155
|
+
|
|
156
|
+
start = frame if failure.invalid_rows is None else failure.invalid_rows
|
|
157
|
+
repro = _shrink_rows(start, as_predicate(schema), max_evals=max_evals)
|
|
158
|
+
assert repro is not None, (
|
|
159
|
+
"diagnose: the validator reported a failure the predicate does not reproduce"
|
|
160
|
+
)
|
|
161
|
+
if columns:
|
|
162
|
+
reduced = minimize_columns(
|
|
163
|
+
Diagnosis(repro=repro, failure=failure), explainer, max_evals=max_evals
|
|
164
|
+
)
|
|
165
|
+
repro = reduced.repro
|
|
166
|
+
return Diagnosis(repro=repro, failure=failure)
|
|
167
|
+
|
|
168
|
+
return diagnose
|