cleanframe-engine 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanframe/__init__.py +169 -0
- cleanframe/__main__.py +5 -0
- cleanframe/_util.py +438 -0
- cleanframe/_version.py +1 -0
- cleanframe/api.py +559 -0
- cleanframe/cli.py +688 -0
- cleanframe/codegen.py +666 -0
- cleanframe/dataio.py +506 -0
- cleanframe/detectors/__init__.py +43 -0
- cleanframe/detectors/base.py +221 -0
- cleanframe/detectors/categories.py +199 -0
- cleanframe/detectors/contacts.py +105 -0
- cleanframe/detectors/currency.py +112 -0
- cleanframe/detectors/dates.py +204 -0
- cleanframe/detectors/dedup.py +150 -0
- cleanframe/detectors/nulls.py +109 -0
- cleanframe/detectors/outliers.py +73 -0
- cleanframe/detectors/schema_mapping.py +125 -0
- cleanframe/detectors/text.py +105 -0
- cleanframe/detectors/units.py +86 -0
- cleanframe/diff.py +369 -0
- cleanframe/drift.py +283 -0
- cleanframe/errors.py +66 -0
- cleanframe/executor.py +229 -0
- cleanframe/fingerprint.py +83 -0
- cleanframe/issues.py +186 -0
- cleanframe/llm.py +811 -0
- cleanframe/ops.py +1245 -0
- cleanframe/planner.py +353 -0
- cleanframe/profile.py +413 -0
- cleanframe/py.typed +1 -0
- cleanframe/quality.py +81 -0
- cleanframe/readfix.py +160 -0
- cleanframe/recipe.py +398 -0
- cleanframe/report.py +345 -0
- cleanframe/result.py +144 -0
- cleanframe/schema.py +259 -0
- cleanframe/streaming.py +354 -0
- cleanframe/types.py +119 -0
- cleanframe/validate.py +363 -0
- cleanframe/workbook.py +370 -0
- cleanframe_engine-0.3.0.dist-info/METADATA +323 -0
- cleanframe_engine-0.3.0.dist-info/RECORD +46 -0
- cleanframe_engine-0.3.0.dist-info/WHEEL +4 -0
- cleanframe_engine-0.3.0.dist-info/entry_points.txt +2 -0
- cleanframe_engine-0.3.0.dist-info/licenses/LICENSE +201 -0
cleanframe/report.py
ADDED
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
"""Self-contained HTML reports — profiling and cleaning results.
|
|
2
|
+
|
|
3
|
+
Two renderers, both returning a complete, standalone HTML document (inline CSS, no
|
|
4
|
+
external assets, light/dark aware):
|
|
5
|
+
|
|
6
|
+
* :func:`render_profile_report` — what ``cleanframe report file.csv`` produces:
|
|
7
|
+
quality score, detected issues, and a column-by-column diagnosis.
|
|
8
|
+
* :func:`render_clean_report` — what a clean produces: the cell-level diff, renames,
|
|
9
|
+
dropped rows, and the quarantine.
|
|
10
|
+
|
|
11
|
+
User data flows into these, so the Jinja environment runs with ``autoescape`` on —
|
|
12
|
+
a column literally named ``<script>`` renders as text, never markup.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
import pandas as pd
|
|
20
|
+
from jinja2 import Environment, select_autoescape
|
|
21
|
+
|
|
22
|
+
from ._version import __version__
|
|
23
|
+
from .diff import CellDiff
|
|
24
|
+
from .issues import Issues
|
|
25
|
+
from .profile import ColumnProfile, DataFrameProfile
|
|
26
|
+
from .quality import QualityScore, quality_score
|
|
27
|
+
|
|
28
|
+
_env = Environment(autoescape=select_autoescape(default=True, default_for_string=True))
|
|
29
|
+
|
|
30
|
+
#: Colour per semantic type badge.
|
|
31
|
+
_TYPE_COLORS = {
|
|
32
|
+
"email": "#2563eb", "phone": "#7c3aed", "date": "#0891b2", "datetime": "#0891b2",
|
|
33
|
+
"currency": "#16a34a", "integer": "#0d9488", "float": "#0d9488", "boolean": "#9333ea",
|
|
34
|
+
"categorical": "#db2777", "id": "#64748b", "url": "#2563eb", "text": "#475569",
|
|
35
|
+
"empty": "#94a3b8",
|
|
36
|
+
}
|
|
37
|
+
_SEV_COLORS = {"error": "#dc2626", "warning": "#d97706", "info": "#2563eb"}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
_CSS = """
|
|
41
|
+
:root {
|
|
42
|
+
--bg: #ffffff; --fg: #0f172a; --muted: #64748b; --card: #f8fafc;
|
|
43
|
+
--border: #e2e8f0; --accent: #4f46e5; --chip-bg: #eef2ff;
|
|
44
|
+
}
|
|
45
|
+
@media (prefers-color-scheme: dark) {
|
|
46
|
+
:root {
|
|
47
|
+
--bg: #0b1120; --fg: #e2e8f0; --muted: #94a3b8; --card: #131c31;
|
|
48
|
+
--border: #24304a; --accent: #818cf8; --chip-bg: #1e253c;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
* { box-sizing: border-box; }
|
|
52
|
+
body {
|
|
53
|
+
margin: 0; background: var(--bg); color: var(--fg);
|
|
54
|
+
font: 15px/1.55 -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Helvetica, Arial, sans-serif;
|
|
55
|
+
}
|
|
56
|
+
.wrap { max-width: 1040px; margin: 0 auto; padding: 32px 20px 80px; }
|
|
57
|
+
header.masthead { display: flex; align-items: baseline; gap: 12px; flex-wrap: wrap;
|
|
58
|
+
border-bottom: 1px solid var(--border); padding-bottom: 16px; margin-bottom: 24px; }
|
|
59
|
+
header.masthead h1 { font-size: 22px; margin: 0; letter-spacing: -0.02em; }
|
|
60
|
+
header.masthead .sub { color: var(--muted); font-size: 14px; }
|
|
61
|
+
.grid { display: grid; gap: 16px; }
|
|
62
|
+
.stats { grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); margin-bottom: 28px; }
|
|
63
|
+
.tile { background: var(--card); border: 1px solid var(--border); border-radius: 12px; padding: 16px 18px; }
|
|
64
|
+
.tile .k { color: var(--muted); font-size: 12px; text-transform: uppercase; letter-spacing: .04em; }
|
|
65
|
+
.tile .v { font-size: 26px; font-weight: 650; margin-top: 4px; letter-spacing: -0.02em; }
|
|
66
|
+
.score { display: flex; align-items: center; gap: 16px; }
|
|
67
|
+
.score .ring { width: 74px; height: 74px; border-radius: 50%; display: grid; place-items: center;
|
|
68
|
+
color: #fff; font-weight: 700; font-size: 22px; flex: none; }
|
|
69
|
+
.score .meta .g { font-weight: 650; font-size: 16px; } .score .meta .l { color: var(--muted); font-size: 13px; }
|
|
70
|
+
h2 { font-size: 15px; text-transform: uppercase; letter-spacing: .05em; color: var(--muted);
|
|
71
|
+
margin: 34px 0 14px; font-weight: 600; }
|
|
72
|
+
.chip { display: inline-block; padding: 2px 9px; border-radius: 999px; font-size: 12px; font-weight: 600;
|
|
73
|
+
color: #fff; }
|
|
74
|
+
.badge { display:inline-block; padding: 1px 8px; border-radius: 6px; font-size: 11px; font-weight: 600;
|
|
75
|
+
background: var(--chip-bg); color: var(--accent); }
|
|
76
|
+
.issue { display: flex; gap: 10px; align-items: baseline; padding: 9px 12px; border: 1px solid var(--border);
|
|
77
|
+
border-radius: 9px; background: var(--card); margin-bottom: 8px; }
|
|
78
|
+
.issue .dot { width: 8px; height: 8px; border-radius: 50%; flex: none; margin-top: 6px; }
|
|
79
|
+
.issue .col { font-weight: 600; font-size: 13px; } .issue .msg { font-size: 14px; }
|
|
80
|
+
.issue .kind { color: var(--muted); font-size: 12px; }
|
|
81
|
+
.cols { grid-template-columns: repeat(auto-fill, minmax(300px, 1fr)); }
|
|
82
|
+
.colcard { background: var(--card); border: 1px solid var(--border); border-radius: 12px; padding: 16px; }
|
|
83
|
+
.colcard .name { font-weight: 650; font-size: 15px; word-break: break-word; }
|
|
84
|
+
.colcard .row { display: flex; justify-content: space-between; font-size: 13px; color: var(--muted);
|
|
85
|
+
margin-top: 4px; } .colcard .row b { color: var(--fg); font-weight: 600; }
|
|
86
|
+
.bar { height: 6px; border-radius: 4px; background: var(--accent); opacity: .85; }
|
|
87
|
+
.mc { margin-top: 10px; } .mc .item { display: grid; grid-template-columns: 1fr auto; gap: 8px;
|
|
88
|
+
align-items: center; font-size: 12px; margin-top: 5px; } .mc .track { background: var(--border);
|
|
89
|
+
border-radius: 4px; overflow: hidden; }
|
|
90
|
+
table.data { width: 100%; border-collapse: collapse; font-size: 13px; }
|
|
91
|
+
.scroll { overflow-x: auto; border: 1px solid var(--border); border-radius: 10px; }
|
|
92
|
+
table.data th, table.data td { text-align: left; padding: 8px 10px; border-bottom: 1px solid var(--border);
|
|
93
|
+
white-space: nowrap; } table.data th { color: var(--muted); font-weight: 600; background: var(--card); }
|
|
94
|
+
.before { color: #dc2626; } .after { color: #16a34a; }
|
|
95
|
+
.arrow { color: var(--muted); padding: 0 6px; }
|
|
96
|
+
code { background: var(--chip-bg); padding: 1px 5px; border-radius: 5px; font-size: 12px; }
|
|
97
|
+
.empty { color: var(--muted); font-style: italic; }
|
|
98
|
+
footer { margin-top: 48px; color: var(--muted); font-size: 12px; border-top: 1px solid var(--border);
|
|
99
|
+
padding-top: 16px; }
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
_BASE = """<!doctype html>
|
|
103
|
+
<html lang="en">
|
|
104
|
+
<head>
|
|
105
|
+
<meta charset="utf-8"/>
|
|
106
|
+
<meta name="viewport" content="width=device-width, initial-scale=1"/>
|
|
107
|
+
<title>{{ title }}</title>
|
|
108
|
+
<style>{{ css|safe }}</style>
|
|
109
|
+
</head>
|
|
110
|
+
<body><div class="wrap">
|
|
111
|
+
<header class="masthead">
|
|
112
|
+
<h1>CleanFrame</h1>
|
|
113
|
+
<span class="sub">{{ subtitle }}</span>
|
|
114
|
+
</header>
|
|
115
|
+
{{ body|safe }}
|
|
116
|
+
<footer>Generated by CleanFrame v{{ version }}. Deterministic · reproducible · reviewable.</footer>
|
|
117
|
+
</div></body></html>
|
|
118
|
+
"""
|
|
119
|
+
|
|
120
|
+
_PROFILE_BODY = """
|
|
121
|
+
<div class="grid stats">
|
|
122
|
+
<div class="tile">
|
|
123
|
+
<div class="k">Quality score</div>
|
|
124
|
+
<div class="score">
|
|
125
|
+
<div class="ring" style="background: {{ q.color }}">{{ q.score }}</div>
|
|
126
|
+
<div class="meta"><div class="g">Grade {{ q.grade }}</div><div class="l">{{ q.label }}</div></div>
|
|
127
|
+
</div>
|
|
128
|
+
</div>
|
|
129
|
+
<div class="tile"><div class="k">Rows</div><div class="v">{{ n_rows }}</div></div>
|
|
130
|
+
<div class="tile"><div class="k">Columns</div><div class="v">{{ n_cols }}</div></div>
|
|
131
|
+
<div class="tile"><div class="k">Issues</div><div class="v">{{ n_issues }}</div>
|
|
132
|
+
<div class="l" style="color:var(--muted);font-size:12px">
|
|
133
|
+
{{ n_error }} error · {{ n_warn }} warning · {{ n_info }} info</div></div>
|
|
134
|
+
<div class="tile"><div class="k">Duplicate rows</div><div class="v">{{ dup_rows }}</div></div>
|
|
135
|
+
</div>
|
|
136
|
+
|
|
137
|
+
{% if issues %}
|
|
138
|
+
<h2>Detected issues</h2>
|
|
139
|
+
{% for i in issues %}
|
|
140
|
+
<div class="issue">
|
|
141
|
+
<span class="dot" style="background: {{ i.color }}"></span>
|
|
142
|
+
<div>
|
|
143
|
+
{% if i.column %}<span class="col">{{ i.column }}</span> {% endif %}
|
|
144
|
+
<span class="msg">{{ i.message }}</span>
|
|
145
|
+
<div class="kind"><span class="badge">{{ i.kind }}</span> · {{ i.severity }}
|
|
146
|
+
· via {{ i.detector }}{% if i.confidence < 1 %} · {{ i.confidence }} conf{% endif %}</div>
|
|
147
|
+
</div>
|
|
148
|
+
</div>
|
|
149
|
+
{% endfor %}
|
|
150
|
+
{% else %}<h2>Detected issues</h2><p class="empty">No issues detected — this file looks clean.</p>{% endif %}
|
|
151
|
+
|
|
152
|
+
<h2>Columns</h2>
|
|
153
|
+
<div class="grid cols">
|
|
154
|
+
{% for c in columns %}
|
|
155
|
+
<div class="colcard">
|
|
156
|
+
<div class="name">{{ c.name }}
|
|
157
|
+
<span class="chip" style="background: {{ c.color }}">{{ c.semantic_type }}</span></div>
|
|
158
|
+
<div class="row"><span>dtype</span><b>{{ c.dtype }}</b></div>
|
|
159
|
+
<div class="row"><span>missing</span><b>{{ c.null_pct }}%</b></div>
|
|
160
|
+
<div class="row"><span>unique</span><b>{{ c.unique_count }}</b></div>
|
|
161
|
+
{% if c.numeric %}<div class="row"><span>range</span><b>{{ c.numeric }}</b></div>{% endif %}
|
|
162
|
+
{% if c.examples %}<div class="row" style="display:block"><span>examples</span>
|
|
163
|
+
<div style="color:var(--fg);margin-top:3px">{% for e in c.examples %}<code>{{ e }}</code> {% endfor %}</div></div>{% endif %}
|
|
164
|
+
{% if c.most_common %}<div class="mc">
|
|
165
|
+
{% for m in c.most_common %}<div class="item"><div class="track"><div class="bar" style="width: {{ m.pct }}%"></div></div>
|
|
166
|
+
<span>{{ m.value }} ({{ m.count }})</span></div>{% endfor %}</div>{% endif %}
|
|
167
|
+
</div>
|
|
168
|
+
{% endfor %}
|
|
169
|
+
</div>
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
_CLEAN_BODY = """
|
|
173
|
+
<div class="grid stats">
|
|
174
|
+
{% if q %}<div class="tile"><div class="k">Quality (before)</div>
|
|
175
|
+
<div class="score"><div class="ring" style="background: {{ q.color }}">{{ q.score }}</div>
|
|
176
|
+
<div class="meta"><div class="g">Grade {{ q.grade }}</div><div class="l">{{ q.label }}</div></div></div></div>{% endif %}
|
|
177
|
+
<div class="tile"><div class="k">Cells changed</div><div class="v">{{ d.changed_cells }}</div></div>
|
|
178
|
+
<div class="tile"><div class="k">Columns changed</div><div class="v">{{ d.changed_columns }}</div></div>
|
|
179
|
+
<div class="tile"><div class="k">Rows</div><div class="v">{{ d.rows_before }} → {{ d.rows_after }}</div>
|
|
180
|
+
<div class="l" style="color:var(--muted);font-size:12px">{{ d.rows_dropped }} dropped</div></div>
|
|
181
|
+
<div class="tile"><div class="k">Quarantined</div><div class="v">{{ quarantine_n }}</div></div>
|
|
182
|
+
</div>
|
|
183
|
+
|
|
184
|
+
{% if renames or added or removed %}
|
|
185
|
+
<h2>Structure</h2>
|
|
186
|
+
{% for src, dst in renames.items() %}<div class="issue"><span class="dot" style="background:#2563eb"></span>
|
|
187
|
+
<div><span class="msg">Renamed <code>{{ src }}</code> <span class="arrow">→</span> <code>{{ dst }}</code></span></div></div>{% endfor %}
|
|
188
|
+
{% for a in added %}<div class="issue"><span class="dot" style="background:#16a34a"></span>
|
|
189
|
+
<div><span class="msg">Added column <code>{{ a }}</code></span></div></div>{% endfor %}
|
|
190
|
+
{% for r in removed %}<div class="issue"><span class="dot" style="background:#dc2626"></span>
|
|
191
|
+
<div><span class="msg">Removed column <code>{{ r }}</code></span></div></div>{% endfor %}
|
|
192
|
+
{% endif %}
|
|
193
|
+
|
|
194
|
+
{% if changes_by_col %}
|
|
195
|
+
<h2>Changed cells</h2>
|
|
196
|
+
{% for col, rows in changes_by_col.items() %}
|
|
197
|
+
<div style="margin-bottom:18px">
|
|
198
|
+
<div style="font-weight:650;margin-bottom:8px">{{ col }} <span class="badge">{{ rows.total }} changed</span></div>
|
|
199
|
+
<div class="scroll"><table class="data">
|
|
200
|
+
<thead><tr><th>row</th><th>before</th><th></th><th>after</th></tr></thead>
|
|
201
|
+
<tbody>
|
|
202
|
+
{% for r in rows.sample %}<tr><td>{{ r.row_id }}</td>
|
|
203
|
+
<td class="before">{{ r.before }}</td><td class="arrow">→</td><td class="after">{{ r.after }}</td></tr>{% endfor %}
|
|
204
|
+
</tbody></table></div>
|
|
205
|
+
{% if rows.total > rows.sample|length %}<div class="l" style="color:var(--muted);font-size:12px;margin-top:6px">
|
|
206
|
+
… and {{ rows.total - rows.sample|length }} more</div>{% endif %}
|
|
207
|
+
</div>
|
|
208
|
+
{% endfor %}
|
|
209
|
+
{% else %}<h2>Changed cells</h2><p class="empty">No cell values changed.</p>{% endif %}
|
|
210
|
+
|
|
211
|
+
{% if quarantine_cols %}
|
|
212
|
+
<h2>Quarantine <span class="badge">{{ quarantine_n }} row(s)</span></h2>
|
|
213
|
+
<div class="scroll"><table class="data">
|
|
214
|
+
<thead><tr>{% for h in quarantine_cols %}<th>{{ h }}</th>{% endfor %}</tr></thead>
|
|
215
|
+
<tbody>{% for row in quarantine_rows %}<tr>{% for cell in row %}<td>{{ cell }}</td>{% endfor %}</tr>{% endfor %}</tbody>
|
|
216
|
+
</table></div>
|
|
217
|
+
{% endif %}
|
|
218
|
+
"""
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _fmt_cell(value: Any) -> str:
|
|
222
|
+
if value is None or (isinstance(value, float) and pd.isna(value)):
|
|
223
|
+
return "∅"
|
|
224
|
+
try:
|
|
225
|
+
if pd.isna(value):
|
|
226
|
+
return "∅"
|
|
227
|
+
except (TypeError, ValueError):
|
|
228
|
+
pass
|
|
229
|
+
return str(value)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _column_context(cp: ColumnProfile) -> dict:
|
|
233
|
+
top = cp.most_common[:5]
|
|
234
|
+
max_count = max((c for _, c in top), default=1) or 1
|
|
235
|
+
numeric = None
|
|
236
|
+
if cp.numeric_stats:
|
|
237
|
+
numeric = f"{cp.numeric_stats['min']:g} – {cp.numeric_stats['max']:g}"
|
|
238
|
+
return {
|
|
239
|
+
"name": cp.name,
|
|
240
|
+
"dtype": cp.dtype,
|
|
241
|
+
"semantic_type": cp.semantic_type,
|
|
242
|
+
"color": _TYPE_COLORS.get(cp.semantic_type, "#475569"),
|
|
243
|
+
"null_pct": round(cp.null_fraction * 100, 1),
|
|
244
|
+
"unique_count": cp.unique_count,
|
|
245
|
+
"numeric": numeric,
|
|
246
|
+
"examples": [_fmt_cell(v) for v in cp.sample_values[:4]],
|
|
247
|
+
"most_common": [
|
|
248
|
+
{"value": _fmt_cell(v), "count": n, "pct": round(100 * n / max_count)} for v, n in top
|
|
249
|
+
],
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _issue_context(issues: Issues) -> list[dict]:
|
|
254
|
+
ordered = sorted(issues, key=lambda i: (-i.severity.rank, str(i.column or ""), i.kind))
|
|
255
|
+
return [
|
|
256
|
+
{
|
|
257
|
+
"column": i.column,
|
|
258
|
+
"message": i.message,
|
|
259
|
+
"kind": i.kind,
|
|
260
|
+
"severity": i.severity.value,
|
|
261
|
+
"detector": i.detector,
|
|
262
|
+
"confidence": round(i.confidence, 2),
|
|
263
|
+
"color": _SEV_COLORS.get(i.severity.value, "#64748b"),
|
|
264
|
+
}
|
|
265
|
+
for i in ordered
|
|
266
|
+
]
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def render_profile_report(
|
|
270
|
+
profile: DataFrameProfile,
|
|
271
|
+
issues: Issues,
|
|
272
|
+
*,
|
|
273
|
+
source: str | None = None,
|
|
274
|
+
quality: QualityScore | None = None,
|
|
275
|
+
) -> str:
|
|
276
|
+
"""Render the profiling report (issues, quality score, per-column diagnosis)."""
|
|
277
|
+
quality = quality or quality_score(profile, issues)
|
|
278
|
+
issue_ctx = _issue_context(issues)
|
|
279
|
+
body = _env.from_string(_PROFILE_BODY).render(
|
|
280
|
+
q=quality,
|
|
281
|
+
n_rows=profile.n_rows,
|
|
282
|
+
n_cols=profile.n_columns,
|
|
283
|
+
n_issues=len(issues),
|
|
284
|
+
n_error=sum(1 for i in issues if i.severity.value == "error"),
|
|
285
|
+
n_warn=sum(1 for i in issues if i.severity.value == "warning"),
|
|
286
|
+
n_info=sum(1 for i in issues if i.severity.value == "info"),
|
|
287
|
+
dup_rows=profile.duplicate_row_count,
|
|
288
|
+
issues=issue_ctx,
|
|
289
|
+
columns=[_column_context(c) for c in profile.columns],
|
|
290
|
+
)
|
|
291
|
+
return _env.from_string(_BASE).render(
|
|
292
|
+
title=f"CleanFrame report — {source}" if source else "CleanFrame report",
|
|
293
|
+
subtitle=source or "data profile",
|
|
294
|
+
css=_CSS,
|
|
295
|
+
version=__version__,
|
|
296
|
+
body=body,
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def render_clean_report(
|
|
301
|
+
diff: CellDiff,
|
|
302
|
+
*,
|
|
303
|
+
quarantine: pd.DataFrame | None = None,
|
|
304
|
+
source: str | None = None,
|
|
305
|
+
quality: QualityScore | None = None,
|
|
306
|
+
max_rows_per_column: int = 12,
|
|
307
|
+
) -> str:
|
|
308
|
+
"""Render the cleaning report (diff, structure changes, quarantine)."""
|
|
309
|
+
changes_by_col = {}
|
|
310
|
+
for col, changes in diff.changes_by_column().items():
|
|
311
|
+
changes_by_col[col] = {
|
|
312
|
+
"total": len(changes),
|
|
313
|
+
"sample": [
|
|
314
|
+
{"row_id": c.row_id, "before": _fmt_cell(c.before), "after": _fmt_cell(c.after)}
|
|
315
|
+
for c in changes[:max_rows_per_column]
|
|
316
|
+
],
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
q_cols: list[str] = []
|
|
320
|
+
q_rows: list[list[str]] = []
|
|
321
|
+
if quarantine is not None and not quarantine.empty:
|
|
322
|
+
q_cols = [str(c) for c in quarantine.columns]
|
|
323
|
+
q_rows = [[_fmt_cell(v) for v in rec] for rec in quarantine.head(50).to_numpy().tolist()]
|
|
324
|
+
|
|
325
|
+
body = _env.from_string(_CLEAN_BODY).render(
|
|
326
|
+
q=quality,
|
|
327
|
+
d=diff.summary(),
|
|
328
|
+
renames=diff.renamed_columns,
|
|
329
|
+
added=diff.added_columns,
|
|
330
|
+
removed=diff.removed_columns,
|
|
331
|
+
changes_by_col=changes_by_col,
|
|
332
|
+
quarantine_n=0 if quarantine is None else int(len(quarantine)),
|
|
333
|
+
quarantine_cols=q_cols,
|
|
334
|
+
quarantine_rows=q_rows,
|
|
335
|
+
)
|
|
336
|
+
return _env.from_string(_BASE).render(
|
|
337
|
+
title=f"CleanFrame diff — {source}" if source else "CleanFrame diff",
|
|
338
|
+
subtitle=source or "cleaning result",
|
|
339
|
+
css=_CSS,
|
|
340
|
+
version=__version__,
|
|
341
|
+
body=body,
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
__all__ = ["render_profile_report", "render_clean_report", "quality_score", "QualityScore"]
|
cleanframe/result.py
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""User-facing result objects returned by the high-level API.
|
|
2
|
+
|
|
3
|
+
:class:`CleanResult` is what ``cf.clean(...)`` hands back — the cleaned dataframe
|
|
4
|
+
plus every durable artifact (recipe, exportable code, diff, quarantine, report) and
|
|
5
|
+
convenience savers so the README's ergonomics work verbatim::
|
|
6
|
+
|
|
7
|
+
result = cf.clean(df, ...)
|
|
8
|
+
result.diff.show()
|
|
9
|
+
result.recipe.save("customer.recipe.yaml")
|
|
10
|
+
result.code.save("clean_customers.py")
|
|
11
|
+
clean_df = result.dataframe
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
import pandas as pd
|
|
20
|
+
|
|
21
|
+
from ._util import write_text
|
|
22
|
+
from .codegen import generate_code
|
|
23
|
+
from .dataio import write_frame
|
|
24
|
+
from .diff import CellDiff
|
|
25
|
+
from .drift import DriftReport
|
|
26
|
+
from .issues import Issues
|
|
27
|
+
from .profile import DataFrameProfile
|
|
28
|
+
from .quality import QualityScore
|
|
29
|
+
from .recipe import Recipe
|
|
30
|
+
from .report import render_clean_report, render_profile_report
|
|
31
|
+
from .validate import ValidationResult
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class CodeArtifact:
|
|
35
|
+
"""Lazily-generated standalone pandas code for a recipe. ``.save()`` writes ``.py``."""
|
|
36
|
+
|
|
37
|
+
def __init__(self, recipe: Recipe, func_name: str = "clean") -> None:
|
|
38
|
+
self._recipe = recipe
|
|
39
|
+
self._func_name = func_name
|
|
40
|
+
self._cached: str | None = None
|
|
41
|
+
|
|
42
|
+
def to_string(self) -> str:
|
|
43
|
+
if self._cached is None:
|
|
44
|
+
self._cached = generate_code(self._recipe, func_name=self._func_name)
|
|
45
|
+
return self._cached
|
|
46
|
+
|
|
47
|
+
def save(self, path: str | Path) -> Path:
|
|
48
|
+
return write_text(path, self.to_string())
|
|
49
|
+
|
|
50
|
+
def __str__(self) -> str:
|
|
51
|
+
return self.to_string()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class Report:
|
|
55
|
+
"""A rendered HTML report. ``.save()`` writes it; displays inline in Jupyter."""
|
|
56
|
+
|
|
57
|
+
def __init__(self, html: str, quality: QualityScore | None = None) -> None:
|
|
58
|
+
self.html = html
|
|
59
|
+
self.quality = quality
|
|
60
|
+
|
|
61
|
+
def save(self, path: str | Path) -> Path:
|
|
62
|
+
return write_text(path, self.html)
|
|
63
|
+
|
|
64
|
+
def _repr_html_(self) -> str: # pragma: no cover - Jupyter hook
|
|
65
|
+
return self.html
|
|
66
|
+
|
|
67
|
+
def __str__(self) -> str:
|
|
68
|
+
return self.html
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass
|
|
72
|
+
class CleanResult:
|
|
73
|
+
"""The full outcome of a clean: cleaned data + every artifact needed to reproduce it."""
|
|
74
|
+
|
|
75
|
+
dataframe: pd.DataFrame
|
|
76
|
+
recipe: Recipe
|
|
77
|
+
diff: CellDiff
|
|
78
|
+
quarantine: pd.DataFrame = field(default_factory=pd.DataFrame)
|
|
79
|
+
issues: Issues = field(default_factory=Issues)
|
|
80
|
+
profile: DataFrameProfile | None = None
|
|
81
|
+
validation_results: list[ValidationResult] = field(default_factory=list)
|
|
82
|
+
quality: QualityScore | None = None
|
|
83
|
+
source: str | None = None
|
|
84
|
+
log: list[str] = field(default_factory=list)
|
|
85
|
+
drift: DriftReport | None = None
|
|
86
|
+
|
|
87
|
+
@property
|
|
88
|
+
def code(self) -> CodeArtifact:
|
|
89
|
+
return CodeArtifact(self.recipe)
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def has_quarantine(self) -> bool:
|
|
93
|
+
return not self.quarantine.empty
|
|
94
|
+
|
|
95
|
+
def show(self, **kwargs) -> None:
|
|
96
|
+
"""Print the cell-level diff, git-diff style."""
|
|
97
|
+
self.diff.show(**kwargs)
|
|
98
|
+
|
|
99
|
+
def report(self, path: str | Path | None = None) -> Report:
|
|
100
|
+
"""Build the HTML cleaning report (diff + quarantine); optionally save it."""
|
|
101
|
+
html = render_clean_report(
|
|
102
|
+
self.diff, quarantine=self.quarantine, source=self.source, quality=self.quality
|
|
103
|
+
)
|
|
104
|
+
report = Report(html, self.quality)
|
|
105
|
+
if path is not None:
|
|
106
|
+
report.save(path)
|
|
107
|
+
return report
|
|
108
|
+
|
|
109
|
+
def summary(self) -> dict:
|
|
110
|
+
s = self.diff.summary()
|
|
111
|
+
s["quarantined"] = int(len(self.quarantine))
|
|
112
|
+
if self.quality is not None:
|
|
113
|
+
s["quality_before"] = self.quality.score
|
|
114
|
+
return s
|
|
115
|
+
|
|
116
|
+
def save_all(self, prefix: str | Path) -> dict[str, Path]:
|
|
117
|
+
"""Save recipe, code, cleaned data, and report next to ``prefix``."""
|
|
118
|
+
prefix = Path(prefix)
|
|
119
|
+
paths = {
|
|
120
|
+
"recipe": self.recipe.save(prefix.with_suffix(".recipe.yaml")),
|
|
121
|
+
"code": self.code.save(prefix.with_suffix(".py")),
|
|
122
|
+
"clean": _save_csv(self.dataframe, prefix.with_suffix(".clean.csv")),
|
|
123
|
+
"report": self.report().save(prefix.with_suffix(".report.html")),
|
|
124
|
+
}
|
|
125
|
+
if self.has_quarantine:
|
|
126
|
+
paths["quarantine"] = write_frame(
|
|
127
|
+
self.quarantine, prefix.with_suffix(".quarantine.csv")
|
|
128
|
+
)
|
|
129
|
+
return paths
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _save_csv(df: pd.DataFrame, path: Path) -> Path:
|
|
133
|
+
return write_frame(df, path)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def build_profile_report_object(
|
|
137
|
+
profile: DataFrameProfile, issues: Issues, *, source: str | None = None,
|
|
138
|
+
quality: QualityScore | None = None,
|
|
139
|
+
) -> Report:
|
|
140
|
+
html = render_profile_report(profile, issues, source=source, quality=quality)
|
|
141
|
+
return Report(html, quality)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
__all__ = ["CleanResult", "Report", "CodeArtifact", "build_profile_report_object"]
|