dference 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dference/__init__.py +33 -0
- dference/_compare.py +726 -0
- dference/_query.py +492 -0
- dference/_widget.py +251 -0
- dference/py.typed +0 -0
- dference/static/widget.css +952 -0
- dference/static/widget.js +1370 -0
- dference-0.1.0.dist-info/METADATA +283 -0
- dference-0.1.0.dist-info/RECORD +11 -0
- dference-0.1.0.dist-info/WHEEL +4 -0
- dference-0.1.0.dist-info/licenses/LICENSE +15 -0
dference/_widget.py
ADDED
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
"""The :class:`DataFrameDiff` anywidget.
|
|
2
|
+
|
|
3
|
+
State that the frontend needs once (column metadata, counts, names) is synced
|
|
4
|
+
as traitlets. Rows are fetched on demand through anywidget custom messages:
|
|
5
|
+
|
|
6
|
+
* frontend → Python ``{"type": "query", "req": n, ...}``
|
|
7
|
+
→ Python → frontend ``{"type": "page", "req": n, "rows": [...], ...}``
|
|
8
|
+
* frontend → Python ``{"type": "export", "req": n, ...}``
|
|
9
|
+
→ Python → frontend ``{"type": "export", "req": n, "filename": ...}`` + CSV buffer
|
|
10
|
+
|
|
11
|
+
Only ``selected_ids`` changes as the user interacts, so in marimo just the
|
|
12
|
+
cells that read the selection re-run - paging, sorting and filtering do not
|
|
13
|
+
trigger any re-execution.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import logging
|
|
19
|
+
from collections.abc import Mapping
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import TYPE_CHECKING, Any
|
|
22
|
+
|
|
23
|
+
import anywidget
|
|
24
|
+
import traitlets
|
|
25
|
+
|
|
26
|
+
from ._compare import STATUS_ORDER, DiffResult, compare
|
|
27
|
+
from ._query import Query, ViewEngine
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from collections.abc import Iterable, Sequence
|
|
31
|
+
|
|
32
|
+
import polars as pl
|
|
33
|
+
|
|
34
|
+
__all__ = ["DataFrameDiff"]
|
|
35
|
+
|
|
36
|
+
_STATIC = Path(__file__).parent / "static"
|
|
37
|
+
_log = logging.getLogger(__name__)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class DataFrameDiff(anywidget.AnyWidget):
|
|
41
|
+
"""Interactive comparison of two DataFrames.
|
|
42
|
+
|
|
43
|
+
Example:
|
|
44
|
+
>>> import marimo as mo
|
|
45
|
+
>>> from dference import DataFrameDiff
|
|
46
|
+
>>> view = mo.ui.anywidget(DataFrameDiff(crm, erp, key="customer_id")) # doctest: +SKIP
|
|
47
|
+
>>> view.value["selected_ids"] # reactive selection # doctest: +SKIP
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
left: Left frame (polars, pandas, pyarrow, …).
|
|
51
|
+
right: Right frame.
|
|
52
|
+
key: Key column(s); the combination must be unique on each side.
|
|
53
|
+
left_name: Display name of the left side.
|
|
54
|
+
right_name: Display name of the right side.
|
|
55
|
+
left_short: Marker for left values; defaults to the first letter of
|
|
56
|
+
``left_name`` (``"L"`` if both names start with the same letter).
|
|
57
|
+
right_short: Marker for right values, analogous to ``left_short``.
|
|
58
|
+
ignore_columns: Columns to leave out of the comparison.
|
|
59
|
+
strict: Require identical dtypes on both sides; ``False`` aligns
|
|
60
|
+
lossless differences (see :func:`~dference.compare`).
|
|
61
|
+
page_size: Initial rows per page (the user can change it).
|
|
62
|
+
**kwargs: Passed on to :class:`anywidget.AnyWidget`.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
_esm = _STATIC / "widget.js"
|
|
66
|
+
_css = _STATIC / "widget.css"
|
|
67
|
+
|
|
68
|
+
columns = traitlets.List(traitlets.Dict()).tag(sync=True)
|
|
69
|
+
summary = traitlets.Dict().tag(sync=True)
|
|
70
|
+
left_name = traitlets.Unicode("left").tag(sync=True)
|
|
71
|
+
right_name = traitlets.Unicode("right").tag(sync=True)
|
|
72
|
+
left_short = traitlets.Unicode("L").tag(sync=True)
|
|
73
|
+
right_short = traitlets.Unicode("R").tag(sync=True)
|
|
74
|
+
page_size = traitlets.Int(10).tag(sync=True)
|
|
75
|
+
#: Row ids checked in the widget (see :meth:`selected_frame`).
|
|
76
|
+
selected_ids = traitlets.List(traitlets.Int()).tag(sync=True)
|
|
77
|
+
|
|
78
|
+
def __init__(
|
|
79
|
+
self,
|
|
80
|
+
left: Any,
|
|
81
|
+
right: Any,
|
|
82
|
+
key: str | Sequence[str],
|
|
83
|
+
*,
|
|
84
|
+
left_name: str = "left",
|
|
85
|
+
right_name: str = "right",
|
|
86
|
+
left_short: str | None = None,
|
|
87
|
+
right_short: str | None = None,
|
|
88
|
+
ignore_columns: Iterable[str] = (),
|
|
89
|
+
strict: bool = True,
|
|
90
|
+
page_size: int = 10,
|
|
91
|
+
**kwargs: Any,
|
|
92
|
+
) -> None:
|
|
93
|
+
result = compare(
|
|
94
|
+
left,
|
|
95
|
+
right,
|
|
96
|
+
key,
|
|
97
|
+
left_name=left_name,
|
|
98
|
+
right_name=right_name,
|
|
99
|
+
ignore_columns=ignore_columns,
|
|
100
|
+
strict=strict,
|
|
101
|
+
)
|
|
102
|
+
self._init_from_result(result, left_short, right_short, page_size, kwargs)
|
|
103
|
+
|
|
104
|
+
@classmethod
|
|
105
|
+
def from_result(
|
|
106
|
+
cls,
|
|
107
|
+
result: DiffResult,
|
|
108
|
+
*,
|
|
109
|
+
left_short: str | None = None,
|
|
110
|
+
right_short: str | None = None,
|
|
111
|
+
page_size: int = 10,
|
|
112
|
+
**kwargs: Any,
|
|
113
|
+
) -> DataFrameDiff:
|
|
114
|
+
"""Create a widget for an existing :func:`~dference.compare` result."""
|
|
115
|
+
self = cls.__new__(cls)
|
|
116
|
+
self._init_from_result(result, left_short, right_short, page_size, kwargs)
|
|
117
|
+
return self
|
|
118
|
+
|
|
119
|
+
def _init_from_result(
|
|
120
|
+
self,
|
|
121
|
+
result: DiffResult,
|
|
122
|
+
left_short: str | None,
|
|
123
|
+
right_short: str | None,
|
|
124
|
+
page_size: int,
|
|
125
|
+
kwargs: dict[str, Any],
|
|
126
|
+
) -> None:
|
|
127
|
+
self._result = result
|
|
128
|
+
self._engine = ViewEngine(result)
|
|
129
|
+
self._last_query = Query()
|
|
130
|
+
ls, rs = short_names(result.left_name, result.right_name, left_short, right_short)
|
|
131
|
+
super().__init__(
|
|
132
|
+
columns=[_column_payload(c) for c in result.columns],
|
|
133
|
+
summary=_summary_payload(result),
|
|
134
|
+
left_name=result.left_name,
|
|
135
|
+
right_name=result.right_name,
|
|
136
|
+
left_short=ls,
|
|
137
|
+
right_short=rs,
|
|
138
|
+
page_size=page_size,
|
|
139
|
+
**kwargs,
|
|
140
|
+
)
|
|
141
|
+
self.on_msg(self._handle_message)
|
|
142
|
+
|
|
143
|
+
# ---- Python API -------------------------------------------------------
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def result(self) -> DiffResult:
|
|
147
|
+
"""The underlying :class:`~dference.DiffResult`."""
|
|
148
|
+
return self._result
|
|
149
|
+
|
|
150
|
+
def selected_frame(self) -> pl.DataFrame:
|
|
151
|
+
"""Rows checked in the widget, as a wide result frame."""
|
|
152
|
+
return self._result.frame(rows=sorted(set(self.selected_ids)))
|
|
153
|
+
|
|
154
|
+
def view_frame(self) -> pl.DataFrame:
|
|
155
|
+
"""Rows matching the filters currently set in the widget, in view order.
|
|
156
|
+
|
|
157
|
+
Note:
|
|
158
|
+
This reflects the last query the frontend sent; it is not reactive
|
|
159
|
+
in marimo (filters do not re-run cells).
|
|
160
|
+
"""
|
|
161
|
+
return self._engine.frame(self._last_query)
|
|
162
|
+
|
|
163
|
+
# ---- messaging ----------------------------------------------------------
|
|
164
|
+
|
|
165
|
+
def _handle_message(self, _widget: Any, content: Any, _buffers: Any) -> None:
|
|
166
|
+
"""Dispatch a custom message from the frontend; never raises."""
|
|
167
|
+
if not isinstance(content, Mapping):
|
|
168
|
+
return
|
|
169
|
+
req = content.get("req")
|
|
170
|
+
try:
|
|
171
|
+
match content.get("type"):
|
|
172
|
+
case "query":
|
|
173
|
+
query = Query.from_message(content, self._result.columns)
|
|
174
|
+
self._last_query = query
|
|
175
|
+
page = self._engine.page(
|
|
176
|
+
query, content.get("page", 0), content.get("page_size", self.page_size)
|
|
177
|
+
)
|
|
178
|
+
self.send({"type": "page", "req": req, **page})
|
|
179
|
+
case "export":
|
|
180
|
+
query = Query.from_message(content, self._result.columns)
|
|
181
|
+
data = self._engine.export_csv(query)
|
|
182
|
+
self.send(
|
|
183
|
+
{"type": "export", "req": req, "filename": "dataframe-diff.csv"},
|
|
184
|
+
buffers=[data],
|
|
185
|
+
)
|
|
186
|
+
case other:
|
|
187
|
+
_log.debug("Ignoring unknown message type %r", other)
|
|
188
|
+
except Exception as exc:
|
|
189
|
+
_log.exception("dference: failed to handle %r message", content.get("type"))
|
|
190
|
+
self.send({"type": "error", "req": req, "message": f"{type(exc).__name__}: {exc}"})
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
# --------------------------------------------------------------------------- #
|
|
194
|
+
# Payload helpers
|
|
195
|
+
# --------------------------------------------------------------------------- #
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def short_names(
|
|
199
|
+
left_name: str, right_name: str, left_short: str | None, right_short: str | None
|
|
200
|
+
) -> tuple[str, str]:
|
|
201
|
+
"""Side markers: first letter of each name unless overridden.
|
|
202
|
+
|
|
203
|
+
Falls back to ``"L"``/``"R"`` if a name is empty or both would get the same
|
|
204
|
+
letter (e.g. ``"Prod"`` vs. ``"Preview"``).
|
|
205
|
+
|
|
206
|
+
Example:
|
|
207
|
+
>>> short_names("CRM", "ERP", None, None)
|
|
208
|
+
('C', 'E')
|
|
209
|
+
>>> short_names("Prod", "Preview", None, None)
|
|
210
|
+
('L', 'R')
|
|
211
|
+
"""
|
|
212
|
+
|
|
213
|
+
def first(name: str) -> str:
|
|
214
|
+
stripped = name.strip()
|
|
215
|
+
return stripped[0].upper() if stripped else ""
|
|
216
|
+
|
|
217
|
+
ls = left_short if left_short is not None else first(left_name)
|
|
218
|
+
rs = right_short if right_short is not None else first(right_name)
|
|
219
|
+
clash = ls == rs
|
|
220
|
+
if left_short is None and (not ls or clash):
|
|
221
|
+
ls = "L"
|
|
222
|
+
if right_short is None and (not rs or clash):
|
|
223
|
+
rs = "R"
|
|
224
|
+
return ls, rs
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _column_payload(c: Any) -> dict[str, Any]:
|
|
228
|
+
return {
|
|
229
|
+
"name": c.name,
|
|
230
|
+
"kind": c.kind,
|
|
231
|
+
"dtype": str(c.dtype),
|
|
232
|
+
"numeric": c.dtype.is_numeric(),
|
|
233
|
+
"ftype": c.filter_type,
|
|
234
|
+
"mismatches": c.mismatches,
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _summary_payload(result: DiffResult) -> dict[str, Any]:
|
|
239
|
+
s = result.summary
|
|
240
|
+
return {
|
|
241
|
+
**{st.value: getattr(s, st.name.lower()) for st in STATUS_ORDER},
|
|
242
|
+
"total": s.total,
|
|
243
|
+
"found": s.found,
|
|
244
|
+
"left_rows": s.left_rows,
|
|
245
|
+
"right_rows": s.right_rows,
|
|
246
|
+
"keys": list(result.keys),
|
|
247
|
+
"compared": list(result.compared),
|
|
248
|
+
"left_only_cols": [c.name for c in result.columns if c.kind == "left_only"],
|
|
249
|
+
"right_only_cols": [c.name for c in result.columns if c.kind == "right_only"],
|
|
250
|
+
"ignored": list(result.ignored),
|
|
251
|
+
}
|
dference/py.typed
ADDED
|
File without changes
|