parity-diff 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
parity/types.py ADDED
@@ -0,0 +1,125 @@
1
+ """Core value types shared across dialects and the diff engine.
2
+
3
+ BUILD_SPEC sketched a ``TableRef`` and a ``Segment`` dataclass. Neither
4
+ survived contact with the implementation: the engine takes a dialect, a
5
+ table name and a key as separate arguments rather than bundling them, and a
6
+ segment is a plain ``(lo, hi)`` tuple whose checksums live in the dict the
7
+ dialect returns. They were carried unused for a while, which is worse than
8
+ not having them - an unused public dataclass invites someone to build on it
9
+ and then drift out of sync with what the code really does.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from dataclasses import dataclass, field
15
+ from enum import Enum
16
+
17
+
18
+ class LogicalType(str, Enum):
19
+ """Engine-independent type categories.
20
+
21
+ Every physical column type a dialect knows about is mapped onto one of
22
+ these, and each one has exactly one canonical text encoding (see
23
+ ``Dialect.normalize``). Two rows are equal iff their canonical encodings
24
+ are byte-identical, so this enum is where cross-engine correctness lives.
25
+ """
26
+
27
+ INTEGER = "integer"
28
+ DECIMAL = "decimal"
29
+ FLOAT = "float"
30
+ BOOLEAN = "boolean"
31
+ STRING = "string"
32
+ DATE = "date"
33
+ TIMESTAMP = "timestamp"
34
+ UNKNOWN = "unknown"
35
+
36
+
37
+ @dataclass(frozen=True)
38
+ class Column:
39
+ name: str
40
+ logical_type: LogicalType
41
+ raw_type: str = ""
42
+
43
+
44
+ @dataclass(frozen=True)
45
+ class KeyStats:
46
+ """What one side reports about its key column, from a single scan.
47
+
48
+ ``rows`` and ``distinct`` ride along with ``min``/``max`` because the
49
+ min/max query already has to visit the key column. Getting uniqueness for
50
+ free matters: a non-unique key silently collapses rows during comparison
51
+ and makes differences vanish, which is the most dangerous failure this
52
+ tool can have.
53
+ """
54
+
55
+ lo: int | None
56
+ hi: int | None
57
+ rows: int
58
+ distinct: int
59
+ #: Rows whose key is not NULL. ``None`` means the dialect did not measure
60
+ #: it, in which case no NULL keys are assumed.
61
+ non_null: int | None = None
62
+
63
+ @property
64
+ def empty(self) -> bool:
65
+ return self.rows == 0
66
+
67
+ @property
68
+ def null_keys(self) -> int:
69
+ return 0 if self.non_null is None else self.rows - self.non_null
70
+
71
+ @property
72
+ def has_null_keys(self) -> bool:
73
+ return self.null_keys > 0
74
+
75
+ @property
76
+ def has_duplicate_keys(self) -> bool:
77
+ # Compare like with like: `count(distinct k)` ignores NULLs, so
78
+ # measuring it against `count(*)` would report every NULL key as a
79
+ # duplicate and send the reader hunting for duplicates that do not
80
+ # exist. NULL keys are diagnosed separately.
81
+ comparable = self.rows if self.non_null is None else self.non_null
82
+ return comparable != self.distinct
83
+
84
+
85
+ @dataclass
86
+ class RowDiff:
87
+ key: int
88
+ kind: str # "only_in_a" | "only_in_b" | "different"
89
+ columns: list[str] = field(default_factory=list)
90
+ values_a: dict[str, str] = field(default_factory=dict)
91
+ values_b: dict[str, str] = field(default_factory=dict)
92
+
93
+
94
+ @dataclass
95
+ class DiffStats:
96
+ queries: int = 0
97
+ segments_checked: int = 0
98
+ rows_downloaded: int = 0
99
+ rows_compared_a: int = 0
100
+ rows_compared_b: int = 0
101
+ seconds: float = 0.0
102
+
103
+
104
+ @dataclass
105
+ class DiffResult:
106
+ diffs: list[RowDiff]
107
+ stats: DiffStats
108
+ columns: list[Column]
109
+ warnings: list[str] = field(default_factory=list)
110
+ #: True when the walk stopped early (``max_diffs``), so ``diffs`` is a
111
+ #: partial answer. CLAUDE.md section 8: never let an approximate result
112
+ #: look exact. Callers must not read ``identical`` as "tables match" when
113
+ #: this is set.
114
+ truncated: bool = False
115
+ #: Decimal places at which DECIMAL/FLOAT columns were compared. Surfaced
116
+ #: in output so a reader knows the comparison was rounded, not exact.
117
+ float_scale: int = 6
118
+
119
+ @property
120
+ def identical(self) -> bool:
121
+ """No differences found *and* the whole key space was walked."""
122
+ return not self.diffs and not self.truncated
123
+
124
+ def by_kind(self, kind: str) -> list[RowDiff]:
125
+ return [d for d in self.diffs if d.kind == kind]
@@ -0,0 +1,268 @@
1
+ Metadata-Version: 2.4
2
+ Name: parity-diff
3
+ Version: 0.1.0
4
+ Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
5
+ Author: Alessio Sorio
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/Aleixiou/parity
8
+ Project-URL: Repository, https://github.com/Aleixiou/parity
9
+ Project-URL: Issues, https://github.com/Aleixiou/parity/issues
10
+ Keywords: data-diff,migration,postgres,duckdb,data-quality,cutover,warehouse,parity
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: System Administrators
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Database
22
+ Classifier: Topic :: Software Development :: Quality Assurance
23
+ Classifier: Topic :: Utilities
24
+ Requires-Python: >=3.10
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ Provides-Extra: duckdb
28
+ Requires-Dist: duckdb>=1.0; extra == "duckdb"
29
+ Provides-Extra: postgres
30
+ Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
31
+ Provides-Extra: all
32
+ Requires-Dist: duckdb>=1.0; extra == "all"
33
+ Requires-Dist: psycopg[binary]>=3.1; extra == "all"
34
+ Dynamic: license-file
35
+
36
+ # parity
37
+
38
+ [![tests](https://github.com/Aleixiou/parity/actions/workflows/tests.yml/badge.svg)](https://github.com/Aleixiou/parity/actions/workflows/tests.yml)
39
+
40
+ **Prove two tables in two different database engines hold the same data —
41
+ without moving the data out of either engine.**
42
+
43
+ Migrations don't fail on translation. They fail at cutover, because nobody can
44
+ prove the new pipeline produces the same data as the old one — so the legacy
45
+ system runs in parallel "just to be safe", forever, at double the cost.
46
+ `parity` is the proof.
47
+
48
+ ```bash
49
+ parity diff \
50
+ --a "postgres://user:pw@legacy-host/warehouse" --a-table public.orders \
51
+ --b "duckdb:///./new.duckdb" --b-table main.orders \
52
+ --key order_id
53
+ ```
54
+
55
+ ```
56
+ ✗ 5 differences in 10,000,000 rows
57
+ 28 queries · 7,628 rows downloaded (0.04% of both tables) · 54.2s
58
+ 1 only in A · 1 only in B · 3 different
59
+
60
+ only in A key 999999999
61
+ only in B key 4000
62
+ different key 13 columns: note
63
+ note A '' B NULL
64
+ different key 6010000 columns: amount
65
+ amount A 700.010000 B 700.000000
66
+ different key 8700000 columns: is_refunded
67
+ is_refunded A NULL B false
68
+
69
+ comparing floats and decimals at 6 decimal places
70
+ ```
71
+
72
+ Exit code `1`. Put it in CI and the build fails until the data agrees.
73
+
74
+ ## How it works
75
+
76
+ `parity` pushes hash aggregation **down into both engines**. It asks each side
77
+ for one checksum per bucket of the key range, compares a handful of integers,
78
+ recurses only into the buckets that disagree, and downloads rows only from
79
+ ranges already proven to differ.
80
+
81
+ On identical tables it downloads **zero rows** and issues **four queries**,
82
+ whether the table has ten thousand rows or ten million.
83
+
84
+ ## Measured
85
+
86
+ 10,000,000 rows per side, PostgreSQL 18.4 ↔ DuckDB 1.5.5, median of five runs
87
+ on one developer laptop (`demo/benchmark.py`):
88
+
89
+ | Scenario | Queries | Rows downloaded | Wall time |
90
+ |---|---|---|---|
91
+ | identical tables | 4 | **0** (0.0000%) | 26.9s |
92
+ | 5 planted differences | 28 | 7,628 (0.0381%) | 54.2s |
93
+
94
+ The query count and the rows-downloaded figures are **exact and
95
+ hardware-independent** — they are properties of the algorithm, and the test
96
+ suite pins them. The wall times are one machine under sustained load; treat
97
+ them as an order of magnitude, not a specification.
98
+
99
+ The five planted differences — a changed decimal, a deleted row, an inserted
100
+ row at key 999,999,999, a `NULL` turned into `''`, and a `FALSE` turned into
101
+ `NULL` — are all found exactly, with no false positives.
102
+
103
+ Cost is roughly **one full hash pass per side**. It is CPU-bound on MD5, not
104
+ IO-bound on key lookup, which is why **an index on the key column makes no
105
+ measurable difference** (measured at 10M: 38.8s without, 37.3s with).
106
+ Raising
107
+ `--bisection-factor` does not speed up the dominant first pass; it only reduces
108
+ round trips on later levels.
109
+
110
+ ## Install
111
+
112
+ ```bash
113
+ pip install "parity-diff[all]" # both engines
114
+ pip install "parity-diff[duckdb]" # DuckDB only — no PostgreSQL driver pulled in
115
+ pip install "parity-diff[postgres]" # PostgreSQL only
116
+ ```
117
+
118
+ > The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
119
+ > empty project. The command you run and the module you import are both
120
+ > `parity`.
121
+
122
+ Python 3.10+. The core has no dependencies; drivers are optional extras and are
123
+ imported lazily, so a DuckDB-only user is never made to install `psycopg`.
124
+
125
+ ## Usage
126
+
127
+ ```
128
+ parity diff --a CONN --a-table TABLE --b CONN --b-table TABLE --key COL
129
+ [--columns a,b,c] [--exclude x,y]
130
+ [--bisection-factor 32] [--threshold 10000] [--float-scale 6]
131
+ [--max-diffs 100] [--json] [--quiet]
132
+ ```
133
+
134
+ **Exit codes** — this is what makes it a CI check:
135
+
136
+ | Code | Meaning |
137
+ |---|---|
138
+ | `0` | identical |
139
+ | `1` | differences found |
140
+ | `2` | error (bad connection string, missing table, unusable key) |
141
+
142
+ An error is never `1`. A CI job can always tell "the tables differ" from "the
143
+ tool could not run".
144
+
145
+ Connection strings:
146
+
147
+ ```
148
+ postgres://user:password@host:port/database (also postgresql://)
149
+ duckdb:///relative/path.duckdb (three slashes = relative)
150
+ duckdb:////var/lib/warehouse.duckdb (four slashes = absolute)
151
+ duckdb:///C:/data/warehouse.duckdb (absolute, Windows)
152
+ duckdb:///:memory:
153
+ ```
154
+
155
+ Table names may be schema-qualified. Unqualified names default to `public` on
156
+ PostgreSQL and `main` on DuckDB.
157
+
158
+ `--json` emits the same content as a machine-readable object, including
159
+ `identical`, `truncated`, per-difference values, and the full stats block.
160
+
161
+ ### In CI
162
+
163
+ ```yaml
164
+ - name: prove the migration is complete
165
+ run: |
166
+ parity diff \
167
+ --a "$LEGACY_URL" --a-table public.orders \
168
+ --b "$WAREHOUSE_URL" --b-table analytics.orders \
169
+ --key order_id --quiet
170
+ ```
171
+
172
+ ## Limitations — read these before trusting a result
173
+
174
+ A parity tool that reports a false match is worse than useless, so these are
175
+ stated plainly rather than buried.
176
+
177
+ - **Integer keys only.** The bisection arithmetic divides the key range. A
178
+ `varchar` or `uuid` key is rejected with a clear message, not guessed at.
179
+ Composite and hashed keys are a planned extension.
180
+ - **Floats and decimals are compared at 6 decimal places** by default. Two
181
+ values differing only in the 7th place are reported as *equal*. This is a
182
+ deliberate cross-engine rounding contract — the two engines do not otherwise
183
+ agree on float text. Change it with `--float-scale`; it always applies to
184
+ both sides, and the scale in force is printed on every run.
185
+ - **Keys must be unique.** A non-unique key is detected up front and rejected.
186
+ Silently collapsing duplicate rows would hide real differences.
187
+ - **`Infinity` and `NaN`** in float columns are compared as those literal
188
+ tokens on both sides. A double whose magnitude reaches 1e32 is not supported
189
+ and fails loudly on DuckDB.
190
+ - **Wide tables are fine.** PostgreSQL caps a function call at 100 arguments,
191
+ so the row concatenation is built as a nested tree; tested to 500 columns.
192
+ - **Supported types:** integer, decimal, float, boolean, string, date,
193
+ timestamp. Anything else is compared as raw text and reported as a warning —
194
+ two engines may render the same JSON or array differently for reasons that
195
+ have nothing to do with the data.
196
+ - **At most 10,000 differences are reported by default.** Each one costs about
197
+ 715 bytes, so two tables that share nothing would need gigabytes rather than
198
+ producing an answer - and pointing the tool at the wrong table or the wrong
199
+ environment is exactly what it exists to catch. Past the limit the run is
200
+ flagged `truncated`, `identical` is never true, and the output says "at
201
+ least N". It does not mean the rest matched. `--max-diffs 0` lifts the limit.
202
+ - **Timezone-aware timestamps are compared as instants, in UTC.** Both
203
+ sessions are pinned to UTC on connect, so the same instant matches whatever
204
+ the two servers' default timezones are. Without that pin, two sides in
205
+ different zones render every `timestamptz` differently and the tool reports
206
+ the entire table as changed.
207
+ - Columns present on only one side are skipped with a warning, not treated as
208
+ differences.
209
+
210
+ ## Supported engines
211
+
212
+ | Engine | Status |
213
+ |---|---|
214
+ | PostgreSQL | supported (tested against 16 and 18) |
215
+ | DuckDB | supported (tested against 1.5) |
216
+ | Snowflake, BigQuery | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
217
+
218
+ ## Scope
219
+
220
+ In scope: proving two tables match, finding exactly which rows and columns
221
+ don't, doing it cheaply on large tables, running in CI.
222
+
223
+ Deliberately out of scope: data quality rules, freshness checks, anomaly
224
+ detection, lineage, cataloguing, orchestration, transformation, a web UI, a
225
+ server, schema migration. If a feature does not help someone answer *"can I
226
+ safely switch off the old system?"*, it does not belong here.
227
+
228
+ ## Development
229
+
230
+ ```bash
231
+ python -m venv .venv
232
+ .venv/Scripts/Activate.ps1 # Windows; source .venv/bin/activate elsewhere
233
+ pip install -e ".[all]" pytest
234
+ pytest
235
+ ```
236
+
237
+ Tests that need PostgreSQL read `PARITY_TEST_PG` (default
238
+ `postgres://parity:parity@127.0.0.1:5432/parity`) and **skip cleanly** when no
239
+ server is reachable — the DuckDB and pure-Python suites still run.
240
+
241
+ A disposable PostgreSQL:
242
+
243
+ ```bash
244
+ docker run -d --name parity-pg -p 55432:5432 \
245
+ -e POSTGRES_USER=parity -e POSTGRES_PASSWORD=parity -e POSTGRES_DB=parity \
246
+ postgres:16-alpine
247
+ export PARITY_TEST_PG="postgres://parity:parity@127.0.0.1:55432/parity"
248
+ ```
249
+
250
+ Reproduce the benchmark:
251
+
252
+ ```bash
253
+ python demo/generate.py --rows 10000000
254
+ python demo/benchmark.py --expect-clean
255
+ python demo/generate.py --rows 10000000 --plant
256
+ python demo/benchmark.py --expect-planted
257
+ ```
258
+
259
+ `CLAUDE.md` holds the verified cross-engine SQL and why each expression is the
260
+ way it is. `BUILD_SPEC.md` is the build plan.
261
+
262
+ ## Changelog
263
+
264
+ See `CHANGELOG.md`.
265
+
266
+ ## License
267
+
268
+ MIT — see `LICENSE`.
@@ -0,0 +1,14 @@
1
+ parity/__init__.py,sha256=9gMr1dCb1XYXshAdj_lgv05BBoh4RON2tm8AlcxyzeE,902
2
+ parity/cli.py,sha256=1A1xHfWD_sDjV7YCArxInV11CTG1W2nA-Gc3cU1aUdc,13454
3
+ parity/engine.py,sha256=k6ryEZy7jP8eXQ735BAW4lkakmHbtVa4iWyuttHCzEk,18901
4
+ parity/types.py,sha256=s1zTT0FS_Hlh3V7XrAbPHD4pCssxt3yXEw9iGPPVTTk,4111
5
+ parity/dialects/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
6
+ parity/dialects/base.py,sha256=tD5GTAGcQJQARA6VpbdUF_tsj_mrwbCr1TlpLXv4AOM,20718
7
+ parity/dialects/duckdb_dialect.py,sha256=Mt9Ucmkdgxagoy_t_yC_y1XOevC4ebDMO4nF0_JR13E,5828
8
+ parity/dialects/postgres_dialect.py,sha256=dq0SHd_eK9AqvyH1YFi2-bV1LD0fvvLCsu3TfGOD1nY,6500
9
+ parity_diff-0.1.0.dist-info/licenses/LICENSE,sha256=J48Z0nBL4u7D3Vrdkh4o4w1kLhPB_3Ke7pq7-e8D3gs,1070
10
+ parity_diff-0.1.0.dist-info/METADATA,sha256=8FmvTeo_nCZQKoCl_adN3wImc3vFXZOjEw59eg0oK2w,10371
11
+ parity_diff-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
12
+ parity_diff-0.1.0.dist-info/entry_points.txt,sha256=5e_vs-8G8VRgoc_VYD-IKHrlelGrk3vwIxMsAFgAdbg,43
13
+ parity_diff-0.1.0.dist-info/top_level.txt,sha256=f4JdYzINvWc3BYfKSKF9JPS87QlPOIxMQRbG1esvjBY,7
14
+ parity_diff-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ parity = parity.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Alessio Sorio
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ parity