parity-diff 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parity/__init__.py +31 -0
- parity/cli.py +361 -0
- parity/dialects/__init__.py +0 -0
- parity/dialects/base.py +487 -0
- parity/dialects/duckdb_dialect.py +132 -0
- parity/dialects/postgres_dialect.py +139 -0
- parity/engine.py +449 -0
- parity/types.py +125 -0
- parity_diff-0.1.0.dist-info/METADATA +268 -0
- parity_diff-0.1.0.dist-info/RECORD +14 -0
- parity_diff-0.1.0.dist-info/WHEEL +5 -0
- parity_diff-0.1.0.dist-info/entry_points.txt +2 -0
- parity_diff-0.1.0.dist-info/licenses/LICENSE +21 -0
- parity_diff-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""PostgreSQL dialect."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from parity.dialects.base import (
|
|
8
|
+
HASH_HEX_CHARS,
|
|
9
|
+
NULL_SENTINEL,
|
|
10
|
+
Dialect,
|
|
11
|
+
sql_literal,
|
|
12
|
+
)
|
|
13
|
+
from parity.types import Column, LogicalType
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class PostgresDialect(Dialect):
|
|
17
|
+
name = "postgres"
|
|
18
|
+
default_schema = "public"
|
|
19
|
+
|
|
20
|
+
def connect(self, connection_string: str) -> None:
|
|
21
|
+
import psycopg
|
|
22
|
+
|
|
23
|
+
# psycopg understands postgres:// and postgresql:// URLs directly.
|
|
24
|
+
self._conn = psycopg.connect(connection_string, autocommit=True)
|
|
25
|
+
# Pin the session to UTC before anything else. `timestamptz` renders
|
|
26
|
+
# through the *session* timezone, so two sides whose sessions differ
|
|
27
|
+
# render the same instant as different text and every row with a
|
|
28
|
+
# timestamptz reports as changed - a false positive that looks exactly
|
|
29
|
+
# like catastrophic data loss. A timestamptz is an instant; comparing
|
|
30
|
+
# instants in UTC is both correct and deterministic. Naive `timestamp`
|
|
31
|
+
# columns carry no zone and are unaffected.
|
|
32
|
+
#
|
|
33
|
+
# Done while autocommit is still on, so it applies to the session
|
|
34
|
+
# rather than to a transaction that later rolls back.
|
|
35
|
+
self._conn.execute("set time zone 'UTC'")
|
|
36
|
+
self._conn.autocommit = False
|
|
37
|
+
# Read-only by construction (CLAUDE.md section 6). Enforced by the
|
|
38
|
+
# server, so no bug in query building can write to a user's database.
|
|
39
|
+
self._conn.read_only = True
|
|
40
|
+
# REPEATABLE READ gives the whole diff one snapshot. Under the default
|
|
41
|
+
# READ COMMITTED every statement sees a fresh snapshot, so a table
|
|
42
|
+
# written to during the walk is a different table at each bisection
|
|
43
|
+
# level - and the tool can then report a difference that never existed
|
|
44
|
+
# at any single point in time, or descend into a range that has since
|
|
45
|
+
# changed. The source side of a migration is live by definition, which
|
|
46
|
+
# makes this the normal case rather than an edge case.
|
|
47
|
+
#
|
|
48
|
+
# The cost is one held snapshot for the duration of the diff, which
|
|
49
|
+
# delays vacuuming dead tuples. At tens of seconds that is the same
|
|
50
|
+
# cost as any analytical query, and far cheaper than an untrustworthy
|
|
51
|
+
# verdict.
|
|
52
|
+
self._conn.isolation_level = psycopg.IsolationLevel.REPEATABLE_READ
|
|
53
|
+
|
|
54
|
+
def cancel(self) -> None:
|
|
55
|
+
# Safe to call from another thread; psycopg opens its own
|
|
56
|
+
# connection to the server to deliver the cancel request.
|
|
57
|
+
self._conn.cancel()
|
|
58
|
+
|
|
59
|
+
def close(self) -> None:
|
|
60
|
+
self._conn.close()
|
|
61
|
+
|
|
62
|
+
def query(self, sql: str) -> list[tuple[Any, ...]]:
|
|
63
|
+
with self._conn.cursor() as cur:
|
|
64
|
+
cur.execute(sql)
|
|
65
|
+
return cur.fetchall()
|
|
66
|
+
|
|
67
|
+
def _exists_but_unreadable(self, schema: str, name: str) -> bool:
|
|
68
|
+
# pg_catalog is world-readable, unlike information_schema, which is
|
|
69
|
+
# filtered to what the current role holds privileges on.
|
|
70
|
+
try:
|
|
71
|
+
rows = self.query(
|
|
72
|
+
"select 1 from pg_catalog.pg_class c "
|
|
73
|
+
"join pg_catalog.pg_namespace n on n.oid = c.relnamespace "
|
|
74
|
+
f"where n.nspname = {sql_literal(schema)} "
|
|
75
|
+
f"and c.relname = {sql_literal(name)} limit 1"
|
|
76
|
+
)
|
|
77
|
+
except Exception: # noqa: BLE001 - diagnosing an error must never replace it
|
|
78
|
+
return False
|
|
79
|
+
return bool(rows)
|
|
80
|
+
|
|
81
|
+
def quote(self, identifier: str) -> str:
|
|
82
|
+
return '"' + identifier.replace('"', '""') + '"'
|
|
83
|
+
|
|
84
|
+
# ----------------------------------------------------------- rendering
|
|
85
|
+
|
|
86
|
+
def normalize(self, column: Column) -> str:
|
|
87
|
+
c = self.quote(column.name)
|
|
88
|
+
t = column.logical_type
|
|
89
|
+
if t is LogicalType.INTEGER:
|
|
90
|
+
expr = f"({c})::text"
|
|
91
|
+
elif t is LogicalType.FLOAT:
|
|
92
|
+
# PostgreSQL renders these as 'Infinity' / '-Infinity' / 'NaN' on
|
|
93
|
+
# its own, but spelling them explicitly keeps the two engines
|
|
94
|
+
# agreeing by construction rather than by coincidence - DuckDB
|
|
95
|
+
# cannot cast them to DECIMAL at all and needs the same tokens.
|
|
96
|
+
# Note NaN must be found by equality, not `c <> c`: PostgreSQL
|
|
97
|
+
# deliberately treats NaN as equal to itself, unlike IEEE 754.
|
|
98
|
+
expr = (
|
|
99
|
+
f"case when {c} = 'Infinity'::float8 then 'Infinity' "
|
|
100
|
+
f"when {c} = '-Infinity'::float8 then '-Infinity' "
|
|
101
|
+
f"when {c} = 'NaN'::float8 then 'NaN' "
|
|
102
|
+
f"else cast(round(({c})::numeric, {self.float_scale}) as text) end"
|
|
103
|
+
)
|
|
104
|
+
elif t is LogicalType.DECIMAL:
|
|
105
|
+
expr = f"cast(round(({c})::numeric, {self.float_scale}) as text)"
|
|
106
|
+
elif t is LogicalType.BOOLEAN:
|
|
107
|
+
# `else` must not swallow NULL. With `case when c then 'true' else
|
|
108
|
+
# 'false' end` a NULL boolean renders as 'false' - identical to a
|
|
109
|
+
# real FALSE - so the coalesce below never fires and NULL-vs-FALSE
|
|
110
|
+
# reports as a match. Both engines agreed on the wrong answer,
|
|
111
|
+
# which is exactly why the encoding tests plant differences.
|
|
112
|
+
expr = f"case when {c} then 'true' when not {c} then 'false' end"
|
|
113
|
+
elif t is LogicalType.DATE:
|
|
114
|
+
expr = f"to_char({c}, 'YYYY-MM-DD')"
|
|
115
|
+
elif t is LogicalType.TIMESTAMP:
|
|
116
|
+
expr = f"to_char({c}, 'YYYY-MM-DD HH24:MI:SS.US')"
|
|
117
|
+
else:
|
|
118
|
+
expr = f"({c})::text"
|
|
119
|
+
return f"coalesce({expr}, '{NULL_SENTINEL}')"
|
|
120
|
+
|
|
121
|
+
def hash_expr(self, text_expr: str) -> str:
|
|
122
|
+
return f"(('x' || substr(md5({text_expr}), 1, {HASH_HEX_CHARS}))::bit(60)::bigint)"
|
|
123
|
+
|
|
124
|
+
def int_div(self, numerator: str, denominator: str) -> str:
|
|
125
|
+
# `div()` is PostgreSQL's exact integer quotient and truncates toward
|
|
126
|
+
# zero, matching Python's `//` for the non-negative operands used here.
|
|
127
|
+
# Plain `/` is right for bigints but silently yields a scaled, rounded
|
|
128
|
+
# result once `wide_int` has promoted an operand to numeric - which is
|
|
129
|
+
# precisely when the bucket boundary has to be exact.
|
|
130
|
+
return f"div(({numerator})::numeric, ({denominator})::numeric)"
|
|
131
|
+
|
|
132
|
+
def wide_int(self, expr: str) -> str:
|
|
133
|
+
# numeric is arbitrary precision: the key offset cannot overflow it.
|
|
134
|
+
return f"(({expr})::numeric)"
|
|
135
|
+
|
|
136
|
+
def sum_wide(self, expr: str) -> str:
|
|
137
|
+
# numeric is arbitrary precision: cannot overflow no matter the row count.
|
|
138
|
+
return f"coalesce(sum(({expr})::numeric), 0)"
|
|
139
|
+
|
parity/engine.py
ADDED
|
@@ -0,0 +1,449 @@
|
|
|
1
|
+
"""Engine-agnostic segmented diff.
|
|
2
|
+
|
|
3
|
+
This module must not import a concrete dialect. It talks to two ``Dialect``
|
|
4
|
+
objects through their contract and knows nothing about SQL - that separation is
|
|
5
|
+
what makes adding Snowflake or BigQuery a single new file.
|
|
6
|
+
|
|
7
|
+
The strategy: split the key range into buckets, ask each side for one checksum
|
|
8
|
+
per bucket (one query per side per level), and recurse only into buckets whose
|
|
9
|
+
checksums disagree. Rows are downloaded solely from ranges already proven to
|
|
10
|
+
differ, and only once those ranges are small.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import time
|
|
16
|
+
from collections.abc import Iterator, Sequence
|
|
17
|
+
from concurrent.futures import Future, ThreadPoolExecutor
|
|
18
|
+
from concurrent.futures import TimeoutError as FuturesTimeout
|
|
19
|
+
from contextlib import contextmanager
|
|
20
|
+
from typing import Any, TypeVar
|
|
21
|
+
|
|
22
|
+
from parity.dialects.base import Dialect, require_matching_scales
|
|
23
|
+
from parity.types import Column, DiffResult, DiffStats, LogicalType, RowDiff
|
|
24
|
+
|
|
25
|
+
#: What a bucket looks like when a side returned no group for it at all.
|
|
26
|
+
#: `group by` only emits non-empty groups, so an absent bucket genuinely holds
|
|
27
|
+
#: zero rows. Two absent buckets match; absent on one side only does not.
|
|
28
|
+
EMPTY = (0, 0)
|
|
29
|
+
|
|
30
|
+
#: DECIMAL and FLOAT render through the same rounded-text encoding, so a column
|
|
31
|
+
#: that is decimal on one side and double on the other still compares correctly.
|
|
32
|
+
#: This is the common migration case and must not raise a warning.
|
|
33
|
+
_NUMERIC_EQUIVALENT = frozenset({LogicalType.DECIMAL, LogicalType.FLOAT})
|
|
34
|
+
|
|
35
|
+
#: Differences retained before the walk stops. Each `RowDiff` costs about
|
|
36
|
+
#: 715 bytes, measured, and the count is linear - so an unbounded walk over
|
|
37
|
+
#: two tables that share nothing needs 8.5 GB at ten million rows, which is
|
|
38
|
+
#: an out-of-memory kill rather than an answer. Pointing the tool at the
|
|
39
|
+
#: wrong table or the wrong environment is exactly the situation a parity
|
|
40
|
+
#: check exists to catch, so it must survive it. Pass `None` for no limit.
|
|
41
|
+
DEFAULT_MAX_DIFFS = 10_000
|
|
42
|
+
|
|
43
|
+
_A = TypeVar("_A")
|
|
44
|
+
_B = TypeVar("_B")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def bucket_bounds(i: int, lo: int, hi: int, n: int) -> tuple[int, int]:
|
|
48
|
+
"""Key range of bucket ``i``, inverting the SQL bucket expression.
|
|
49
|
+
|
|
50
|
+
SQL computes ``bucket = (key - lo) * n / (hi - lo)`` with *truncating*
|
|
51
|
+
integer division. The inverse of that is ceiling division. Getting this
|
|
52
|
+
wrong makes the walker skip key ranges while still reporting a clean
|
|
53
|
+
match - the worst failure this tool can have - so it is isolated here and
|
|
54
|
+
property-tested against the SQL formula in ``tests/test_engine.py``.
|
|
55
|
+
"""
|
|
56
|
+
span = hi - lo
|
|
57
|
+
b_lo = lo + -(-(i * span) // n)
|
|
58
|
+
b_hi = lo + -(-((i + 1) * span) // n)
|
|
59
|
+
return b_lo, b_hi
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
#: How often the main thread wakes while waiting on the two sides.
|
|
63
|
+
_POLL_SECONDS = 0.2
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _gather(fa: Future[_A], fb: Future[_B]) -> tuple[_A, _B]:
|
|
67
|
+
"""Wait for both sides, staying interruptible while doing so.
|
|
68
|
+
|
|
69
|
+
``Future.result()`` with no timeout blocks in a lock acquire that Windows
|
|
70
|
+
will not deliver a KeyboardInterrupt through, so Ctrl-C is not noticed
|
|
71
|
+
until the query returns on its own - measured at 17 seconds into a 40
|
|
72
|
+
second diff. Passing a timeout makes the wait a series of short sleeps the
|
|
73
|
+
interrupt can land between, at a cost of one cheap wakeup every fifth of a
|
|
74
|
+
second against queries that run for tens of seconds.
|
|
75
|
+
"""
|
|
76
|
+
while True:
|
|
77
|
+
try:
|
|
78
|
+
return fa.result(timeout=_POLL_SECONDS), fb.result(timeout=_POLL_SECONDS)
|
|
79
|
+
except FuturesTimeout:
|
|
80
|
+
continue
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@contextmanager
|
|
84
|
+
def _cancel_on_interrupt(a: Dialect, b: Dialect) -> Iterator[None]:
|
|
85
|
+
"""Turn a Ctrl-C into an actual abort of both in-flight queries.
|
|
86
|
+
|
|
87
|
+
Both sides are queried on worker threads, so the interrupt lands on the
|
|
88
|
+
main thread while the workers sit blocked in the database driver. Exiting
|
|
89
|
+
the `ThreadPoolExecutor` context then *waits* for those queries to finish -
|
|
90
|
+
so without this, Ctrl-C on the ten-minute diff someone actually wants to
|
|
91
|
+
abort does nothing for ten minutes.
|
|
92
|
+
|
|
93
|
+
Cancelling makes the workers' queries raise, the threads end, and the pool
|
|
94
|
+
shuts down promptly. The original KeyboardInterrupt is re-raised either way.
|
|
95
|
+
"""
|
|
96
|
+
try:
|
|
97
|
+
yield
|
|
98
|
+
except BaseException:
|
|
99
|
+
for side in (a, b):
|
|
100
|
+
try:
|
|
101
|
+
side.cancel()
|
|
102
|
+
except Exception: # noqa: BLE001, S110
|
|
103
|
+
# A failed cancel must never replace the real exception -
|
|
104
|
+
# the KeyboardInterrupt below is what the caller needs.
|
|
105
|
+
pass
|
|
106
|
+
raise
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _is_tz_aware(column: Column) -> bool:
|
|
110
|
+
"""Whether a column carries a timezone, judged from the engine's own name.
|
|
111
|
+
|
|
112
|
+
Both engines report `timestamp with time zone` (DuckDB in upper case), and
|
|
113
|
+
`map_type` folds it onto TIMESTAMP by prefix - so the logical type cannot
|
|
114
|
+
tell these apart and the raw name is the only signal there is.
|
|
115
|
+
"""
|
|
116
|
+
return "with time zone" in column.raw_type.lower()
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _select_columns(
|
|
120
|
+
cols_a: dict[str, Column],
|
|
121
|
+
cols_b: dict[str, Column],
|
|
122
|
+
key: str,
|
|
123
|
+
columns: Sequence[str] | None,
|
|
124
|
+
exclude: Sequence[str],
|
|
125
|
+
warnings: list[str],
|
|
126
|
+
) -> list[str]:
|
|
127
|
+
"""Decide which columns to compare, explaining anything dropped."""
|
|
128
|
+
both = (set(cols_a) & set(cols_b)) - {key}
|
|
129
|
+
excluded = set(exclude)
|
|
130
|
+
|
|
131
|
+
unknown_exclude = excluded - set(cols_a) - set(cols_b)
|
|
132
|
+
if unknown_exclude:
|
|
133
|
+
warnings.append(
|
|
134
|
+
f"--exclude named columns that exist on neither side: "
|
|
135
|
+
f"{sorted(unknown_exclude)}"
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
shared = sorted(both - excluded)
|
|
139
|
+
if columns:
|
|
140
|
+
requested = list(dict.fromkeys(columns)) # de-duplicate, keep order
|
|
141
|
+
nowhere = [c for c in requested if c not in cols_a and c not in cols_b]
|
|
142
|
+
one_side = [c for c in requested if c not in nowhere and c not in both]
|
|
143
|
+
dropped = [c for c in requested if c in excluded]
|
|
144
|
+
# Order matters: the key is present on both sides but excluded from
|
|
145
|
+
# `both`, so it would otherwise be misreported as one-sided.
|
|
146
|
+
if key in requested:
|
|
147
|
+
raise ValueError(
|
|
148
|
+
f"--columns named the key column {key!r}; the key is how rows "
|
|
149
|
+
f"are matched up, not something compared between them"
|
|
150
|
+
)
|
|
151
|
+
if nowhere:
|
|
152
|
+
raise ValueError(f"--columns named unknown columns: {nowhere}")
|
|
153
|
+
if one_side:
|
|
154
|
+
raise ValueError(
|
|
155
|
+
f"--columns named columns present on only one side: {one_side}"
|
|
156
|
+
)
|
|
157
|
+
if dropped:
|
|
158
|
+
raise ValueError(
|
|
159
|
+
f"--columns and --exclude both name: {dropped}"
|
|
160
|
+
)
|
|
161
|
+
shared = [c for c in shared if c in set(requested)]
|
|
162
|
+
|
|
163
|
+
for side, only in (
|
|
164
|
+
("A", sorted(set(cols_a) - set(cols_b) - {key})),
|
|
165
|
+
("B", sorted(set(cols_b) - set(cols_a) - {key})),
|
|
166
|
+
):
|
|
167
|
+
if only:
|
|
168
|
+
warnings.append(f"not compared, present only on side {side}: {only}")
|
|
169
|
+
|
|
170
|
+
# A column whose logical type differs between sides renders through a
|
|
171
|
+
# different canonical encoding, so every row would report as changed. That
|
|
172
|
+
# looks like a catastrophic data difference but is really a schema
|
|
173
|
+
# difference, so name it explicitly.
|
|
174
|
+
for name in shared:
|
|
175
|
+
ta, tb = cols_a[name].logical_type, cols_b[name].logical_type
|
|
176
|
+
if ta is tb or {ta, tb} <= _NUMERIC_EQUIVALENT:
|
|
177
|
+
continue
|
|
178
|
+
warnings.append(
|
|
179
|
+
f"column {name!r} is {cols_a[name].raw_type or ta.value} on side A "
|
|
180
|
+
f"but {cols_b[name].raw_type or tb.value} on side B; values are "
|
|
181
|
+
f"compared as text and will very likely all differ"
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
# Timezone-awareness is invisible to the logical type - both sides map to
|
|
185
|
+
# TIMESTAMP - but it is a real semantic difference and `timestamptz` to
|
|
186
|
+
# `timestamp` is one of the commonest migration changes there is. Sessions
|
|
187
|
+
# are pinned to UTC, so a migration that stored UTC compares clean. One
|
|
188
|
+
# that stored local wall-clock reports *every* row as different, and
|
|
189
|
+
# without this line there is nothing pointing at which axis to look along.
|
|
190
|
+
for name in shared:
|
|
191
|
+
aware_a = _is_tz_aware(cols_a[name])
|
|
192
|
+
if aware_a is _is_tz_aware(cols_b[name]):
|
|
193
|
+
continue
|
|
194
|
+
aware, naive = ("A", "B") if aware_a else ("B", "A")
|
|
195
|
+
warnings.append(
|
|
196
|
+
f"column {name!r} is timezone-aware on side {aware} but not on "
|
|
197
|
+
f"side {naive}; both are read in UTC, so a migration that stored "
|
|
198
|
+
f"local wall-clock time rather than UTC will show every row as "
|
|
199
|
+
f"different"
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
unknown = sorted(
|
|
203
|
+
{n for n in shared if cols_a[n].logical_type is LogicalType.UNKNOWN}
|
|
204
|
+
| {n for n in shared if cols_b[n].logical_type is LogicalType.UNKNOWN}
|
|
205
|
+
)
|
|
206
|
+
if unknown:
|
|
207
|
+
warnings.append(
|
|
208
|
+
f"unmapped types, compared as raw text (may differ across engines "
|
|
209
|
+
f"for reasons other than the data): {unknown}"
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
if not shared:
|
|
213
|
+
warnings.append(
|
|
214
|
+
"no comparable columns: only the presence of each key is checked, "
|
|
215
|
+
"not row contents"
|
|
216
|
+
)
|
|
217
|
+
return shared
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def diff(
|
|
221
|
+
a: Dialect,
|
|
222
|
+
b: Dialect,
|
|
223
|
+
a_table: str,
|
|
224
|
+
b_table: str,
|
|
225
|
+
key: str,
|
|
226
|
+
columns: Sequence[str] | None = None,
|
|
227
|
+
exclude: Sequence[str] = (),
|
|
228
|
+
bisection_factor: int = 32,
|
|
229
|
+
threshold: int = 10_000,
|
|
230
|
+
max_diffs: int | None = DEFAULT_MAX_DIFFS,
|
|
231
|
+
) -> DiffResult:
|
|
232
|
+
"""Compare ``a_table`` on side ``a`` with ``b_table`` on side ``b``."""
|
|
233
|
+
started = time.perf_counter()
|
|
234
|
+
stats = DiffStats()
|
|
235
|
+
warnings: list[str] = []
|
|
236
|
+
|
|
237
|
+
if bisection_factor < 2:
|
|
238
|
+
raise ValueError(f"bisection_factor must be >= 2, got {bisection_factor}")
|
|
239
|
+
if threshold < 1:
|
|
240
|
+
raise ValueError(f"threshold must be >= 1, got {threshold}")
|
|
241
|
+
# Checked before any query: two sides rounding floats differently would
|
|
242
|
+
# report every float row as changed.
|
|
243
|
+
require_matching_scales(a, b)
|
|
244
|
+
|
|
245
|
+
# Both sides in parallel from here on. One pool for the whole walk - the
|
|
246
|
+
# comparison is almost entirely IO-wait on two independent engines.
|
|
247
|
+
# Order matters: context managers exit in reverse, so `_cancel_on_interrupt`
|
|
248
|
+
# must be the *inner* one. The pool's own exit blocks waiting for its
|
|
249
|
+
# threads, so cancelling has to happen before that, not after.
|
|
250
|
+
with ThreadPoolExecutor(
|
|
251
|
+
max_workers=2, thread_name_prefix="parity"
|
|
252
|
+
) as pool, _cancel_on_interrupt(a, b):
|
|
253
|
+
|
|
254
|
+
def both(
|
|
255
|
+
fn_name: str,
|
|
256
|
+
*args_a: Any,
|
|
257
|
+
_args_b: tuple[Any, ...] | None = None,
|
|
258
|
+
) -> tuple[Any, Any]:
|
|
259
|
+
fa = pool.submit(getattr(a, fn_name), a_table, *args_a)
|
|
260
|
+
fb = pool.submit(getattr(b, fn_name), b_table, *(_args_b or args_a))
|
|
261
|
+
stats.queries += 2
|
|
262
|
+
return _gather(fa, fb)
|
|
263
|
+
|
|
264
|
+
cols_a_list, cols_b_list = both("columns")
|
|
265
|
+
cols_a = {c.name: c for c in cols_a_list}
|
|
266
|
+
cols_b = {c.name: c for c in cols_b_list}
|
|
267
|
+
# Introspection is metadata, not a scan; do not inflate the query count
|
|
268
|
+
# users read as "how much work did this cost".
|
|
269
|
+
stats.queries -= 2
|
|
270
|
+
|
|
271
|
+
for side, table, cols in (("A", a_table, cols_a), ("B", b_table, cols_b)):
|
|
272
|
+
if key not in cols:
|
|
273
|
+
raise ValueError(
|
|
274
|
+
f"[side {side}] key column {key!r} is not in {table}. "
|
|
275
|
+
f"Columns are: {sorted(cols)}"
|
|
276
|
+
)
|
|
277
|
+
col = cols[key]
|
|
278
|
+
if col.logical_type is not LogicalType.INTEGER:
|
|
279
|
+
# The bisection arithmetic divides the key range. A varchar or
|
|
280
|
+
# uuid key would otherwise fail as a cast error mid-walk.
|
|
281
|
+
raise ValueError(
|
|
282
|
+
f"[side {side}] key column {key!r} in {table} is "
|
|
283
|
+
f"{col.raw_type or col.logical_type.value}, not an integer. "
|
|
284
|
+
f"Only integer keys are supported."
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
shared = _select_columns(cols_a, cols_b, key, columns, exclude, warnings)
|
|
288
|
+
a_cols = [cols_a[c] for c in shared]
|
|
289
|
+
b_cols = [cols_b[c] for c in shared]
|
|
290
|
+
|
|
291
|
+
ks_a, ks_b = both("key_stats", key)
|
|
292
|
+
|
|
293
|
+
# A non-unique key is fatal, not a warning: `fetch_range` maps key to
|
|
294
|
+
# row, so duplicates collapse and their differences disappear.
|
|
295
|
+
# Reporting "identical" for a table we could not actually compare is
|
|
296
|
+
# the one outcome this tool must never produce.
|
|
297
|
+
for side, table, ks in (("A", a_table, ks_a), ("B", b_table, ks_b)):
|
|
298
|
+
# NULL keys first: `count(distinct)` ignores NULLs, so checking
|
|
299
|
+
# uniqueness alone would report a NULL key as a duplicate and send
|
|
300
|
+
# the reader hunting for duplicates that do not exist.
|
|
301
|
+
if ks.has_null_keys:
|
|
302
|
+
raise ValueError(
|
|
303
|
+
f"[side {side}] key column {key!r} in {table} contains "
|
|
304
|
+
f"{ks.null_keys:,} NULL value(s). A row with no key cannot "
|
|
305
|
+
f"be matched to anything on the other side."
|
|
306
|
+
)
|
|
307
|
+
if ks.has_duplicate_keys:
|
|
308
|
+
raise ValueError(
|
|
309
|
+
f"[side {side}] key column {key!r} in {table} is not "
|
|
310
|
+
f"unique: {ks.rows:,} rows but only {ks.distinct:,} "
|
|
311
|
+
f"distinct keys. Rows cannot be compared one-to-one."
|
|
312
|
+
)
|
|
313
|
+
stats.rows_compared_a, stats.rows_compared_b = ks_a.rows, ks_b.rows
|
|
314
|
+
|
|
315
|
+
diffs: list[RowDiff] = []
|
|
316
|
+
truncated = False
|
|
317
|
+
#: Differences or key ranges the walk knowingly did not look at. Any
|
|
318
|
+
#: non-zero value means the answer is partial.
|
|
319
|
+
unchecked = 0
|
|
320
|
+
bounds = [v for v in (ks_a.lo, ks_a.hi, ks_b.lo, ks_b.hi) if v is not None]
|
|
321
|
+
|
|
322
|
+
if bounds:
|
|
323
|
+
# Half-open [lo, hi): +1 so the largest key is inside the range.
|
|
324
|
+
lo, hi = min(bounds), max(bounds) + 1
|
|
325
|
+
queue: list[tuple[int, int]] = [(lo, hi)]
|
|
326
|
+
|
|
327
|
+
def limit_reached() -> bool:
|
|
328
|
+
return max_diffs is not None and len(diffs) >= max_diffs
|
|
329
|
+
|
|
330
|
+
while queue:
|
|
331
|
+
if limit_reached():
|
|
332
|
+
unchecked += len(queue)
|
|
333
|
+
break
|
|
334
|
+
|
|
335
|
+
s_lo, s_hi = queue.pop()
|
|
336
|
+
span = s_hi - s_lo
|
|
337
|
+
if span <= 0:
|
|
338
|
+
continue
|
|
339
|
+
stats.segments_checked += 1
|
|
340
|
+
|
|
341
|
+
if span <= 1:
|
|
342
|
+
_compare_rows(
|
|
343
|
+
pool, a, b, a_table, b_table, key,
|
|
344
|
+
a_cols, b_cols, s_lo, s_hi, diffs, stats,
|
|
345
|
+
)
|
|
346
|
+
continue
|
|
347
|
+
|
|
348
|
+
n = min(bisection_factor, span)
|
|
349
|
+
fa = pool.submit(
|
|
350
|
+
a.segment_checksums, a_table, key, a_cols, s_lo, s_hi, n
|
|
351
|
+
)
|
|
352
|
+
fb = pool.submit(
|
|
353
|
+
b.segment_checksums, b_table, key, b_cols, s_lo, s_hi, n
|
|
354
|
+
)
|
|
355
|
+
cs_a, cs_b = _gather(fa, fb)
|
|
356
|
+
stats.queries += 2
|
|
357
|
+
|
|
358
|
+
differing = [
|
|
359
|
+
i for i in range(n)
|
|
360
|
+
if cs_a.get(i, EMPTY) != cs_b.get(i, EMPTY)
|
|
361
|
+
]
|
|
362
|
+
for position, i in enumerate(differing):
|
|
363
|
+
# The limit has to be honoured inside the level too. A
|
|
364
|
+
# single bucket can yield thousands of differences, so
|
|
365
|
+
# checking only between queue pops would blow past
|
|
366
|
+
# max_diffs and still call the walk complete.
|
|
367
|
+
if limit_reached():
|
|
368
|
+
unchecked += len(differing) - position + len(queue)
|
|
369
|
+
break
|
|
370
|
+
va, vb = cs_a.get(i, EMPTY), cs_b.get(i, EMPTY)
|
|
371
|
+
b_lo_i, b_hi_i = bucket_bounds(i, s_lo, s_hi, n)
|
|
372
|
+
if max(va[0], vb[0]) <= threshold or b_hi_i - b_lo_i <= 1:
|
|
373
|
+
_compare_rows(
|
|
374
|
+
pool, a, b, a_table, b_table, key,
|
|
375
|
+
a_cols, b_cols, b_lo_i, b_hi_i, diffs, stats,
|
|
376
|
+
)
|
|
377
|
+
else:
|
|
378
|
+
queue.append((b_lo_i, b_hi_i))
|
|
379
|
+
|
|
380
|
+
diffs.sort(key=lambda d: d.key)
|
|
381
|
+
if max_diffs is not None and len(diffs) > max_diffs:
|
|
382
|
+
# A single bucket download can overshoot the limit by a lot. Report the
|
|
383
|
+
# first `max_diffs` in key order and say the rest were not listed.
|
|
384
|
+
unchecked += len(diffs) - max_diffs
|
|
385
|
+
del diffs[max_diffs:]
|
|
386
|
+
if unchecked:
|
|
387
|
+
# Partial answers get a flag on the result, not merely a warning
|
|
388
|
+
# string, so no caller can mistake one for a clean comparison.
|
|
389
|
+
truncated = True
|
|
390
|
+
warnings.append(
|
|
391
|
+
f"stopped at the --max-diffs limit of {max_diffs}; "
|
|
392
|
+
f"{unchecked} further difference(s) or key range(s) were not "
|
|
393
|
+
f"reported, so this is a partial answer"
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
stats.seconds = time.perf_counter() - started
|
|
397
|
+
return DiffResult(
|
|
398
|
+
diffs, stats, a_cols, warnings,
|
|
399
|
+
truncated=truncated, float_scale=a.float_scale,
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _compare_rows(
|
|
404
|
+
pool: ThreadPoolExecutor,
|
|
405
|
+
a: Dialect,
|
|
406
|
+
b: Dialect,
|
|
407
|
+
a_table: str,
|
|
408
|
+
b_table: str,
|
|
409
|
+
key: str,
|
|
410
|
+
a_cols: list[Column],
|
|
411
|
+
b_cols: list[Column],
|
|
412
|
+
lo: int,
|
|
413
|
+
hi: int,
|
|
414
|
+
diffs: list[RowDiff],
|
|
415
|
+
stats: DiffStats,
|
|
416
|
+
) -> None:
|
|
417
|
+
"""Download a proven-different range from both sides and diff it locally."""
|
|
418
|
+
fa = pool.submit(a.fetch_range, a_table, key, a_cols, lo, hi)
|
|
419
|
+
fb = pool.submit(b.fetch_range, b_table, key, b_cols, lo, hi)
|
|
420
|
+
rows_a, rows_b = _gather(fa, fb)
|
|
421
|
+
stats.queries += 2
|
|
422
|
+
stats.rows_downloaded += len(rows_a) + len(rows_b)
|
|
423
|
+
|
|
424
|
+
names = [c.name for c in a_cols]
|
|
425
|
+
# `strict=True` on every zip below is load-bearing, not tidiness. Both sides
|
|
426
|
+
# render the same column list in the same order, so the tuples must be the
|
|
427
|
+
# same length as `names`. If that invariant ever broke, a plain zip would
|
|
428
|
+
# silently truncate and simply not report the trailing columns - a
|
|
429
|
+
# difference the tool found and then dropped, which is the one outcome it
|
|
430
|
+
# must never produce. Better to raise.
|
|
431
|
+
for k in sorted(set(rows_a) | set(rows_b)):
|
|
432
|
+
ra, rb = rows_a.get(k), rows_b.get(k)
|
|
433
|
+
if ra is None:
|
|
434
|
+
diffs.append(
|
|
435
|
+
RowDiff(k, "only_in_b", names, {}, dict(zip(names, rb or (), strict=True)))
|
|
436
|
+
)
|
|
437
|
+
elif rb is None:
|
|
438
|
+
diffs.append(
|
|
439
|
+
RowDiff(k, "only_in_a", names, dict(zip(names, ra, strict=True)), {})
|
|
440
|
+
)
|
|
441
|
+
elif ra != rb:
|
|
442
|
+
changed = [n for n, x, y in zip(names, ra, rb, strict=True) if x != y]
|
|
443
|
+
diffs.append(
|
|
444
|
+
RowDiff(
|
|
445
|
+
k, "different", changed,
|
|
446
|
+
{n: v for n, v in zip(names, ra, strict=True) if n in changed},
|
|
447
|
+
{n: v for n, v in zip(names, rb, strict=True) if n in changed},
|
|
448
|
+
)
|
|
449
|
+
)
|