parity-diff 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,139 @@
1
+ """PostgreSQL dialect."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from parity.dialects.base import (
8
+ HASH_HEX_CHARS,
9
+ NULL_SENTINEL,
10
+ Dialect,
11
+ sql_literal,
12
+ )
13
+ from parity.types import Column, LogicalType
14
+
15
+
16
+ class PostgresDialect(Dialect):
17
+ name = "postgres"
18
+ default_schema = "public"
19
+
20
+ def connect(self, connection_string: str) -> None:
21
+ import psycopg
22
+
23
+ # psycopg understands postgres:// and postgresql:// URLs directly.
24
+ self._conn = psycopg.connect(connection_string, autocommit=True)
25
+ # Pin the session to UTC before anything else. `timestamptz` renders
26
+ # through the *session* timezone, so two sides whose sessions differ
27
+ # render the same instant as different text and every row with a
28
+ # timestamptz reports as changed - a false positive that looks exactly
29
+ # like catastrophic data loss. A timestamptz is an instant; comparing
30
+ # instants in UTC is both correct and deterministic. Naive `timestamp`
31
+ # columns carry no zone and are unaffected.
32
+ #
33
+ # Done while autocommit is still on, so it applies to the session
34
+ # rather than to a transaction that later rolls back.
35
+ self._conn.execute("set time zone 'UTC'")
36
+ self._conn.autocommit = False
37
+ # Read-only by construction (CLAUDE.md section 6). Enforced by the
38
+ # server, so no bug in query building can write to a user's database.
39
+ self._conn.read_only = True
40
+ # REPEATABLE READ gives the whole diff one snapshot. Under the default
41
+ # READ COMMITTED every statement sees a fresh snapshot, so a table
42
+ # written to during the walk is a different table at each bisection
43
+ # level - and the tool can then report a difference that never existed
44
+ # at any single point in time, or descend into a range that has since
45
+ # changed. The source side of a migration is live by definition, which
46
+ # makes this the normal case rather than an edge case.
47
+ #
48
+ # The cost is one held snapshot for the duration of the diff, which
49
+ # delays vacuuming dead tuples. At tens of seconds that is the same
50
+ # cost as any analytical query, and far cheaper than an untrustworthy
51
+ # verdict.
52
+ self._conn.isolation_level = psycopg.IsolationLevel.REPEATABLE_READ
53
+
54
+ def cancel(self) -> None:
55
+ # Safe to call from another thread; psycopg opens its own
56
+ # connection to the server to deliver the cancel request.
57
+ self._conn.cancel()
58
+
59
+ def close(self) -> None:
60
+ self._conn.close()
61
+
62
+ def query(self, sql: str) -> list[tuple[Any, ...]]:
63
+ with self._conn.cursor() as cur:
64
+ cur.execute(sql)
65
+ return cur.fetchall()
66
+
67
+ def _exists_but_unreadable(self, schema: str, name: str) -> bool:
68
+ # pg_catalog is world-readable, unlike information_schema, which is
69
+ # filtered to what the current role holds privileges on.
70
+ try:
71
+ rows = self.query(
72
+ "select 1 from pg_catalog.pg_class c "
73
+ "join pg_catalog.pg_namespace n on n.oid = c.relnamespace "
74
+ f"where n.nspname = {sql_literal(schema)} "
75
+ f"and c.relname = {sql_literal(name)} limit 1"
76
+ )
77
+ except Exception: # noqa: BLE001 - diagnosing an error must never replace it
78
+ return False
79
+ return bool(rows)
80
+
81
+ def quote(self, identifier: str) -> str:
82
+ return '"' + identifier.replace('"', '""') + '"'
83
+
84
+ # ----------------------------------------------------------- rendering
85
+
86
+ def normalize(self, column: Column) -> str:
87
+ c = self.quote(column.name)
88
+ t = column.logical_type
89
+ if t is LogicalType.INTEGER:
90
+ expr = f"({c})::text"
91
+ elif t is LogicalType.FLOAT:
92
+ # PostgreSQL renders these as 'Infinity' / '-Infinity' / 'NaN' on
93
+ # its own, but spelling them explicitly keeps the two engines
94
+ # agreeing by construction rather than by coincidence - DuckDB
95
+ # cannot cast them to DECIMAL at all and needs the same tokens.
96
+ # Note NaN must be found by equality, not `c <> c`: PostgreSQL
97
+ # deliberately treats NaN as equal to itself, unlike IEEE 754.
98
+ expr = (
99
+ f"case when {c} = 'Infinity'::float8 then 'Infinity' "
100
+ f"when {c} = '-Infinity'::float8 then '-Infinity' "
101
+ f"when {c} = 'NaN'::float8 then 'NaN' "
102
+ f"else cast(round(({c})::numeric, {self.float_scale}) as text) end"
103
+ )
104
+ elif t is LogicalType.DECIMAL:
105
+ expr = f"cast(round(({c})::numeric, {self.float_scale}) as text)"
106
+ elif t is LogicalType.BOOLEAN:
107
+ # `else` must not swallow NULL. With `case when c then 'true' else
108
+ # 'false' end` a NULL boolean renders as 'false' - identical to a
109
+ # real FALSE - so the coalesce below never fires and NULL-vs-FALSE
110
+ # reports as a match. Both engines agreed on the wrong answer,
111
+ # which is exactly why the encoding tests plant differences.
112
+ expr = f"case when {c} then 'true' when not {c} then 'false' end"
113
+ elif t is LogicalType.DATE:
114
+ expr = f"to_char({c}, 'YYYY-MM-DD')"
115
+ elif t is LogicalType.TIMESTAMP:
116
+ expr = f"to_char({c}, 'YYYY-MM-DD HH24:MI:SS.US')"
117
+ else:
118
+ expr = f"({c})::text"
119
+ return f"coalesce({expr}, '{NULL_SENTINEL}')"
120
+
121
+ def hash_expr(self, text_expr: str) -> str:
122
+ return f"(('x' || substr(md5({text_expr}), 1, {HASH_HEX_CHARS}))::bit(60)::bigint)"
123
+
124
+ def int_div(self, numerator: str, denominator: str) -> str:
125
+ # `div()` is PostgreSQL's exact integer quotient and truncates toward
126
+ # zero, matching Python's `//` for the non-negative operands used here.
127
+ # Plain `/` is right for bigints but silently yields a scaled, rounded
128
+ # result once `wide_int` has promoted an operand to numeric - which is
129
+ # precisely when the bucket boundary has to be exact.
130
+ return f"div(({numerator})::numeric, ({denominator})::numeric)"
131
+
132
+ def wide_int(self, expr: str) -> str:
133
+ # numeric is arbitrary precision: the key offset cannot overflow it.
134
+ return f"(({expr})::numeric)"
135
+
136
+ def sum_wide(self, expr: str) -> str:
137
+ # numeric is arbitrary precision: cannot overflow no matter the row count.
138
+ return f"coalesce(sum(({expr})::numeric), 0)"
139
+
parity/engine.py ADDED
@@ -0,0 +1,449 @@
1
+ """Engine-agnostic segmented diff.
2
+
3
+ This module must not import a concrete dialect. It talks to two ``Dialect``
4
+ objects through their contract and knows nothing about SQL - that separation is
5
+ what makes adding Snowflake or BigQuery a single new file.
6
+
7
+ The strategy: split the key range into buckets, ask each side for one checksum
8
+ per bucket (one query per side per level), and recurse only into buckets whose
9
+ checksums disagree. Rows are downloaded solely from ranges already proven to
10
+ differ, and only once those ranges are small.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import time
16
+ from collections.abc import Iterator, Sequence
17
+ from concurrent.futures import Future, ThreadPoolExecutor
18
+ from concurrent.futures import TimeoutError as FuturesTimeout
19
+ from contextlib import contextmanager
20
+ from typing import Any, TypeVar
21
+
22
+ from parity.dialects.base import Dialect, require_matching_scales
23
+ from parity.types import Column, DiffResult, DiffStats, LogicalType, RowDiff
24
+
25
+ #: What a bucket looks like when a side returned no group for it at all.
26
+ #: `group by` only emits non-empty groups, so an absent bucket genuinely holds
27
+ #: zero rows. Two absent buckets match; absent on one side only does not.
28
+ EMPTY = (0, 0)
29
+
30
+ #: DECIMAL and FLOAT render through the same rounded-text encoding, so a column
31
+ #: that is decimal on one side and double on the other still compares correctly.
32
+ #: This is the common migration case and must not raise a warning.
33
+ _NUMERIC_EQUIVALENT = frozenset({LogicalType.DECIMAL, LogicalType.FLOAT})
34
+
35
+ #: Differences retained before the walk stops. Each `RowDiff` costs about
36
+ #: 715 bytes, measured, and the count is linear - so an unbounded walk over
37
+ #: two tables that share nothing needs 8.5 GB at ten million rows, which is
38
+ #: an out-of-memory kill rather than an answer. Pointing the tool at the
39
+ #: wrong table or the wrong environment is exactly the situation a parity
40
+ #: check exists to catch, so it must survive it. Pass `None` for no limit.
41
+ DEFAULT_MAX_DIFFS = 10_000
42
+
43
+ _A = TypeVar("_A")
44
+ _B = TypeVar("_B")
45
+
46
+
47
+ def bucket_bounds(i: int, lo: int, hi: int, n: int) -> tuple[int, int]:
48
+ """Key range of bucket ``i``, inverting the SQL bucket expression.
49
+
50
+ SQL computes ``bucket = (key - lo) * n / (hi - lo)`` with *truncating*
51
+ integer division. The inverse of that is ceiling division. Getting this
52
+ wrong makes the walker skip key ranges while still reporting a clean
53
+ match - the worst failure this tool can have - so it is isolated here and
54
+ property-tested against the SQL formula in ``tests/test_engine.py``.
55
+ """
56
+ span = hi - lo
57
+ b_lo = lo + -(-(i * span) // n)
58
+ b_hi = lo + -(-((i + 1) * span) // n)
59
+ return b_lo, b_hi
60
+
61
+
62
+ #: How often the main thread wakes while waiting on the two sides.
63
+ _POLL_SECONDS = 0.2
64
+
65
+
66
+ def _gather(fa: Future[_A], fb: Future[_B]) -> tuple[_A, _B]:
67
+ """Wait for both sides, staying interruptible while doing so.
68
+
69
+ ``Future.result()`` with no timeout blocks in a lock acquire that Windows
70
+ will not deliver a KeyboardInterrupt through, so Ctrl-C is not noticed
71
+ until the query returns on its own - measured at 17 seconds into a 40
72
+ second diff. Passing a timeout makes the wait a series of short sleeps the
73
+ interrupt can land between, at a cost of one cheap wakeup every fifth of a
74
+ second against queries that run for tens of seconds.
75
+ """
76
+ while True:
77
+ try:
78
+ return fa.result(timeout=_POLL_SECONDS), fb.result(timeout=_POLL_SECONDS)
79
+ except FuturesTimeout:
80
+ continue
81
+
82
+
83
+ @contextmanager
84
+ def _cancel_on_interrupt(a: Dialect, b: Dialect) -> Iterator[None]:
85
+ """Turn a Ctrl-C into an actual abort of both in-flight queries.
86
+
87
+ Both sides are queried on worker threads, so the interrupt lands on the
88
+ main thread while the workers sit blocked in the database driver. Exiting
89
+ the `ThreadPoolExecutor` context then *waits* for those queries to finish -
90
+ so without this, Ctrl-C on the ten-minute diff someone actually wants to
91
+ abort does nothing for ten minutes.
92
+
93
+ Cancelling makes the workers' queries raise, the threads end, and the pool
94
+ shuts down promptly. The original KeyboardInterrupt is re-raised either way.
95
+ """
96
+ try:
97
+ yield
98
+ except BaseException:
99
+ for side in (a, b):
100
+ try:
101
+ side.cancel()
102
+ except Exception: # noqa: BLE001, S110
103
+ # A failed cancel must never replace the real exception -
104
+ # the KeyboardInterrupt below is what the caller needs.
105
+ pass
106
+ raise
107
+
108
+
109
+ def _is_tz_aware(column: Column) -> bool:
110
+ """Whether a column carries a timezone, judged from the engine's own name.
111
+
112
+ Both engines report `timestamp with time zone` (DuckDB in upper case), and
113
+ `map_type` folds it onto TIMESTAMP by prefix - so the logical type cannot
114
+ tell these apart and the raw name is the only signal there is.
115
+ """
116
+ return "with time zone" in column.raw_type.lower()
117
+
118
+
119
+ def _select_columns(
120
+ cols_a: dict[str, Column],
121
+ cols_b: dict[str, Column],
122
+ key: str,
123
+ columns: Sequence[str] | None,
124
+ exclude: Sequence[str],
125
+ warnings: list[str],
126
+ ) -> list[str]:
127
+ """Decide which columns to compare, explaining anything dropped."""
128
+ both = (set(cols_a) & set(cols_b)) - {key}
129
+ excluded = set(exclude)
130
+
131
+ unknown_exclude = excluded - set(cols_a) - set(cols_b)
132
+ if unknown_exclude:
133
+ warnings.append(
134
+ f"--exclude named columns that exist on neither side: "
135
+ f"{sorted(unknown_exclude)}"
136
+ )
137
+
138
+ shared = sorted(both - excluded)
139
+ if columns:
140
+ requested = list(dict.fromkeys(columns)) # de-duplicate, keep order
141
+ nowhere = [c for c in requested if c not in cols_a and c not in cols_b]
142
+ one_side = [c for c in requested if c not in nowhere and c not in both]
143
+ dropped = [c for c in requested if c in excluded]
144
+ # Order matters: the key is present on both sides but excluded from
145
+ # `both`, so it would otherwise be misreported as one-sided.
146
+ if key in requested:
147
+ raise ValueError(
148
+ f"--columns named the key column {key!r}; the key is how rows "
149
+ f"are matched up, not something compared between them"
150
+ )
151
+ if nowhere:
152
+ raise ValueError(f"--columns named unknown columns: {nowhere}")
153
+ if one_side:
154
+ raise ValueError(
155
+ f"--columns named columns present on only one side: {one_side}"
156
+ )
157
+ if dropped:
158
+ raise ValueError(
159
+ f"--columns and --exclude both name: {dropped}"
160
+ )
161
+ shared = [c for c in shared if c in set(requested)]
162
+
163
+ for side, only in (
164
+ ("A", sorted(set(cols_a) - set(cols_b) - {key})),
165
+ ("B", sorted(set(cols_b) - set(cols_a) - {key})),
166
+ ):
167
+ if only:
168
+ warnings.append(f"not compared, present only on side {side}: {only}")
169
+
170
+ # A column whose logical type differs between sides renders through a
171
+ # different canonical encoding, so every row would report as changed. That
172
+ # looks like a catastrophic data difference but is really a schema
173
+ # difference, so name it explicitly.
174
+ for name in shared:
175
+ ta, tb = cols_a[name].logical_type, cols_b[name].logical_type
176
+ if ta is tb or {ta, tb} <= _NUMERIC_EQUIVALENT:
177
+ continue
178
+ warnings.append(
179
+ f"column {name!r} is {cols_a[name].raw_type or ta.value} on side A "
180
+ f"but {cols_b[name].raw_type or tb.value} on side B; values are "
181
+ f"compared as text and will very likely all differ"
182
+ )
183
+
184
+ # Timezone-awareness is invisible to the logical type - both sides map to
185
+ # TIMESTAMP - but it is a real semantic difference and `timestamptz` to
186
+ # `timestamp` is one of the commonest migration changes there is. Sessions
187
+ # are pinned to UTC, so a migration that stored UTC compares clean. One
188
+ # that stored local wall-clock reports *every* row as different, and
189
+ # without this line there is nothing pointing at which axis to look along.
190
+ for name in shared:
191
+ aware_a = _is_tz_aware(cols_a[name])
192
+ if aware_a is _is_tz_aware(cols_b[name]):
193
+ continue
194
+ aware, naive = ("A", "B") if aware_a else ("B", "A")
195
+ warnings.append(
196
+ f"column {name!r} is timezone-aware on side {aware} but not on "
197
+ f"side {naive}; both are read in UTC, so a migration that stored "
198
+ f"local wall-clock time rather than UTC will show every row as "
199
+ f"different"
200
+ )
201
+
202
+ unknown = sorted(
203
+ {n for n in shared if cols_a[n].logical_type is LogicalType.UNKNOWN}
204
+ | {n for n in shared if cols_b[n].logical_type is LogicalType.UNKNOWN}
205
+ )
206
+ if unknown:
207
+ warnings.append(
208
+ f"unmapped types, compared as raw text (may differ across engines "
209
+ f"for reasons other than the data): {unknown}"
210
+ )
211
+
212
+ if not shared:
213
+ warnings.append(
214
+ "no comparable columns: only the presence of each key is checked, "
215
+ "not row contents"
216
+ )
217
+ return shared
218
+
219
+
220
+ def diff(
221
+ a: Dialect,
222
+ b: Dialect,
223
+ a_table: str,
224
+ b_table: str,
225
+ key: str,
226
+ columns: Sequence[str] | None = None,
227
+ exclude: Sequence[str] = (),
228
+ bisection_factor: int = 32,
229
+ threshold: int = 10_000,
230
+ max_diffs: int | None = DEFAULT_MAX_DIFFS,
231
+ ) -> DiffResult:
232
+ """Compare ``a_table`` on side ``a`` with ``b_table`` on side ``b``."""
233
+ started = time.perf_counter()
234
+ stats = DiffStats()
235
+ warnings: list[str] = []
236
+
237
+ if bisection_factor < 2:
238
+ raise ValueError(f"bisection_factor must be >= 2, got {bisection_factor}")
239
+ if threshold < 1:
240
+ raise ValueError(f"threshold must be >= 1, got {threshold}")
241
+ # Checked before any query: two sides rounding floats differently would
242
+ # report every float row as changed.
243
+ require_matching_scales(a, b)
244
+
245
+ # Both sides in parallel from here on. One pool for the whole walk - the
246
+ # comparison is almost entirely IO-wait on two independent engines.
247
+ # Order matters: context managers exit in reverse, so `_cancel_on_interrupt`
248
+ # must be the *inner* one. The pool's own exit blocks waiting for its
249
+ # threads, so cancelling has to happen before that, not after.
250
+ with ThreadPoolExecutor(
251
+ max_workers=2, thread_name_prefix="parity"
252
+ ) as pool, _cancel_on_interrupt(a, b):
253
+
254
+ def both(
255
+ fn_name: str,
256
+ *args_a: Any,
257
+ _args_b: tuple[Any, ...] | None = None,
258
+ ) -> tuple[Any, Any]:
259
+ fa = pool.submit(getattr(a, fn_name), a_table, *args_a)
260
+ fb = pool.submit(getattr(b, fn_name), b_table, *(_args_b or args_a))
261
+ stats.queries += 2
262
+ return _gather(fa, fb)
263
+
264
+ cols_a_list, cols_b_list = both("columns")
265
+ cols_a = {c.name: c for c in cols_a_list}
266
+ cols_b = {c.name: c for c in cols_b_list}
267
+ # Introspection is metadata, not a scan; do not inflate the query count
268
+ # users read as "how much work did this cost".
269
+ stats.queries -= 2
270
+
271
+ for side, table, cols in (("A", a_table, cols_a), ("B", b_table, cols_b)):
272
+ if key not in cols:
273
+ raise ValueError(
274
+ f"[side {side}] key column {key!r} is not in {table}. "
275
+ f"Columns are: {sorted(cols)}"
276
+ )
277
+ col = cols[key]
278
+ if col.logical_type is not LogicalType.INTEGER:
279
+ # The bisection arithmetic divides the key range. A varchar or
280
+ # uuid key would otherwise fail as a cast error mid-walk.
281
+ raise ValueError(
282
+ f"[side {side}] key column {key!r} in {table} is "
283
+ f"{col.raw_type or col.logical_type.value}, not an integer. "
284
+ f"Only integer keys are supported."
285
+ )
286
+
287
+ shared = _select_columns(cols_a, cols_b, key, columns, exclude, warnings)
288
+ a_cols = [cols_a[c] for c in shared]
289
+ b_cols = [cols_b[c] for c in shared]
290
+
291
+ ks_a, ks_b = both("key_stats", key)
292
+
293
+ # A non-unique key is fatal, not a warning: `fetch_range` maps key to
294
+ # row, so duplicates collapse and their differences disappear.
295
+ # Reporting "identical" for a table we could not actually compare is
296
+ # the one outcome this tool must never produce.
297
+ for side, table, ks in (("A", a_table, ks_a), ("B", b_table, ks_b)):
298
+ # NULL keys first: `count(distinct)` ignores NULLs, so checking
299
+ # uniqueness alone would report a NULL key as a duplicate and send
300
+ # the reader hunting for duplicates that do not exist.
301
+ if ks.has_null_keys:
302
+ raise ValueError(
303
+ f"[side {side}] key column {key!r} in {table} contains "
304
+ f"{ks.null_keys:,} NULL value(s). A row with no key cannot "
305
+ f"be matched to anything on the other side."
306
+ )
307
+ if ks.has_duplicate_keys:
308
+ raise ValueError(
309
+ f"[side {side}] key column {key!r} in {table} is not "
310
+ f"unique: {ks.rows:,} rows but only {ks.distinct:,} "
311
+ f"distinct keys. Rows cannot be compared one-to-one."
312
+ )
313
+ stats.rows_compared_a, stats.rows_compared_b = ks_a.rows, ks_b.rows
314
+
315
+ diffs: list[RowDiff] = []
316
+ truncated = False
317
+ #: Differences or key ranges the walk knowingly did not look at. Any
318
+ #: non-zero value means the answer is partial.
319
+ unchecked = 0
320
+ bounds = [v for v in (ks_a.lo, ks_a.hi, ks_b.lo, ks_b.hi) if v is not None]
321
+
322
+ if bounds:
323
+ # Half-open [lo, hi): +1 so the largest key is inside the range.
324
+ lo, hi = min(bounds), max(bounds) + 1
325
+ queue: list[tuple[int, int]] = [(lo, hi)]
326
+
327
+ def limit_reached() -> bool:
328
+ return max_diffs is not None and len(diffs) >= max_diffs
329
+
330
+ while queue:
331
+ if limit_reached():
332
+ unchecked += len(queue)
333
+ break
334
+
335
+ s_lo, s_hi = queue.pop()
336
+ span = s_hi - s_lo
337
+ if span <= 0:
338
+ continue
339
+ stats.segments_checked += 1
340
+
341
+ if span <= 1:
342
+ _compare_rows(
343
+ pool, a, b, a_table, b_table, key,
344
+ a_cols, b_cols, s_lo, s_hi, diffs, stats,
345
+ )
346
+ continue
347
+
348
+ n = min(bisection_factor, span)
349
+ fa = pool.submit(
350
+ a.segment_checksums, a_table, key, a_cols, s_lo, s_hi, n
351
+ )
352
+ fb = pool.submit(
353
+ b.segment_checksums, b_table, key, b_cols, s_lo, s_hi, n
354
+ )
355
+ cs_a, cs_b = _gather(fa, fb)
356
+ stats.queries += 2
357
+
358
+ differing = [
359
+ i for i in range(n)
360
+ if cs_a.get(i, EMPTY) != cs_b.get(i, EMPTY)
361
+ ]
362
+ for position, i in enumerate(differing):
363
+ # The limit has to be honoured inside the level too. A
364
+ # single bucket can yield thousands of differences, so
365
+ # checking only between queue pops would blow past
366
+ # max_diffs and still call the walk complete.
367
+ if limit_reached():
368
+ unchecked += len(differing) - position + len(queue)
369
+ break
370
+ va, vb = cs_a.get(i, EMPTY), cs_b.get(i, EMPTY)
371
+ b_lo_i, b_hi_i = bucket_bounds(i, s_lo, s_hi, n)
372
+ if max(va[0], vb[0]) <= threshold or b_hi_i - b_lo_i <= 1:
373
+ _compare_rows(
374
+ pool, a, b, a_table, b_table, key,
375
+ a_cols, b_cols, b_lo_i, b_hi_i, diffs, stats,
376
+ )
377
+ else:
378
+ queue.append((b_lo_i, b_hi_i))
379
+
380
+ diffs.sort(key=lambda d: d.key)
381
+ if max_diffs is not None and len(diffs) > max_diffs:
382
+ # A single bucket download can overshoot the limit by a lot. Report the
383
+ # first `max_diffs` in key order and say the rest were not listed.
384
+ unchecked += len(diffs) - max_diffs
385
+ del diffs[max_diffs:]
386
+ if unchecked:
387
+ # Partial answers get a flag on the result, not merely a warning
388
+ # string, so no caller can mistake one for a clean comparison.
389
+ truncated = True
390
+ warnings.append(
391
+ f"stopped at the --max-diffs limit of {max_diffs}; "
392
+ f"{unchecked} further difference(s) or key range(s) were not "
393
+ f"reported, so this is a partial answer"
394
+ )
395
+
396
+ stats.seconds = time.perf_counter() - started
397
+ return DiffResult(
398
+ diffs, stats, a_cols, warnings,
399
+ truncated=truncated, float_scale=a.float_scale,
400
+ )
401
+
402
+
403
+ def _compare_rows(
404
+ pool: ThreadPoolExecutor,
405
+ a: Dialect,
406
+ b: Dialect,
407
+ a_table: str,
408
+ b_table: str,
409
+ key: str,
410
+ a_cols: list[Column],
411
+ b_cols: list[Column],
412
+ lo: int,
413
+ hi: int,
414
+ diffs: list[RowDiff],
415
+ stats: DiffStats,
416
+ ) -> None:
417
+ """Download a proven-different range from both sides and diff it locally."""
418
+ fa = pool.submit(a.fetch_range, a_table, key, a_cols, lo, hi)
419
+ fb = pool.submit(b.fetch_range, b_table, key, b_cols, lo, hi)
420
+ rows_a, rows_b = _gather(fa, fb)
421
+ stats.queries += 2
422
+ stats.rows_downloaded += len(rows_a) + len(rows_b)
423
+
424
+ names = [c.name for c in a_cols]
425
+ # `strict=True` on every zip below is load-bearing, not tidiness. Both sides
426
+ # render the same column list in the same order, so the tuples must be the
427
+ # same length as `names`. If that invariant ever broke, a plain zip would
428
+ # silently truncate and simply not report the trailing columns - a
429
+ # difference the tool found and then dropped, which is the one outcome it
430
+ # must never produce. Better to raise.
431
+ for k in sorted(set(rows_a) | set(rows_b)):
432
+ ra, rb = rows_a.get(k), rows_b.get(k)
433
+ if ra is None:
434
+ diffs.append(
435
+ RowDiff(k, "only_in_b", names, {}, dict(zip(names, rb or (), strict=True)))
436
+ )
437
+ elif rb is None:
438
+ diffs.append(
439
+ RowDiff(k, "only_in_a", names, dict(zip(names, ra, strict=True)), {})
440
+ )
441
+ elif ra != rb:
442
+ changed = [n for n, x, y in zip(names, ra, rb, strict=True) if x != y]
443
+ diffs.append(
444
+ RowDiff(
445
+ k, "different", changed,
446
+ {n: v for n, v in zip(names, ra, strict=True) if n in changed},
447
+ {n: v for n, v in zip(names, rb, strict=True) if n in changed},
448
+ )
449
+ )