parity-diff 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {parity_diff-0.2.2 → parity_diff-0.2.3}/CONTRIBUTING.md +41 -0
  2. {parity_diff-0.2.2/src/parity_diff.egg-info → parity_diff-0.2.3}/PKG-INFO +1 -1
  3. {parity_diff-0.2.2 → parity_diff-0.2.3}/pyproject.toml +1 -1
  4. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/__init__.py +1 -1
  5. {parity_diff-0.2.2 → parity_diff-0.2.3/src/parity_diff.egg-info}/PKG-INFO +1 -1
  6. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/SOURCES.txt +3 -0
  7. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/conftest.py +10 -0
  8. parity_diff-0.2.3/tests/test_identical.py +222 -0
  9. parity_diff-0.2.3/tests/test_identical_live.py +382 -0
  10. parity_diff-0.2.3/tests/test_mysql_postgres.py +106 -0
  11. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_properties.py +2 -1
  12. {parity_diff-0.2.2 → parity_diff-0.2.3}/LICENSE +0 -0
  13. {parity_diff-0.2.2 → parity_diff-0.2.3}/MANIFEST.in +0 -0
  14. {parity_diff-0.2.2 → parity_diff-0.2.3}/README.md +0 -0
  15. {parity_diff-0.2.2 → parity_diff-0.2.3}/setup.cfg +0 -0
  16. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/cli.py +0 -0
  17. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/__init__.py +0 -0
  18. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/base.py +0 -0
  19. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/duckdb_dialect.py +0 -0
  20. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/mysql_dialect.py +0 -0
  21. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/postgres_dialect.py +0 -0
  22. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/snowflake_dialect.py +0 -0
  23. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/engine.py +0 -0
  24. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/types.py +0 -0
  25. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/dependency_links.txt +0 -0
  26. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/entry_points.txt +0 -0
  27. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/requires.txt +0 -0
  28. {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/top_level.txt +0 -0
  29. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/fakes.py +0 -0
  30. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_cli.py +0 -0
  31. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_encoding.py +0 -0
  32. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_engine.py +0 -0
  33. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_fuzz_encoding.py +0 -0
  34. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_integration.py +0 -0
  35. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_mysql.py +0 -0
  36. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_snowflake.py +0 -0
  37. {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_snowflake_offline.py +0 -0
@@ -97,6 +97,34 @@ MySQL, added after v0.1.0, turned up two more that a warehouse dialect may hit:
97
97
  sentinel silently became `N`. Build such bytes from `CHAR`/hex, not a
98
98
  literal, and verify by hashing rather than by reading the SQL.
99
99
 
100
+ Snowflake, the first warehouse (v0.2.2), added three more — the sort a warehouse
101
+ is especially likely to spring:
102
+
103
+ 9. **Your engine may have neither a bit-cast nor `CONV` to reach 60 bits.**
104
+ Snowflake had no `bit(60)::bigint` and no `conv(hex,16,10)`. What it did have
105
+ is `md5_number_upper64(x)`, the top 64 bits of the digest as a number, and
106
+ `floor(that / 16)` drops the low 4 to land on the same 60-bit prefix - the
107
+ fourth distinct path to `648541476951500027`. Find your engine's own route;
108
+ the constant is the contract, not the SQL that reaches it.
109
+
110
+ 10. **Integers and decimals may share one type name.** Snowflake reports both as
111
+ `NUMBER` and only `numeric_scale` (0 = integer) tells them apart, so its
112
+ `columns()` reads the scale instead of trusting `data_type`. Trust the type
113
+ name and an integer key renders as `42.000000` and never matches another
114
+ engine's `42`. If your engine collapses numeric types like this, override
115
+ `columns()`.
116
+
117
+ 11. **Identifier case-folding is the engine's, and it is not universal.**
118
+ Snowflake upper-cases unquoted identifiers where PostgreSQL and DuckDB
119
+ lower-case them. The engine matches keys and columns case-insensitively for
120
+ exactly this reason (`_fold_columns`), but the *table* name is looked up in
121
+ the case the engine stored, so `--a-table orders` against a Snowflake
122
+ `ORDERS` fails with a near-miss hint. And some warehouses (Snowflake among
123
+ them) offer only READ COMMITTED, so unlike PostgreSQL the walk cannot be
124
+ pinned to one snapshot - a real limitation to document, not hide. None of
125
+ these three showed up in the docs; they surfaced only against a live
126
+ account, which is why the rule below is not negotiable.
127
+
100
128
  ### Proving it works
101
129
 
102
130
  A dialect is not done until `tests/test_encoding.py` passes against it. That
@@ -136,6 +164,19 @@ reason next to it rather than being switched off globally.
136
164
  PostgreSQL-backed tests read `PARITY_TEST_PG` and skip cleanly when nothing is
137
165
  listening, so the suite is useful with only DuckDB installed.
138
166
 
167
+ The generative suites hunt for the one failure that matters most — a real
168
+ difference reported as identical. To run that hunt deeper (many more generated
169
+ cases per property, at the cost of time), scale it with `PARITY_DEEP`:
170
+
171
+ ```bash
172
+ PARITY_DEEP=20 pytest tests/test_identical.py tests/test_properties.py
173
+ ```
174
+
175
+ The default of `1` keeps the everyday suite fast; the deep run is the repeatable
176
+ "make sure there is no abnormality" check on the identical guarantee. A dozen
177
+ seeds at `PARITY_DEEP=6` have turned up nothing — but the point is that anyone
178
+ can re-run it.
179
+
139
180
  ## Scope
140
181
 
141
182
  Before proposing a feature, check it against the question the tool exists to
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parity-diff
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
5
5
  Author: Alessio Sorio
6
6
  License-Expression: MIT
@@ -3,7 +3,7 @@
3
3
  # PyPI by an empty project. The import name, the CLI command and the repo are
4
4
  # all still `parity`: `pip install parity-diff` gives you `parity ...`.
5
5
  name = "parity-diff"
6
- version = "0.2.2"
6
+ version = "0.2.3"
7
7
  description = "Prove two tables in two different database engines hold the same data - without moving the data out of either engine."
8
8
  readme = "README.md"
9
9
  license = "MIT"
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  from typing import Any
15
15
 
16
- __version__ = "0.2.2"
16
+ __version__ = "0.2.3"
17
17
 
18
18
  __all__ = ["__version__", "diff", "get_dialect"]
19
19
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parity-diff
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
5
5
  Author: Alessio Sorio
6
6
  License-Expression: MIT
@@ -25,8 +25,11 @@ tests/test_cli.py
25
25
  tests/test_encoding.py
26
26
  tests/test_engine.py
27
27
  tests/test_fuzz_encoding.py
28
+ tests/test_identical.py
29
+ tests/test_identical_live.py
28
30
  tests/test_integration.py
29
31
  tests/test_mysql.py
32
+ tests/test_mysql_postgres.py
30
33
  tests/test_properties.py
31
34
  tests/test_snowflake.py
32
35
  tests/test_snowflake_offline.py
@@ -12,6 +12,16 @@ import pytest
12
12
 
13
13
  from parity.dialects.base import get_dialect
14
14
 
15
+ #: Multiplier for Hypothesis example counts. `PARITY_DEEP=20 pytest` runs the
16
+ #: generative suites twenty times deeper - the repeatable "make sure there is no
17
+ #: abnormality" hunt for the identical check, without changing the fast default.
18
+ DEEP = max(1, int(os.environ.get("PARITY_DEEP", "1")))
19
+
20
+
21
+ def deep_examples(n: int) -> int:
22
+ """Scale a Hypothesis `max_examples` by the PARITY_DEEP factor (default 1)."""
23
+ return n * DEEP
24
+
15
25
  #: CLAUDE.md section 7 documents a Docker container on port 55432. A native
16
26
  #: install on 5432 is equally valid, so the endpoint is configurable.
17
27
  PG_URL = os.environ.get(
@@ -0,0 +1,222 @@
1
+ """The identical check, stress-tested from every offline angle.
2
+
3
+ A parity tool that reports a false match is worse than useless (CLAUDE.md 8), so
4
+ the property that identical tables report identical - and that identical-except-
5
+ one-thing never does - deserves its own dedicated hunt. These run against the
6
+ in-memory `FakeDialect`, so they exercise the *bisection, matching and checksum*
7
+ logic at high volume with hostile data; the cross-engine *encoding* side of the
8
+ same promise lives in test_encoding.py and the live suites.
9
+
10
+ Two deliberate choices about the field separator (0x1f):
11
+ - The "identical stays identical" tests include it in the data on purpose - the
12
+ same bytes on both sides must compare equal however hostile.
13
+ - The false-identical hunter excludes it, because `concat_ws` smearing on a
14
+ separator that appears in real data is a documented limitation (CLAUDE.md
15
+ 4.3), not an abnormality, and planting it would test the wrong thing.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import pytest
21
+
22
+ pytest.importorskip("hypothesis")
23
+
24
+ from conftest import deep_examples
25
+ from fakes import DictTable, FakeDialect, SyntheticTable
26
+ from hypothesis import assume, given, settings
27
+ from hypothesis import strategies as st
28
+
29
+ from parity.engine import diff
30
+ from parity.types import Column, LogicalType
31
+
32
+ # Hostile text: full Unicode, control characters, even NUL - the fake is pure
33
+ # Python, so the point is that identical *bytes* stay identical however ugly.
34
+ # Surrogates are excluded only because they have no encoding at all.
35
+ _FULL = st.text(
36
+ alphabet=st.characters(blacklist_categories=("Cs",)), max_size=18
37
+ )
38
+ # The same, minus the field separator, for tests that plant a difference.
39
+ _SAFE = st.text(
40
+ alphabet=st.characters(blacklist_characters="\x1f", blacklist_categories=("Cs",)),
41
+ max_size=18,
42
+ )
43
+ _KEY = st.integers(min_value=-(10**12), max_value=10**12)
44
+
45
+
46
+ def _table(text=_SAFE, min_rows=0, max_cols=6):
47
+ """A strategy for (columns, rows): 1..max_cols string columns, sparse keys."""
48
+
49
+ @st.composite
50
+ def build(draw):
51
+ """Draw one (columns, rows) pair for the strategy above."""
52
+ ncols = draw(st.integers(min_value=1, max_value=max_cols))
53
+ cols = [Column(f"c{i}", LogicalType.STRING, "varchar") for i in range(ncols)]
54
+ keys = draw(
55
+ st.lists(_KEY, unique=True, min_size=min_rows, max_size=30)
56
+ )
57
+ rows = {k: tuple(draw(text) for _ in range(ncols)) for k in keys}
58
+ return cols, rows
59
+
60
+ return build()
61
+
62
+
63
+ def _oracle(rows_a: dict, rows_b: dict) -> list[tuple[int, str]]:
64
+ """The truth, computed the dumb way, for the change hunter."""
65
+ out = []
66
+ for k in set(rows_a) | set(rows_b):
67
+ a, b = rows_a.get(k), rows_b.get(k)
68
+ if a is None:
69
+ out.append((k, "only_in_b"))
70
+ elif b is None:
71
+ out.append((k, "only_in_a"))
72
+ elif a != b:
73
+ out.append((k, "different"))
74
+ return sorted(out)
75
+
76
+
77
+ def _run(cols_a, rows_a, cols_b, rows_b, **kwargs):
78
+ """Diff two explicit (columns, rows) sides through the engine."""
79
+ a = FakeDialect(DictTable(cols_a, rows_a), side="A")
80
+ b = FakeDialect(DictTable(cols_b, rows_b), side="B")
81
+ return diff(a, b, "a.t", "b.t", "id", **kwargs)
82
+
83
+
84
+ def _kinds(result) -> list[tuple[int, str]]:
85
+ """Result as a sorted (key, kind) list."""
86
+ return sorted((d.key, d.kind) for d in result.diffs)
87
+
88
+
89
+ # ---------------------------------------------------------------------------
90
+ # Identical stays identical - however hostile the data, whatever the knobs.
91
+ # ---------------------------------------------------------------------------
92
+
93
+
94
+ @settings(max_examples=deep_examples(500))
95
+ @given(
96
+ data=_table(text=_FULL),
97
+ bisection_factor=st.integers(min_value=2, max_value=64),
98
+ threshold=st.integers(min_value=1, max_value=100),
99
+ )
100
+ def test_identical_tables_are_identical_under_every_knob(data, bisection_factor, threshold):
101
+ """A table against itself: identical, no diffs, zero rows moved - for any
102
+ fan-out and threshold, and even with separators and NULs in the data."""
103
+ cols, rows = data
104
+ result = _run(
105
+ cols, rows, cols, dict(rows),
106
+ bisection_factor=bisection_factor, threshold=threshold,
107
+ )
108
+ assert result.identical
109
+ assert result.diffs == []
110
+ assert result.stats.rows_downloaded == 0
111
+
112
+
113
+ @settings(max_examples=300)
114
+ @given(data=_table(text=_FULL, min_rows=1))
115
+ def test_identical_downloads_zero_and_costs_four_queries(data):
116
+ """The headline efficiency claim on any non-empty table: 4 queries
117
+ (2 key_stats + 2 first-level checksums) and nothing downloaded."""
118
+ cols, rows = data
119
+ result = _run(cols, rows, cols, dict(rows))
120
+ assert result.identical
121
+ assert result.stats.rows_downloaded == 0
122
+ assert result.stats.queries == 4
123
+
124
+
125
+ @settings(max_examples=300)
126
+ @given(data=_table(text=_FULL, min_rows=1), perm=st.randoms(use_true_random=False))
127
+ def test_identical_is_independent_of_column_order(data, perm):
128
+ """Columns are matched by name, so the same data with columns in a different
129
+ order on side B still compares identical - order is not content."""
130
+ cols, rows = data
131
+ order = list(range(len(cols)))
132
+ perm.shuffle(order)
133
+ cols_b = [cols[i] for i in order]
134
+ rows_b = {k: tuple(v[i] for i in order) for k, v in rows.items()}
135
+ result = _run(cols, rows, cols_b, rows_b)
136
+ assert result.identical
137
+ assert result.stats.rows_downloaded == 0
138
+
139
+
140
+ @settings(max_examples=100, deadline=None) # large-table walks are legitimately slow
141
+ @given(n=st.integers(min_value=1, max_value=200_000), bf=st.integers(min_value=2, max_value=64))
142
+ def test_identical_at_scale_downloads_nothing(n, bf):
143
+ """Two identical generated tables of up to 200k rows: still zero download,
144
+ and the query count stays tiny (a logarithmic walk that never recurses)."""
145
+ a = FakeDialect(SyntheticTable(n), side="A")
146
+ b = FakeDialect(SyntheticTable(n), side="B")
147
+ result = diff(a, b, "t", "t", "id", bisection_factor=bf)
148
+ assert result.identical
149
+ assert result.stats.rows_downloaded == 0
150
+ assert result.stats.queries == 4
151
+
152
+
153
+ @settings(max_examples=200)
154
+ @given(data=_table(text=_FULL))
155
+ def test_the_identical_check_is_deterministic(data):
156
+ """Running the identical check twice gives the same verdict and counts -
157
+ no dependence on dict ordering, threads, or hash seeding."""
158
+ cols, rows = data
159
+ first = _run(cols, rows, cols, dict(rows))
160
+ second = _run(cols, rows, cols, dict(rows))
161
+ assert first.identical == second.identical
162
+ assert _kinds(first) == _kinds(second)
163
+ assert first.stats.rows_downloaded == second.stats.rows_downloaded
164
+ assert first.stats.queries == second.stats.queries
165
+
166
+
167
+ # ---------------------------------------------------------------------------
168
+ # The false-identical hunter: one minimal change must never read as identical.
169
+ # ---------------------------------------------------------------------------
170
+
171
+
172
+ @settings(max_examples=deep_examples(600))
173
+ @given(data=_table(text=_SAFE, min_rows=1), seed=st.randoms(use_true_random=False), newval=_SAFE)
174
+ def test_a_single_planted_change_is_never_called_identical(data, seed, newval):
175
+ """Take an identical pair, apply exactly one change - alter a cell, delete a
176
+ row, or insert a row - and assert the walk never reports identical and finds
177
+ exactly what a brute-force comparison would. This is the failure the whole
178
+ tool exists to prevent, concentrated onto the hardest case: one difference.
179
+ """
180
+ cols, rows = data
181
+ rows_b = dict(rows)
182
+ keys = sorted(rows)
183
+ kind = seed.choice(["change", "delete", "insert"])
184
+ if kind == "change":
185
+ k = seed.choice(keys)
186
+ col = seed.randrange(len(cols))
187
+ v = list(rows_b[k])
188
+ v[col] = newval
189
+ rows_b[k] = tuple(v)
190
+ elif kind == "delete":
191
+ del rows_b[seed.choice(keys)]
192
+ else: # insert a key not already present
193
+ newk = seed.randint(-(10**12), 10**12)
194
+ assume(newk not in rows_b)
195
+ rows_b[newk] = tuple(newval for _ in cols)
196
+
197
+ assume(rows_b != rows) # a change that changed nothing is not a test
198
+
199
+ result = _run(cols, rows, cols, rows_b)
200
+ assert not result.identical, "a real difference was reported as identical"
201
+ assert _kinds(result) == _oracle(rows, rows_b)
202
+ assert result.stats.rows_downloaded > 0
203
+
204
+
205
+ @settings(max_examples=300)
206
+ @given(data=_table(text=_SAFE, min_rows=1), seed=st.randoms(use_true_random=False))
207
+ def test_null_versus_value_on_one_cell_is_found(data, seed):
208
+ """The migration bug class: a value on one side, absent (a different value)
209
+ on the other, in a single cell, must be caught - never smoothed to a match.
210
+ """
211
+ cols, rows = data
212
+ k = seed.choice(sorted(rows))
213
+ col = seed.randrange(len(cols))
214
+ v = list(rows[k])
215
+ # Flip the cell to something guaranteed different from what is there.
216
+ v[col] = v[col] + "␀" if v[col] != "␀" else "x"
217
+ rows_b = dict(rows)
218
+ rows_b[k] = tuple(v)
219
+ result = _run(cols, rows, cols, rows_b)
220
+ assert not result.identical
221
+ assert _kinds(result) == [(k, "different")]
222
+ assert result.diffs[0].columns == [f"c{col}"]
@@ -0,0 +1,382 @@
1
+ """The identical check across engines, over key types and values the core
2
+ integration suite does not reach.
3
+
4
+ `test_integration.py` proves identical tables match across PostgreSQL and DuckDB
5
+ with an *integer* key. This file widens the identical check where a cross-engine
6
+ abnormality would actually hide: the *hashed* key paths (text, composite, uuid),
7
+ which bucket by a hash of the key's rendered text, and edge-case values
8
+ (bigint extremes, non-finite floats, high-precision decimals, astral-plane
9
+ Unicode, an all-NULL row). Every table is built by a deterministic expression
10
+ spelled identically on both engines, so any reported difference is the tool
11
+ rendering the same value two ways - the exact false-positive the identical check
12
+ must never produce.
13
+
14
+ Skips cleanly without PostgreSQL.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import hashlib
20
+
21
+ import pytest
22
+ from conftest import PG_SCHEMA, duckdb_write, open_duckdb, open_pg
23
+
24
+ from parity.engine import diff
25
+
26
+ pytestmark = pytest.mark.postgres
27
+
28
+ N = 5_000
29
+
30
+ # md5(i) agrees across engines, giving a deterministic text/uuid key. Each SELECT
31
+ # is portable SQL that produces byte-identical data on PostgreSQL and DuckDB.
32
+ TABLES = {
33
+ "text_key": (
34
+ "uid",
35
+ f"""
36
+ select md5(i::varchar) as uid,
37
+ (i % 97)::integer as customer_id,
38
+ ((i * 7 % 100000) / 100.0)::decimal(12,2) as amount,
39
+ case when i % 13 = 0 then null
40
+ else 'n' || i::varchar end as note
41
+ from generate_series(1, {N}) as s(i)
42
+ """,
43
+ ),
44
+ "composite_key": (
45
+ "grp,id",
46
+ f"""
47
+ select (i % 100)::integer as grp,
48
+ i::bigint as id,
49
+ ((i * 3 % 50000) / 100.0)::decimal(12,2) as amount,
50
+ (i % 7 = 0) as flag
51
+ from generate_series(1, {N}) as s(i)
52
+ """,
53
+ ),
54
+ "uuid_key": (
55
+ "uid",
56
+ f"""
57
+ select cast(
58
+ substr(md5(i::varchar), 1, 8) || '-' ||
59
+ substr(md5(i::varchar), 9, 4) || '-' ||
60
+ substr(md5(i::varchar), 13, 4) || '-' ||
61
+ substr(md5(i::varchar), 17, 4) || '-' ||
62
+ substr(md5(i::varchar), 21, 12) as uuid) as uid,
63
+ (i % 97)::integer as customer_id
64
+ from generate_series(1, {N}) as s(i)
65
+ """,
66
+ ),
67
+ # One row per hostile value: bigint extremes, a non-finite float, a
68
+ # high-precision decimal, an astral-plane emoji, an empty string, and an
69
+ # all-NULL row. `{{f}}` is the engine's double type, filled at build time.
70
+ "edge_values": (
71
+ "id",
72
+ """
73
+ select cast(1 as bigint) as id,
74
+ cast('9223372036854775807' as bigint) as big,
75
+ cast(123456789012.345678 as decimal(38,6)) as dec,
76
+ cast('Infinity' as {f}) as flt,
77
+ timestamp '2024-02-29 13:04:05.123456' as ts,
78
+ 'añ日\U0001f600' as txt
79
+ union all select cast(2 as bigint), cast('-9223372036854775808' as bigint),
80
+ cast(-0.000001 as decimal(38,6)), cast('-Infinity' as {f}),
81
+ timestamp '2000-01-01 00:00:00', ''
82
+ union all select cast(3 as bigint), cast(0 as bigint),
83
+ cast(0 as decimal(38,6)), cast('NaN' as {f}),
84
+ timestamp '2099-12-31 23:59:59.999999', null
85
+ union all select cast(4 as bigint), null,
86
+ null, cast(1.5 as {f}), null, 'plain'
87
+ """,
88
+ ),
89
+ }
90
+
91
+
92
+ def _build(cursor_exec, table_sql, float_type: str) -> None:
93
+ """Create every table via the given execute callable (one engine).
94
+
95
+ `float_type` is that engine's double spelling - PostgreSQL wants
96
+ ``double precision`` where DuckDB wants ``double`` - filled into the edge
97
+ table's non-finite-float columns.
98
+ """
99
+ for name, (_key, select) in TABLES.items():
100
+ cursor_exec(f"create table {table_sql(name)} as {select.format(f=float_type)}")
101
+
102
+
103
+ @pytest.fixture(scope="module")
104
+ def duck(tmp_path_factory):
105
+ """Side B: all tables in one DuckDB file, opened read-only."""
106
+ path = str(tmp_path_factory.mktemp("ident_live") / "b.duckdb")
107
+ con = duckdb_write(path)
108
+ try:
109
+ _build(con.execute, lambda n: f"main.{n}", "double")
110
+ finally:
111
+ con.close()
112
+ d = open_duckdb(path, side="B")
113
+ yield d
114
+ d.close()
115
+
116
+
117
+ SCHEMA = f"{PG_SCHEMA}_identlive"
118
+
119
+
120
+ @pytest.fixture(scope="module")
121
+ def pg(pg_url):
122
+ """Side A: all tables in a schema the test owns, opened read-only."""
123
+ import psycopg
124
+
125
+ con = psycopg.connect(pg_url, autocommit=True)
126
+ try:
127
+ con.execute(f"drop schema if exists {SCHEMA} cascade")
128
+ con.execute(f"create schema {SCHEMA}")
129
+ _build(con.execute, lambda n: f"{SCHEMA}.{n}", "double precision")
130
+ finally:
131
+ con.close()
132
+ d = open_pg(pg_url, side="A")
133
+ yield d
134
+ d.close()
135
+
136
+
137
+ @pytest.mark.parametrize("name", list(TABLES))
138
+ def test_identical_across_engines_by_key_type(pg, duck, name):
139
+ """The same table, same data, one hashed key type at a time: identical,
140
+ and not one row moved across the network."""
141
+ key = TABLES[name][0].split(",") # a list, so composite keys resolve
142
+ result = diff(pg, duck, f"{SCHEMA}.{name}", f"main.{name}", key)
143
+ assert result.identical, (
144
+ f"{name}: identical data reported {len(result.diffs)} diffs "
145
+ f"{[(d.key, d.kind, d.columns) for d in result.diffs[:3]]}"
146
+ )
147
+ assert result.stats.rows_downloaded == 0
148
+
149
+
150
+ def test_a_representation_change_across_engines_reads_identical(pg_url, tmp_path):
151
+ """The migration scenario: the same values stored with *different declared
152
+ types* on each engine - integer vs bigint, numeric(12,2) vs double,
153
+ varchar vs text - must still read identical end to end. Storing a value a
154
+ different way is not changing it; the diff must not mistake it for one.
155
+ """
156
+ import psycopg
157
+
158
+ n = 2_000
159
+ schema = f"{PG_SCHEMA}_repr"
160
+ pg_sql = f"""
161
+ create table {schema}.repr as
162
+ select i::bigint as id,
163
+ (i / 100.0)::numeric(12,2) as amount,
164
+ ('r' || i::text)::varchar(50) as name,
165
+ (timestamp '2024-01-01 00:00:00'
166
+ + (i % 86400) * interval '1 second') as ts
167
+ from generate_series(1, {n}) as s(i)
168
+ """
169
+ duck_sql = f"""
170
+ create table repr as
171
+ select i::integer as id,
172
+ (i / 100.0)::double as amount,
173
+ ('r' || i::varchar) as name,
174
+ (timestamp '2024-01-01 00:00:00'
175
+ + (i % 86400) * interval '1 second') as ts
176
+ from generate_series(1, {n}) as s(i)
177
+ """
178
+ con = psycopg.connect(pg_url, autocommit=True)
179
+ try:
180
+ con.execute(f"drop schema if exists {schema} cascade")
181
+ con.execute(f"create schema {schema}")
182
+ con.execute(pg_sql)
183
+ finally:
184
+ con.close()
185
+ path = str(tmp_path / "repr.duckdb")
186
+ dcon = duckdb_write(path)
187
+ try:
188
+ dcon.execute(duck_sql)
189
+ finally:
190
+ dcon.close()
191
+
192
+ a = open_pg(pg_url, side="A")
193
+ b = open_duckdb(path, side="B")
194
+ try:
195
+ result = diff(a, b, f"{schema}.repr", "main.repr", "id")
196
+ finally:
197
+ a.close()
198
+ b.close()
199
+ con = psycopg.connect(pg_url, autocommit=True)
200
+ try:
201
+ con.execute(f"drop schema if exists {schema} cascade")
202
+ finally:
203
+ con.close()
204
+
205
+ assert result.identical, (
206
+ "a pure representation change was reported as a data difference: "
207
+ f"{[(d.key, d.columns) for d in result.diffs[:5]]}"
208
+ )
209
+ assert result.stats.rows_downloaded == 0
210
+
211
+
212
+ def test_a_hashed_key_table_still_finds_a_planted_difference(pg, tmp_path):
213
+ """The other half of the promise: the identical check on a hashed key must
214
+ never MISS a real difference either. Plant one changed row in a text-keyed
215
+ copy and confirm the walk reports exactly it, keyed by the real text
216
+ identity - never by the 60-bit bucket hash, which could collide.
217
+ """
218
+ path = str(tmp_path / "text_key_perturbed.duckdb")
219
+ con = duckdb_write(path)
220
+ try:
221
+ con.execute(f"create table text_key as {TABLES['text_key'][1]}")
222
+ # md5('1234') is the uid of the row generated for i = 1234.
223
+ con.execute("update text_key set amount = amount + 0.01 where uid = md5('1234')")
224
+ finally:
225
+ con.close()
226
+
227
+ b = open_duckdb(path, side="B")
228
+ try:
229
+ result = diff(pg, b, f"{SCHEMA}.text_key", "main.text_key", "uid")
230
+ finally:
231
+ b.close()
232
+
233
+ # The same md5 the engines compute, for cross-engine agreement not security.
234
+ uid = hashlib.md5(b"1234", usedforsecurity=False).hexdigest()
235
+ assert [(d.key, d.kind) for d in result.diffs] == [(uid, "different")]
236
+ assert result.diffs[0].columns == ["amount"]
237
+
238
+
239
+ def test_a_very_wide_table_reads_identical_across_engines(pg_url, tmp_path):
240
+ """A table with more columns than PostgreSQL's 100-argument `concat_ws`
241
+ limit forces `row_text` to build a nested tree of `concat_ws` calls. The two
242
+ engines must build the *same* tree over the same values, or an identical
243
+ wide table - an ordinary denormalised fact table - would report every row as
244
+ different. Also plants one change to prove the wide path finds a real one.
245
+ """
246
+ import psycopg
247
+
248
+ ncols = 120 # over PostgreSQL's 99-argument concat_ws limit
249
+ cols = ", ".join(f"('v' || (i + {n})::varchar) as c{n}" for n in range(ncols))
250
+ select = f"select i::bigint as id, {cols} from generate_series(1, 500) as s(i)"
251
+ schema = f"{PG_SCHEMA}_wide"
252
+
253
+ con = psycopg.connect(pg_url, autocommit=True)
254
+ try:
255
+ con.execute(f"drop schema if exists {schema} cascade")
256
+ con.execute(f"create schema {schema}")
257
+ con.execute(f"create table {schema}.wide as {select}")
258
+ finally:
259
+ con.close()
260
+ path = str(tmp_path / "wide.duckdb")
261
+ dcon = duckdb_write(path)
262
+ try:
263
+ dcon.execute(f"create table wide as {select}")
264
+ dcon.execute("create table wide_p as select * from wide")
265
+ dcon.execute("update wide_p set c50 = 'CHANGED' where id = 321")
266
+ finally:
267
+ dcon.close()
268
+
269
+ a = open_pg(pg_url, side="A")
270
+ b = open_duckdb(path, side="B")
271
+ try:
272
+ same = diff(a, b, f"{schema}.wide", "main.wide", "id")
273
+ changed = diff(a, b, f"{schema}.wide", "main.wide_p", "id")
274
+ finally:
275
+ a.close()
276
+ b.close()
277
+ con = psycopg.connect(pg_url, autocommit=True)
278
+ try:
279
+ con.execute(f"drop schema if exists {schema} cascade")
280
+ finally:
281
+ con.close()
282
+
283
+ assert same.identical, (
284
+ "a 120-column identical table reported differences - the nested "
285
+ f"concat_ws trees disagree: {[(d.key, d.columns) for d in same.diffs[:3]]}"
286
+ )
287
+ assert same.stats.rows_downloaded == 0
288
+ assert [(d.key, d.kind) for d in changed.diffs] == [(321, "different")]
289
+ assert changed.diffs[0].columns == ["c50"]
290
+
291
+
292
+ def test_a_same_content_swap_in_one_bucket_is_caught_live(pg_url, tmp_path):
293
+ """The exact 0.2.1 bug, reproduced across two real engines.
294
+
295
+ Every row carries identical non-key content, and one side holds key 2500
296
+ while the other holds the adjacent key 2501 - a delete and an insert of the
297
+ same content in the same bucket. Before the key was folded into each
298
+ bucket's checksum the counts balanced and the content sums matched, so both
299
+ rows vanished: a false "identical". They must both be found.
300
+ """
301
+ import psycopg
302
+
303
+ # Shared keys are 1..5000 except {2500, 2501}; PostgreSQL adds 2500, DuckDB
304
+ # adds 2501. Identical content everywhere, so only the key distinguishes them.
305
+ body = "1.00::decimal(12,2) as amount, 'x' as status"
306
+ pg_select = f"select i::bigint as id, {body} from generate_series(1,5000) as s(i) where i <> 2501"
307
+ duck_select = f"select i::bigint as id, {body} from generate_series(1,5000) as s(i) where i <> 2500"
308
+ schema = f"{PG_SCHEMA}_swap"
309
+
310
+ con = psycopg.connect(pg_url, autocommit=True)
311
+ try:
312
+ con.execute(f"drop schema if exists {schema} cascade")
313
+ con.execute(f"create schema {schema}")
314
+ con.execute(f"create table {schema}.swap as {pg_select}")
315
+ finally:
316
+ con.close()
317
+ path = str(tmp_path / "swap.duckdb")
318
+ dcon = duckdb_write(path)
319
+ try:
320
+ dcon.execute(f"create table swap as {duck_select}")
321
+ finally:
322
+ dcon.close()
323
+
324
+ a = open_pg(pg_url, side="A")
325
+ b = open_duckdb(path, side="B")
326
+ try:
327
+ result = diff(a, b, f"{schema}.swap", "main.swap", "id")
328
+ finally:
329
+ a.close()
330
+ b.close()
331
+ con = psycopg.connect(pg_url, autocommit=True)
332
+ try:
333
+ con.execute(f"drop schema if exists {schema} cascade")
334
+ finally:
335
+ con.close()
336
+
337
+ assert not result.identical, "a same-content insert+delete cancelled to a false match"
338
+ assert [(d.key, d.kind) for d in result.diffs] == [
339
+ (2500, "only_in_a"),
340
+ (2501, "only_in_b"),
341
+ ]
342
+
343
+
344
+ @pytest.mark.parametrize("rows,queries", [(0, 2), (1, 4)])
345
+ def test_degenerate_identical_tables_across_engines(pg_url, tmp_path, rows, queries):
346
+ """The degenerate sizes where off-by-ones hide: an empty table costs two
347
+ queries (key_stats only, nothing to bisect) and a one-row table costs four
348
+ (2 key_stats + 2 checksums), both identical and both moving zero rows."""
349
+ import psycopg
350
+
351
+ select = f"select i::bigint as id, ('v' || i::text) as note from generate_series(1, {rows}) as s(i)"
352
+ schema = f"{PG_SCHEMA}_degen"
353
+ con = psycopg.connect(pg_url, autocommit=True)
354
+ try:
355
+ con.execute(f"drop schema if exists {schema} cascade")
356
+ con.execute(f"create schema {schema}")
357
+ con.execute(f"create table {schema}.t as {select}")
358
+ finally:
359
+ con.close()
360
+ path = str(tmp_path / "degen.duckdb")
361
+ dcon = duckdb_write(path)
362
+ try:
363
+ dcon.execute(f"create table t as {select}")
364
+ finally:
365
+ dcon.close()
366
+
367
+ a = open_pg(pg_url, side="A")
368
+ b = open_duckdb(path, side="B")
369
+ try:
370
+ result = diff(a, b, f"{schema}.t", "main.t", "id")
371
+ finally:
372
+ a.close()
373
+ b.close()
374
+ con = psycopg.connect(pg_url, autocommit=True)
375
+ try:
376
+ con.execute(f"drop schema if exists {schema} cascade")
377
+ finally:
378
+ con.close()
379
+
380
+ assert result.identical
381
+ assert result.stats.rows_downloaded == 0
382
+ assert result.stats.queries == queries
@@ -0,0 +1,106 @@
1
+ """The identical check between the two most common migration endpoints.
2
+
3
+ MySQL and PostgreSQL are each verified against DuckDB elsewhere, so byte-equal
4
+ canonical text makes them agree with each other by transitivity - but a
5
+ MySQL -> PostgreSQL migration is common enough that the pair deserves a direct
6
+ test, with neither side being the reference engine. Identical data built by
7
+ each engine's own dialect must read identical; a single planted change must be
8
+ found exactly.
9
+
10
+ Skips unless both a MySQL and a PostgreSQL server are reachable.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import pytest
16
+ from conftest import PG_SCHEMA, open_mysql, open_pg
17
+
18
+ from parity.engine import diff
19
+
20
+ pytestmark = [pytest.mark.postgres, pytest.mark.mysql]
21
+
22
+ N = 2_000
23
+ SCHEMA = f"{PG_SCHEMA}_mypg"
24
+
25
+ # The same rows, spelled in each engine's own SQL. `flag` is a plain int on both
26
+ # (not a boolean) so this tests data agreement, not the boolean-vs-int trap.
27
+ PG_BUILD = f"""
28
+ create table {SCHEMA}.xeng as
29
+ select i::bigint as id,
30
+ ((i * 7 % 100000) / 100.0)::decimal(12,2) as amount,
31
+ (case when i % 3 = 0 then 'paid'
32
+ when i % 3 = 1 then 'open' else 'void' end)::varchar(10) as status,
33
+ (i % 11 = 0)::int as flag,
34
+ (case when i % 13 = 0 then null
35
+ else 'n' || i::text end)::varchar(20) as note
36
+ from generate_series(1, {N}) as s(i)
37
+ """
38
+
39
+ MYSQL_BUILD = f"""
40
+ create table xeng as
41
+ with recursive seq(i) as (
42
+ select 1 union all select i + 1 from seq where i < {N}
43
+ )
44
+ select cast(i as signed) as id,
45
+ cast((i * 7 mod 100000) / 100.0 as decimal(12,2)) as amount,
46
+ (case when i mod 3 = 0 then 'paid'
47
+ when i mod 3 = 1 then 'open' else 'void' end) as status,
48
+ cast(i mod 11 = 0 as signed) as flag,
49
+ (case when i mod 13 = 0 then null
50
+ else concat('n', i) end) as note
51
+ from seq
52
+ """
53
+
54
+
55
+ @pytest.fixture(scope="module")
56
+ def pg(pg_url):
57
+ """Side A: PostgreSQL, its own schema, opened read-only."""
58
+ import psycopg
59
+
60
+ con = psycopg.connect(pg_url, autocommit=True)
61
+ try:
62
+ con.execute(f"drop schema if exists {SCHEMA} cascade")
63
+ con.execute(f"create schema {SCHEMA}")
64
+ con.execute(PG_BUILD)
65
+ finally:
66
+ con.close()
67
+ d = open_pg(pg_url, side="A")
68
+ yield d
69
+ d.close()
70
+
71
+
72
+ @pytest.fixture(scope="module")
73
+ def my(mysql_url):
74
+ """Side B: MySQL, identical data, opened read-only."""
75
+ d = open_mysql(mysql_url, side="B")
76
+ cur = d._conn.cursor()
77
+ try:
78
+ cur.execute("set session cte_max_recursion_depth = 1000000")
79
+ cur.execute("drop table if exists xeng")
80
+ cur.execute("drop table if exists xeng_p")
81
+ cur.execute(MYSQL_BUILD)
82
+ # A perturbed copy for the planted-difference test, built once.
83
+ cur.execute("create table xeng_p as select * from xeng")
84
+ cur.execute("update xeng_p set amount = amount + 0.01 where id = 1234")
85
+ d._conn.commit()
86
+ finally:
87
+ cur.close()
88
+ yield d
89
+ d.close()
90
+
91
+
92
+ def test_mysql_and_postgres_agree_on_identical_data(pg, my):
93
+ """The same rows on each engine read identical, with zero download."""
94
+ result = diff(pg, my, f"{SCHEMA}.xeng", "xeng", "id")
95
+ assert result.identical, (
96
+ "MySQL and PostgreSQL disagreed on identical data: "
97
+ f"{[(d.key, d.columns, d.values_a, d.values_b) for d in result.diffs[:5]]}"
98
+ )
99
+ assert result.stats.rows_downloaded == 0
100
+
101
+
102
+ def test_a_planted_change_between_mysql_and_postgres_is_found(pg, my):
103
+ """One changed decimal is reported as exactly that row and column."""
104
+ result = diff(pg, my, f"{SCHEMA}.xeng", "xeng_p", "id")
105
+ assert [(d.key, d.kind) for d in result.diffs] == [(1234, "different")]
106
+ assert result.diffs[0].columns == ["amount"]
@@ -22,6 +22,7 @@ import pytest
22
22
  # dependency must not look like a broken build.
23
23
  pytest.importorskip("hypothesis")
24
24
 
25
+ from conftest import deep_examples
25
26
  from fakes import DictTable, FakeDialect
26
27
  from hypothesis import given, settings
27
28
  from hypothesis import strategies as st
@@ -81,7 +82,7 @@ def _kinds(result) -> list[tuple[int, str]]:
81
82
  # ---------------------------------------------------------------------------
82
83
 
83
84
 
84
- @settings(max_examples=400)
85
+ @settings(max_examples=deep_examples(400))
85
86
  @given(a=_table, b=_table)
86
87
  def test_diff_matches_a_brute_force_oracle(a, b):
87
88
  """For any two tables, parity's result equals comparing every row directly.
File without changes
File without changes
File without changes
File without changes
File without changes