parity-diff 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parity_diff-0.2.2 → parity_diff-0.2.3}/CONTRIBUTING.md +41 -0
- {parity_diff-0.2.2/src/parity_diff.egg-info → parity_diff-0.2.3}/PKG-INFO +1 -1
- {parity_diff-0.2.2 → parity_diff-0.2.3}/pyproject.toml +1 -1
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/__init__.py +1 -1
- {parity_diff-0.2.2 → parity_diff-0.2.3/src/parity_diff.egg-info}/PKG-INFO +1 -1
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/SOURCES.txt +3 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/conftest.py +10 -0
- parity_diff-0.2.3/tests/test_identical.py +222 -0
- parity_diff-0.2.3/tests/test_identical_live.py +382 -0
- parity_diff-0.2.3/tests/test_mysql_postgres.py +106 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_properties.py +2 -1
- {parity_diff-0.2.2 → parity_diff-0.2.3}/LICENSE +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/MANIFEST.in +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/README.md +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/setup.cfg +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/cli.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/__init__.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/base.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/duckdb_dialect.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/mysql_dialect.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/postgres_dialect.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/dialects/snowflake_dialect.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/engine.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity/types.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/dependency_links.txt +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/entry_points.txt +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/requires.txt +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/src/parity_diff.egg-info/top_level.txt +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/fakes.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_cli.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_encoding.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_engine.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_fuzz_encoding.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_integration.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_mysql.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_snowflake.py +0 -0
- {parity_diff-0.2.2 → parity_diff-0.2.3}/tests/test_snowflake_offline.py +0 -0
|
@@ -97,6 +97,34 @@ MySQL, added after v0.1.0, turned up two more that a warehouse dialect may hit:
|
|
|
97
97
|
sentinel silently became `N`. Build such bytes from `CHAR`/hex, not a
|
|
98
98
|
literal, and verify by hashing rather than by reading the SQL.
|
|
99
99
|
|
|
100
|
+
Snowflake, the first warehouse (v0.2.2), added three more — the sort a warehouse
|
|
101
|
+
is especially likely to spring:
|
|
102
|
+
|
|
103
|
+
9. **Your engine may have neither a bit-cast nor `CONV` to reach 60 bits.**
|
|
104
|
+
Snowflake had no `bit(60)::bigint` and no `conv(hex,16,10)`. What it did have
|
|
105
|
+
is `md5_number_upper64(x)`, the top 64 bits of the digest as a number, and
|
|
106
|
+
`floor(that / 16)` drops the low 4 to land on the same 60-bit prefix - the
|
|
107
|
+
fourth distinct path to `648541476951500027`. Find your engine's own route;
|
|
108
|
+
the constant is the contract, not the SQL that reaches it.
|
|
109
|
+
|
|
110
|
+
10. **Integers and decimals may share one type name.** Snowflake reports both as
|
|
111
|
+
`NUMBER` and only `numeric_scale` (0 = integer) tells them apart, so its
|
|
112
|
+
`columns()` reads the scale instead of trusting `data_type`. Trust the type
|
|
113
|
+
name and an integer key renders as `42.000000` and never matches another
|
|
114
|
+
engine's `42`. If your engine collapses numeric types like this, override
|
|
115
|
+
`columns()`.
|
|
116
|
+
|
|
117
|
+
11. **Identifier case-folding is the engine's, and it is not universal.**
|
|
118
|
+
Snowflake upper-cases unquoted identifiers where PostgreSQL and DuckDB
|
|
119
|
+
lower-case them. The engine matches keys and columns case-insensitively for
|
|
120
|
+
exactly this reason (`_fold_columns`), but the *table* name is looked up in
|
|
121
|
+
the case the engine stored, so `--a-table orders` against a Snowflake
|
|
122
|
+
`ORDERS` fails with a near-miss hint. And some warehouses (Snowflake among
|
|
123
|
+
them) offer only READ COMMITTED, so unlike PostgreSQL the walk cannot be
|
|
124
|
+
pinned to one snapshot - a real limitation to document, not hide. None of
|
|
125
|
+
these three showed up in the docs; they surfaced only against a live
|
|
126
|
+
account, which is why the rule below is not negotiable.
|
|
127
|
+
|
|
100
128
|
### Proving it works
|
|
101
129
|
|
|
102
130
|
A dialect is not done until `tests/test_encoding.py` passes against it. That
|
|
@@ -136,6 +164,19 @@ reason next to it rather than being switched off globally.
|
|
|
136
164
|
PostgreSQL-backed tests read `PARITY_TEST_PG` and skip cleanly when nothing is
|
|
137
165
|
listening, so the suite is useful with only DuckDB installed.
|
|
138
166
|
|
|
167
|
+
The generative suites hunt for the one failure that matters most — a real
|
|
168
|
+
difference reported as identical. To run that hunt deeper (many more generated
|
|
169
|
+
cases per property, at the cost of time), scale it with `PARITY_DEEP`:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
PARITY_DEEP=20 pytest tests/test_identical.py tests/test_properties.py
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
The default of `1` keeps the everyday suite fast; the deep run is the repeatable
|
|
176
|
+
"make sure there is no abnormality" check on the identical guarantee. A dozen
|
|
177
|
+
seeds at `PARITY_DEEP=6` have turned up nothing — but the point is that anyone
|
|
178
|
+
can re-run it.
|
|
179
|
+
|
|
139
180
|
## Scope
|
|
140
181
|
|
|
141
182
|
Before proposing a feature, check it against the question the tool exists to
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
# PyPI by an empty project. The import name, the CLI command and the repo are
|
|
4
4
|
# all still `parity`: `pip install parity-diff` gives you `parity ...`.
|
|
5
5
|
name = "parity-diff"
|
|
6
|
-
version = "0.2.
|
|
6
|
+
version = "0.2.3"
|
|
7
7
|
description = "Prove two tables in two different database engines hold the same data - without moving the data out of either engine."
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
license = "MIT"
|
|
@@ -25,8 +25,11 @@ tests/test_cli.py
|
|
|
25
25
|
tests/test_encoding.py
|
|
26
26
|
tests/test_engine.py
|
|
27
27
|
tests/test_fuzz_encoding.py
|
|
28
|
+
tests/test_identical.py
|
|
29
|
+
tests/test_identical_live.py
|
|
28
30
|
tests/test_integration.py
|
|
29
31
|
tests/test_mysql.py
|
|
32
|
+
tests/test_mysql_postgres.py
|
|
30
33
|
tests/test_properties.py
|
|
31
34
|
tests/test_snowflake.py
|
|
32
35
|
tests/test_snowflake_offline.py
|
|
@@ -12,6 +12,16 @@ import pytest
|
|
|
12
12
|
|
|
13
13
|
from parity.dialects.base import get_dialect
|
|
14
14
|
|
|
15
|
+
#: Multiplier for Hypothesis example counts. `PARITY_DEEP=20 pytest` runs the
|
|
16
|
+
#: generative suites twenty times deeper - the repeatable "make sure there is no
|
|
17
|
+
#: abnormality" hunt for the identical check, without changing the fast default.
|
|
18
|
+
DEEP = max(1, int(os.environ.get("PARITY_DEEP", "1")))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def deep_examples(n: int) -> int:
|
|
22
|
+
"""Scale a Hypothesis `max_examples` by the PARITY_DEEP factor (default 1)."""
|
|
23
|
+
return n * DEEP
|
|
24
|
+
|
|
15
25
|
#: CLAUDE.md section 7 documents a Docker container on port 55432. A native
|
|
16
26
|
#: install on 5432 is equally valid, so the endpoint is configurable.
|
|
17
27
|
PG_URL = os.environ.get(
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""The identical check, stress-tested from every offline angle.
|
|
2
|
+
|
|
3
|
+
A parity tool that reports a false match is worse than useless (CLAUDE.md 8), so
|
|
4
|
+
the property that identical tables report identical - and that identical-except-
|
|
5
|
+
one-thing never does - deserves its own dedicated hunt. These run against the
|
|
6
|
+
in-memory `FakeDialect`, so they exercise the *bisection, matching and checksum*
|
|
7
|
+
logic at high volume with hostile data; the cross-engine *encoding* side of the
|
|
8
|
+
same promise lives in test_encoding.py and the live suites.
|
|
9
|
+
|
|
10
|
+
Two deliberate choices about the field separator (0x1f):
|
|
11
|
+
- The "identical stays identical" tests include it in the data on purpose - the
|
|
12
|
+
same bytes on both sides must compare equal however hostile.
|
|
13
|
+
- The false-identical hunter excludes it, because `concat_ws` smearing on a
|
|
14
|
+
separator that appears in real data is a documented limitation (CLAUDE.md
|
|
15
|
+
4.3), not an abnormality, and planting it would test the wrong thing.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import pytest
|
|
21
|
+
|
|
22
|
+
pytest.importorskip("hypothesis")
|
|
23
|
+
|
|
24
|
+
from conftest import deep_examples
|
|
25
|
+
from fakes import DictTable, FakeDialect, SyntheticTable
|
|
26
|
+
from hypothesis import assume, given, settings
|
|
27
|
+
from hypothesis import strategies as st
|
|
28
|
+
|
|
29
|
+
from parity.engine import diff
|
|
30
|
+
from parity.types import Column, LogicalType
|
|
31
|
+
|
|
32
|
+
# Hostile text: full Unicode, control characters, even NUL - the fake is pure
|
|
33
|
+
# Python, so the point is that identical *bytes* stay identical however ugly.
|
|
34
|
+
# Surrogates are excluded only because they have no encoding at all.
|
|
35
|
+
_FULL = st.text(
|
|
36
|
+
alphabet=st.characters(blacklist_categories=("Cs",)), max_size=18
|
|
37
|
+
)
|
|
38
|
+
# The same, minus the field separator, for tests that plant a difference.
|
|
39
|
+
_SAFE = st.text(
|
|
40
|
+
alphabet=st.characters(blacklist_characters="\x1f", blacklist_categories=("Cs",)),
|
|
41
|
+
max_size=18,
|
|
42
|
+
)
|
|
43
|
+
_KEY = st.integers(min_value=-(10**12), max_value=10**12)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _table(text=_SAFE, min_rows=0, max_cols=6):
|
|
47
|
+
"""A strategy for (columns, rows): 1..max_cols string columns, sparse keys."""
|
|
48
|
+
|
|
49
|
+
@st.composite
|
|
50
|
+
def build(draw):
|
|
51
|
+
"""Draw one (columns, rows) pair for the strategy above."""
|
|
52
|
+
ncols = draw(st.integers(min_value=1, max_value=max_cols))
|
|
53
|
+
cols = [Column(f"c{i}", LogicalType.STRING, "varchar") for i in range(ncols)]
|
|
54
|
+
keys = draw(
|
|
55
|
+
st.lists(_KEY, unique=True, min_size=min_rows, max_size=30)
|
|
56
|
+
)
|
|
57
|
+
rows = {k: tuple(draw(text) for _ in range(ncols)) for k in keys}
|
|
58
|
+
return cols, rows
|
|
59
|
+
|
|
60
|
+
return build()
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _oracle(rows_a: dict, rows_b: dict) -> list[tuple[int, str]]:
|
|
64
|
+
"""The truth, computed the dumb way, for the change hunter."""
|
|
65
|
+
out = []
|
|
66
|
+
for k in set(rows_a) | set(rows_b):
|
|
67
|
+
a, b = rows_a.get(k), rows_b.get(k)
|
|
68
|
+
if a is None:
|
|
69
|
+
out.append((k, "only_in_b"))
|
|
70
|
+
elif b is None:
|
|
71
|
+
out.append((k, "only_in_a"))
|
|
72
|
+
elif a != b:
|
|
73
|
+
out.append((k, "different"))
|
|
74
|
+
return sorted(out)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _run(cols_a, rows_a, cols_b, rows_b, **kwargs):
|
|
78
|
+
"""Diff two explicit (columns, rows) sides through the engine."""
|
|
79
|
+
a = FakeDialect(DictTable(cols_a, rows_a), side="A")
|
|
80
|
+
b = FakeDialect(DictTable(cols_b, rows_b), side="B")
|
|
81
|
+
return diff(a, b, "a.t", "b.t", "id", **kwargs)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _kinds(result) -> list[tuple[int, str]]:
|
|
85
|
+
"""Result as a sorted (key, kind) list."""
|
|
86
|
+
return sorted((d.key, d.kind) for d in result.diffs)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# ---------------------------------------------------------------------------
|
|
90
|
+
# Identical stays identical - however hostile the data, whatever the knobs.
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@settings(max_examples=deep_examples(500))
|
|
95
|
+
@given(
|
|
96
|
+
data=_table(text=_FULL),
|
|
97
|
+
bisection_factor=st.integers(min_value=2, max_value=64),
|
|
98
|
+
threshold=st.integers(min_value=1, max_value=100),
|
|
99
|
+
)
|
|
100
|
+
def test_identical_tables_are_identical_under_every_knob(data, bisection_factor, threshold):
|
|
101
|
+
"""A table against itself: identical, no diffs, zero rows moved - for any
|
|
102
|
+
fan-out and threshold, and even with separators and NULs in the data."""
|
|
103
|
+
cols, rows = data
|
|
104
|
+
result = _run(
|
|
105
|
+
cols, rows, cols, dict(rows),
|
|
106
|
+
bisection_factor=bisection_factor, threshold=threshold,
|
|
107
|
+
)
|
|
108
|
+
assert result.identical
|
|
109
|
+
assert result.diffs == []
|
|
110
|
+
assert result.stats.rows_downloaded == 0
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@settings(max_examples=300)
|
|
114
|
+
@given(data=_table(text=_FULL, min_rows=1))
|
|
115
|
+
def test_identical_downloads_zero_and_costs_four_queries(data):
|
|
116
|
+
"""The headline efficiency claim on any non-empty table: 4 queries
|
|
117
|
+
(2 key_stats + 2 first-level checksums) and nothing downloaded."""
|
|
118
|
+
cols, rows = data
|
|
119
|
+
result = _run(cols, rows, cols, dict(rows))
|
|
120
|
+
assert result.identical
|
|
121
|
+
assert result.stats.rows_downloaded == 0
|
|
122
|
+
assert result.stats.queries == 4
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@settings(max_examples=300)
|
|
126
|
+
@given(data=_table(text=_FULL, min_rows=1), perm=st.randoms(use_true_random=False))
|
|
127
|
+
def test_identical_is_independent_of_column_order(data, perm):
|
|
128
|
+
"""Columns are matched by name, so the same data with columns in a different
|
|
129
|
+
order on side B still compares identical - order is not content."""
|
|
130
|
+
cols, rows = data
|
|
131
|
+
order = list(range(len(cols)))
|
|
132
|
+
perm.shuffle(order)
|
|
133
|
+
cols_b = [cols[i] for i in order]
|
|
134
|
+
rows_b = {k: tuple(v[i] for i in order) for k, v in rows.items()}
|
|
135
|
+
result = _run(cols, rows, cols_b, rows_b)
|
|
136
|
+
assert result.identical
|
|
137
|
+
assert result.stats.rows_downloaded == 0
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@settings(max_examples=100, deadline=None) # large-table walks are legitimately slow
|
|
141
|
+
@given(n=st.integers(min_value=1, max_value=200_000), bf=st.integers(min_value=2, max_value=64))
|
|
142
|
+
def test_identical_at_scale_downloads_nothing(n, bf):
|
|
143
|
+
"""Two identical generated tables of up to 200k rows: still zero download,
|
|
144
|
+
and the query count stays tiny (a logarithmic walk that never recurses)."""
|
|
145
|
+
a = FakeDialect(SyntheticTable(n), side="A")
|
|
146
|
+
b = FakeDialect(SyntheticTable(n), side="B")
|
|
147
|
+
result = diff(a, b, "t", "t", "id", bisection_factor=bf)
|
|
148
|
+
assert result.identical
|
|
149
|
+
assert result.stats.rows_downloaded == 0
|
|
150
|
+
assert result.stats.queries == 4
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@settings(max_examples=200)
|
|
154
|
+
@given(data=_table(text=_FULL))
|
|
155
|
+
def test_the_identical_check_is_deterministic(data):
|
|
156
|
+
"""Running the identical check twice gives the same verdict and counts -
|
|
157
|
+
no dependence on dict ordering, threads, or hash seeding."""
|
|
158
|
+
cols, rows = data
|
|
159
|
+
first = _run(cols, rows, cols, dict(rows))
|
|
160
|
+
second = _run(cols, rows, cols, dict(rows))
|
|
161
|
+
assert first.identical == second.identical
|
|
162
|
+
assert _kinds(first) == _kinds(second)
|
|
163
|
+
assert first.stats.rows_downloaded == second.stats.rows_downloaded
|
|
164
|
+
assert first.stats.queries == second.stats.queries
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
# ---------------------------------------------------------------------------
|
|
168
|
+
# The false-identical hunter: one minimal change must never read as identical.
|
|
169
|
+
# ---------------------------------------------------------------------------
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@settings(max_examples=deep_examples(600))
|
|
173
|
+
@given(data=_table(text=_SAFE, min_rows=1), seed=st.randoms(use_true_random=False), newval=_SAFE)
|
|
174
|
+
def test_a_single_planted_change_is_never_called_identical(data, seed, newval):
|
|
175
|
+
"""Take an identical pair, apply exactly one change - alter a cell, delete a
|
|
176
|
+
row, or insert a row - and assert the walk never reports identical and finds
|
|
177
|
+
exactly what a brute-force comparison would. This is the failure the whole
|
|
178
|
+
tool exists to prevent, concentrated onto the hardest case: one difference.
|
|
179
|
+
"""
|
|
180
|
+
cols, rows = data
|
|
181
|
+
rows_b = dict(rows)
|
|
182
|
+
keys = sorted(rows)
|
|
183
|
+
kind = seed.choice(["change", "delete", "insert"])
|
|
184
|
+
if kind == "change":
|
|
185
|
+
k = seed.choice(keys)
|
|
186
|
+
col = seed.randrange(len(cols))
|
|
187
|
+
v = list(rows_b[k])
|
|
188
|
+
v[col] = newval
|
|
189
|
+
rows_b[k] = tuple(v)
|
|
190
|
+
elif kind == "delete":
|
|
191
|
+
del rows_b[seed.choice(keys)]
|
|
192
|
+
else: # insert a key not already present
|
|
193
|
+
newk = seed.randint(-(10**12), 10**12)
|
|
194
|
+
assume(newk not in rows_b)
|
|
195
|
+
rows_b[newk] = tuple(newval for _ in cols)
|
|
196
|
+
|
|
197
|
+
assume(rows_b != rows) # a change that changed nothing is not a test
|
|
198
|
+
|
|
199
|
+
result = _run(cols, rows, cols, rows_b)
|
|
200
|
+
assert not result.identical, "a real difference was reported as identical"
|
|
201
|
+
assert _kinds(result) == _oracle(rows, rows_b)
|
|
202
|
+
assert result.stats.rows_downloaded > 0
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@settings(max_examples=300)
|
|
206
|
+
@given(data=_table(text=_SAFE, min_rows=1), seed=st.randoms(use_true_random=False))
|
|
207
|
+
def test_null_versus_value_on_one_cell_is_found(data, seed):
|
|
208
|
+
"""The migration bug class: a value on one side, absent (a different value)
|
|
209
|
+
on the other, in a single cell, must be caught - never smoothed to a match.
|
|
210
|
+
"""
|
|
211
|
+
cols, rows = data
|
|
212
|
+
k = seed.choice(sorted(rows))
|
|
213
|
+
col = seed.randrange(len(cols))
|
|
214
|
+
v = list(rows[k])
|
|
215
|
+
# Flip the cell to something guaranteed different from what is there.
|
|
216
|
+
v[col] = v[col] + "␀" if v[col] != "␀" else "x"
|
|
217
|
+
rows_b = dict(rows)
|
|
218
|
+
rows_b[k] = tuple(v)
|
|
219
|
+
result = _run(cols, rows, cols, rows_b)
|
|
220
|
+
assert not result.identical
|
|
221
|
+
assert _kinds(result) == [(k, "different")]
|
|
222
|
+
assert result.diffs[0].columns == [f"c{col}"]
|
|
@@ -0,0 +1,382 @@
|
|
|
1
|
+
"""The identical check across engines, over key types and values the core
|
|
2
|
+
integration suite does not reach.
|
|
3
|
+
|
|
4
|
+
`test_integration.py` proves identical tables match across PostgreSQL and DuckDB
|
|
5
|
+
with an *integer* key. This file widens the identical check where a cross-engine
|
|
6
|
+
abnormality would actually hide: the *hashed* key paths (text, composite, uuid),
|
|
7
|
+
which bucket by a hash of the key's rendered text, and edge-case values
|
|
8
|
+
(bigint extremes, non-finite floats, high-precision decimals, astral-plane
|
|
9
|
+
Unicode, an all-NULL row). Every table is built by a deterministic expression
|
|
10
|
+
spelled identically on both engines, so any reported difference is the tool
|
|
11
|
+
rendering the same value two ways - the exact false-positive the identical check
|
|
12
|
+
must never produce.
|
|
13
|
+
|
|
14
|
+
Skips cleanly without PostgreSQL.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
|
|
21
|
+
import pytest
|
|
22
|
+
from conftest import PG_SCHEMA, duckdb_write, open_duckdb, open_pg
|
|
23
|
+
|
|
24
|
+
from parity.engine import diff
|
|
25
|
+
|
|
26
|
+
pytestmark = pytest.mark.postgres
|
|
27
|
+
|
|
28
|
+
N = 5_000
|
|
29
|
+
|
|
30
|
+
# md5(i) agrees across engines, giving a deterministic text/uuid key. Each SELECT
|
|
31
|
+
# is portable SQL that produces byte-identical data on PostgreSQL and DuckDB.
|
|
32
|
+
TABLES = {
|
|
33
|
+
"text_key": (
|
|
34
|
+
"uid",
|
|
35
|
+
f"""
|
|
36
|
+
select md5(i::varchar) as uid,
|
|
37
|
+
(i % 97)::integer as customer_id,
|
|
38
|
+
((i * 7 % 100000) / 100.0)::decimal(12,2) as amount,
|
|
39
|
+
case when i % 13 = 0 then null
|
|
40
|
+
else 'n' || i::varchar end as note
|
|
41
|
+
from generate_series(1, {N}) as s(i)
|
|
42
|
+
""",
|
|
43
|
+
),
|
|
44
|
+
"composite_key": (
|
|
45
|
+
"grp,id",
|
|
46
|
+
f"""
|
|
47
|
+
select (i % 100)::integer as grp,
|
|
48
|
+
i::bigint as id,
|
|
49
|
+
((i * 3 % 50000) / 100.0)::decimal(12,2) as amount,
|
|
50
|
+
(i % 7 = 0) as flag
|
|
51
|
+
from generate_series(1, {N}) as s(i)
|
|
52
|
+
""",
|
|
53
|
+
),
|
|
54
|
+
"uuid_key": (
|
|
55
|
+
"uid",
|
|
56
|
+
f"""
|
|
57
|
+
select cast(
|
|
58
|
+
substr(md5(i::varchar), 1, 8) || '-' ||
|
|
59
|
+
substr(md5(i::varchar), 9, 4) || '-' ||
|
|
60
|
+
substr(md5(i::varchar), 13, 4) || '-' ||
|
|
61
|
+
substr(md5(i::varchar), 17, 4) || '-' ||
|
|
62
|
+
substr(md5(i::varchar), 21, 12) as uuid) as uid,
|
|
63
|
+
(i % 97)::integer as customer_id
|
|
64
|
+
from generate_series(1, {N}) as s(i)
|
|
65
|
+
""",
|
|
66
|
+
),
|
|
67
|
+
# One row per hostile value: bigint extremes, a non-finite float, a
|
|
68
|
+
# high-precision decimal, an astral-plane emoji, an empty string, and an
|
|
69
|
+
# all-NULL row. `{{f}}` is the engine's double type, filled at build time.
|
|
70
|
+
"edge_values": (
|
|
71
|
+
"id",
|
|
72
|
+
"""
|
|
73
|
+
select cast(1 as bigint) as id,
|
|
74
|
+
cast('9223372036854775807' as bigint) as big,
|
|
75
|
+
cast(123456789012.345678 as decimal(38,6)) as dec,
|
|
76
|
+
cast('Infinity' as {f}) as flt,
|
|
77
|
+
timestamp '2024-02-29 13:04:05.123456' as ts,
|
|
78
|
+
'añ日\U0001f600' as txt
|
|
79
|
+
union all select cast(2 as bigint), cast('-9223372036854775808' as bigint),
|
|
80
|
+
cast(-0.000001 as decimal(38,6)), cast('-Infinity' as {f}),
|
|
81
|
+
timestamp '2000-01-01 00:00:00', ''
|
|
82
|
+
union all select cast(3 as bigint), cast(0 as bigint),
|
|
83
|
+
cast(0 as decimal(38,6)), cast('NaN' as {f}),
|
|
84
|
+
timestamp '2099-12-31 23:59:59.999999', null
|
|
85
|
+
union all select cast(4 as bigint), null,
|
|
86
|
+
null, cast(1.5 as {f}), null, 'plain'
|
|
87
|
+
""",
|
|
88
|
+
),
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _build(cursor_exec, table_sql, float_type: str) -> None:
|
|
93
|
+
"""Create every table via the given execute callable (one engine).
|
|
94
|
+
|
|
95
|
+
`float_type` is that engine's double spelling - PostgreSQL wants
|
|
96
|
+
``double precision`` where DuckDB wants ``double`` - filled into the edge
|
|
97
|
+
table's non-finite-float columns.
|
|
98
|
+
"""
|
|
99
|
+
for name, (_key, select) in TABLES.items():
|
|
100
|
+
cursor_exec(f"create table {table_sql(name)} as {select.format(f=float_type)}")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@pytest.fixture(scope="module")
|
|
104
|
+
def duck(tmp_path_factory):
|
|
105
|
+
"""Side B: all tables in one DuckDB file, opened read-only."""
|
|
106
|
+
path = str(tmp_path_factory.mktemp("ident_live") / "b.duckdb")
|
|
107
|
+
con = duckdb_write(path)
|
|
108
|
+
try:
|
|
109
|
+
_build(con.execute, lambda n: f"main.{n}", "double")
|
|
110
|
+
finally:
|
|
111
|
+
con.close()
|
|
112
|
+
d = open_duckdb(path, side="B")
|
|
113
|
+
yield d
|
|
114
|
+
d.close()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
SCHEMA = f"{PG_SCHEMA}_identlive"
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@pytest.fixture(scope="module")
|
|
121
|
+
def pg(pg_url):
|
|
122
|
+
"""Side A: all tables in a schema the test owns, opened read-only."""
|
|
123
|
+
import psycopg
|
|
124
|
+
|
|
125
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
126
|
+
try:
|
|
127
|
+
con.execute(f"drop schema if exists {SCHEMA} cascade")
|
|
128
|
+
con.execute(f"create schema {SCHEMA}")
|
|
129
|
+
_build(con.execute, lambda n: f"{SCHEMA}.{n}", "double precision")
|
|
130
|
+
finally:
|
|
131
|
+
con.close()
|
|
132
|
+
d = open_pg(pg_url, side="A")
|
|
133
|
+
yield d
|
|
134
|
+
d.close()
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@pytest.mark.parametrize("name", list(TABLES))
|
|
138
|
+
def test_identical_across_engines_by_key_type(pg, duck, name):
|
|
139
|
+
"""The same table, same data, one hashed key type at a time: identical,
|
|
140
|
+
and not one row moved across the network."""
|
|
141
|
+
key = TABLES[name][0].split(",") # a list, so composite keys resolve
|
|
142
|
+
result = diff(pg, duck, f"{SCHEMA}.{name}", f"main.{name}", key)
|
|
143
|
+
assert result.identical, (
|
|
144
|
+
f"{name}: identical data reported {len(result.diffs)} diffs "
|
|
145
|
+
f"{[(d.key, d.kind, d.columns) for d in result.diffs[:3]]}"
|
|
146
|
+
)
|
|
147
|
+
assert result.stats.rows_downloaded == 0
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def test_a_representation_change_across_engines_reads_identical(pg_url, tmp_path):
|
|
151
|
+
"""The migration scenario: the same values stored with *different declared
|
|
152
|
+
types* on each engine - integer vs bigint, numeric(12,2) vs double,
|
|
153
|
+
varchar vs text - must still read identical end to end. Storing a value a
|
|
154
|
+
different way is not changing it; the diff must not mistake it for one.
|
|
155
|
+
"""
|
|
156
|
+
import psycopg
|
|
157
|
+
|
|
158
|
+
n = 2_000
|
|
159
|
+
schema = f"{PG_SCHEMA}_repr"
|
|
160
|
+
pg_sql = f"""
|
|
161
|
+
create table {schema}.repr as
|
|
162
|
+
select i::bigint as id,
|
|
163
|
+
(i / 100.0)::numeric(12,2) as amount,
|
|
164
|
+
('r' || i::text)::varchar(50) as name,
|
|
165
|
+
(timestamp '2024-01-01 00:00:00'
|
|
166
|
+
+ (i % 86400) * interval '1 second') as ts
|
|
167
|
+
from generate_series(1, {n}) as s(i)
|
|
168
|
+
"""
|
|
169
|
+
duck_sql = f"""
|
|
170
|
+
create table repr as
|
|
171
|
+
select i::integer as id,
|
|
172
|
+
(i / 100.0)::double as amount,
|
|
173
|
+
('r' || i::varchar) as name,
|
|
174
|
+
(timestamp '2024-01-01 00:00:00'
|
|
175
|
+
+ (i % 86400) * interval '1 second') as ts
|
|
176
|
+
from generate_series(1, {n}) as s(i)
|
|
177
|
+
"""
|
|
178
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
179
|
+
try:
|
|
180
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
181
|
+
con.execute(f"create schema {schema}")
|
|
182
|
+
con.execute(pg_sql)
|
|
183
|
+
finally:
|
|
184
|
+
con.close()
|
|
185
|
+
path = str(tmp_path / "repr.duckdb")
|
|
186
|
+
dcon = duckdb_write(path)
|
|
187
|
+
try:
|
|
188
|
+
dcon.execute(duck_sql)
|
|
189
|
+
finally:
|
|
190
|
+
dcon.close()
|
|
191
|
+
|
|
192
|
+
a = open_pg(pg_url, side="A")
|
|
193
|
+
b = open_duckdb(path, side="B")
|
|
194
|
+
try:
|
|
195
|
+
result = diff(a, b, f"{schema}.repr", "main.repr", "id")
|
|
196
|
+
finally:
|
|
197
|
+
a.close()
|
|
198
|
+
b.close()
|
|
199
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
200
|
+
try:
|
|
201
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
202
|
+
finally:
|
|
203
|
+
con.close()
|
|
204
|
+
|
|
205
|
+
assert result.identical, (
|
|
206
|
+
"a pure representation change was reported as a data difference: "
|
|
207
|
+
f"{[(d.key, d.columns) for d in result.diffs[:5]]}"
|
|
208
|
+
)
|
|
209
|
+
assert result.stats.rows_downloaded == 0
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def test_a_hashed_key_table_still_finds_a_planted_difference(pg, tmp_path):
|
|
213
|
+
"""The other half of the promise: the identical check on a hashed key must
|
|
214
|
+
never MISS a real difference either. Plant one changed row in a text-keyed
|
|
215
|
+
copy and confirm the walk reports exactly it, keyed by the real text
|
|
216
|
+
identity - never by the 60-bit bucket hash, which could collide.
|
|
217
|
+
"""
|
|
218
|
+
path = str(tmp_path / "text_key_perturbed.duckdb")
|
|
219
|
+
con = duckdb_write(path)
|
|
220
|
+
try:
|
|
221
|
+
con.execute(f"create table text_key as {TABLES['text_key'][1]}")
|
|
222
|
+
# md5('1234') is the uid of the row generated for i = 1234.
|
|
223
|
+
con.execute("update text_key set amount = amount + 0.01 where uid = md5('1234')")
|
|
224
|
+
finally:
|
|
225
|
+
con.close()
|
|
226
|
+
|
|
227
|
+
b = open_duckdb(path, side="B")
|
|
228
|
+
try:
|
|
229
|
+
result = diff(pg, b, f"{SCHEMA}.text_key", "main.text_key", "uid")
|
|
230
|
+
finally:
|
|
231
|
+
b.close()
|
|
232
|
+
|
|
233
|
+
# The same md5 the engines compute, for cross-engine agreement not security.
|
|
234
|
+
uid = hashlib.md5(b"1234", usedforsecurity=False).hexdigest()
|
|
235
|
+
assert [(d.key, d.kind) for d in result.diffs] == [(uid, "different")]
|
|
236
|
+
assert result.diffs[0].columns == ["amount"]
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def test_a_very_wide_table_reads_identical_across_engines(pg_url, tmp_path):
|
|
240
|
+
"""A table with more columns than PostgreSQL's 100-argument `concat_ws`
|
|
241
|
+
limit forces `row_text` to build a nested tree of `concat_ws` calls. The two
|
|
242
|
+
engines must build the *same* tree over the same values, or an identical
|
|
243
|
+
wide table - an ordinary denormalised fact table - would report every row as
|
|
244
|
+
different. Also plants one change to prove the wide path finds a real one.
|
|
245
|
+
"""
|
|
246
|
+
import psycopg
|
|
247
|
+
|
|
248
|
+
ncols = 120 # over PostgreSQL's 99-argument concat_ws limit
|
|
249
|
+
cols = ", ".join(f"('v' || (i + {n})::varchar) as c{n}" for n in range(ncols))
|
|
250
|
+
select = f"select i::bigint as id, {cols} from generate_series(1, 500) as s(i)"
|
|
251
|
+
schema = f"{PG_SCHEMA}_wide"
|
|
252
|
+
|
|
253
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
254
|
+
try:
|
|
255
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
256
|
+
con.execute(f"create schema {schema}")
|
|
257
|
+
con.execute(f"create table {schema}.wide as {select}")
|
|
258
|
+
finally:
|
|
259
|
+
con.close()
|
|
260
|
+
path = str(tmp_path / "wide.duckdb")
|
|
261
|
+
dcon = duckdb_write(path)
|
|
262
|
+
try:
|
|
263
|
+
dcon.execute(f"create table wide as {select}")
|
|
264
|
+
dcon.execute("create table wide_p as select * from wide")
|
|
265
|
+
dcon.execute("update wide_p set c50 = 'CHANGED' where id = 321")
|
|
266
|
+
finally:
|
|
267
|
+
dcon.close()
|
|
268
|
+
|
|
269
|
+
a = open_pg(pg_url, side="A")
|
|
270
|
+
b = open_duckdb(path, side="B")
|
|
271
|
+
try:
|
|
272
|
+
same = diff(a, b, f"{schema}.wide", "main.wide", "id")
|
|
273
|
+
changed = diff(a, b, f"{schema}.wide", "main.wide_p", "id")
|
|
274
|
+
finally:
|
|
275
|
+
a.close()
|
|
276
|
+
b.close()
|
|
277
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
278
|
+
try:
|
|
279
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
280
|
+
finally:
|
|
281
|
+
con.close()
|
|
282
|
+
|
|
283
|
+
assert same.identical, (
|
|
284
|
+
"a 120-column identical table reported differences - the nested "
|
|
285
|
+
f"concat_ws trees disagree: {[(d.key, d.columns) for d in same.diffs[:3]]}"
|
|
286
|
+
)
|
|
287
|
+
assert same.stats.rows_downloaded == 0
|
|
288
|
+
assert [(d.key, d.kind) for d in changed.diffs] == [(321, "different")]
|
|
289
|
+
assert changed.diffs[0].columns == ["c50"]
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def test_a_same_content_swap_in_one_bucket_is_caught_live(pg_url, tmp_path):
|
|
293
|
+
"""The exact 0.2.1 bug, reproduced across two real engines.
|
|
294
|
+
|
|
295
|
+
Every row carries identical non-key content, and one side holds key 2500
|
|
296
|
+
while the other holds the adjacent key 2501 - a delete and an insert of the
|
|
297
|
+
same content in the same bucket. Before the key was folded into each
|
|
298
|
+
bucket's checksum the counts balanced and the content sums matched, so both
|
|
299
|
+
rows vanished: a false "identical". They must both be found.
|
|
300
|
+
"""
|
|
301
|
+
import psycopg
|
|
302
|
+
|
|
303
|
+
# Shared keys are 1..5000 except {2500, 2501}; PostgreSQL adds 2500, DuckDB
|
|
304
|
+
# adds 2501. Identical content everywhere, so only the key distinguishes them.
|
|
305
|
+
body = "1.00::decimal(12,2) as amount, 'x' as status"
|
|
306
|
+
pg_select = f"select i::bigint as id, {body} from generate_series(1,5000) as s(i) where i <> 2501"
|
|
307
|
+
duck_select = f"select i::bigint as id, {body} from generate_series(1,5000) as s(i) where i <> 2500"
|
|
308
|
+
schema = f"{PG_SCHEMA}_swap"
|
|
309
|
+
|
|
310
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
311
|
+
try:
|
|
312
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
313
|
+
con.execute(f"create schema {schema}")
|
|
314
|
+
con.execute(f"create table {schema}.swap as {pg_select}")
|
|
315
|
+
finally:
|
|
316
|
+
con.close()
|
|
317
|
+
path = str(tmp_path / "swap.duckdb")
|
|
318
|
+
dcon = duckdb_write(path)
|
|
319
|
+
try:
|
|
320
|
+
dcon.execute(f"create table swap as {duck_select}")
|
|
321
|
+
finally:
|
|
322
|
+
dcon.close()
|
|
323
|
+
|
|
324
|
+
a = open_pg(pg_url, side="A")
|
|
325
|
+
b = open_duckdb(path, side="B")
|
|
326
|
+
try:
|
|
327
|
+
result = diff(a, b, f"{schema}.swap", "main.swap", "id")
|
|
328
|
+
finally:
|
|
329
|
+
a.close()
|
|
330
|
+
b.close()
|
|
331
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
332
|
+
try:
|
|
333
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
334
|
+
finally:
|
|
335
|
+
con.close()
|
|
336
|
+
|
|
337
|
+
assert not result.identical, "a same-content insert+delete cancelled to a false match"
|
|
338
|
+
assert [(d.key, d.kind) for d in result.diffs] == [
|
|
339
|
+
(2500, "only_in_a"),
|
|
340
|
+
(2501, "only_in_b"),
|
|
341
|
+
]
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
@pytest.mark.parametrize("rows,queries", [(0, 2), (1, 4)])
|
|
345
|
+
def test_degenerate_identical_tables_across_engines(pg_url, tmp_path, rows, queries):
|
|
346
|
+
"""The degenerate sizes where off-by-ones hide: an empty table costs two
|
|
347
|
+
queries (key_stats only, nothing to bisect) and a one-row table costs four
|
|
348
|
+
(2 key_stats + 2 checksums), both identical and both moving zero rows."""
|
|
349
|
+
import psycopg
|
|
350
|
+
|
|
351
|
+
select = f"select i::bigint as id, ('v' || i::text) as note from generate_series(1, {rows}) as s(i)"
|
|
352
|
+
schema = f"{PG_SCHEMA}_degen"
|
|
353
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
354
|
+
try:
|
|
355
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
356
|
+
con.execute(f"create schema {schema}")
|
|
357
|
+
con.execute(f"create table {schema}.t as {select}")
|
|
358
|
+
finally:
|
|
359
|
+
con.close()
|
|
360
|
+
path = str(tmp_path / "degen.duckdb")
|
|
361
|
+
dcon = duckdb_write(path)
|
|
362
|
+
try:
|
|
363
|
+
dcon.execute(f"create table t as {select}")
|
|
364
|
+
finally:
|
|
365
|
+
dcon.close()
|
|
366
|
+
|
|
367
|
+
a = open_pg(pg_url, side="A")
|
|
368
|
+
b = open_duckdb(path, side="B")
|
|
369
|
+
try:
|
|
370
|
+
result = diff(a, b, f"{schema}.t", "main.t", "id")
|
|
371
|
+
finally:
|
|
372
|
+
a.close()
|
|
373
|
+
b.close()
|
|
374
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
375
|
+
try:
|
|
376
|
+
con.execute(f"drop schema if exists {schema} cascade")
|
|
377
|
+
finally:
|
|
378
|
+
con.close()
|
|
379
|
+
|
|
380
|
+
assert result.identical
|
|
381
|
+
assert result.stats.rows_downloaded == 0
|
|
382
|
+
assert result.stats.queries == queries
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""The identical check between the two most common migration endpoints.
|
|
2
|
+
|
|
3
|
+
MySQL and PostgreSQL are each verified against DuckDB elsewhere, so byte-equal
|
|
4
|
+
canonical text makes them agree with each other by transitivity - but a
|
|
5
|
+
MySQL -> PostgreSQL migration is common enough that the pair deserves a direct
|
|
6
|
+
test, with neither side being the reference engine. Identical data built by
|
|
7
|
+
each engine's own dialect must read identical; a single planted change must be
|
|
8
|
+
found exactly.
|
|
9
|
+
|
|
10
|
+
Skips unless both a MySQL and a PostgreSQL server are reachable.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import pytest
|
|
16
|
+
from conftest import PG_SCHEMA, open_mysql, open_pg
|
|
17
|
+
|
|
18
|
+
from parity.engine import diff
|
|
19
|
+
|
|
20
|
+
pytestmark = [pytest.mark.postgres, pytest.mark.mysql]
|
|
21
|
+
|
|
22
|
+
N = 2_000
|
|
23
|
+
SCHEMA = f"{PG_SCHEMA}_mypg"
|
|
24
|
+
|
|
25
|
+
# The same rows, spelled in each engine's own SQL. `flag` is a plain int on both
|
|
26
|
+
# (not a boolean) so this tests data agreement, not the boolean-vs-int trap.
|
|
27
|
+
PG_BUILD = f"""
|
|
28
|
+
create table {SCHEMA}.xeng as
|
|
29
|
+
select i::bigint as id,
|
|
30
|
+
((i * 7 % 100000) / 100.0)::decimal(12,2) as amount,
|
|
31
|
+
(case when i % 3 = 0 then 'paid'
|
|
32
|
+
when i % 3 = 1 then 'open' else 'void' end)::varchar(10) as status,
|
|
33
|
+
(i % 11 = 0)::int as flag,
|
|
34
|
+
(case when i % 13 = 0 then null
|
|
35
|
+
else 'n' || i::text end)::varchar(20) as note
|
|
36
|
+
from generate_series(1, {N}) as s(i)
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
MYSQL_BUILD = f"""
|
|
40
|
+
create table xeng as
|
|
41
|
+
with recursive seq(i) as (
|
|
42
|
+
select 1 union all select i + 1 from seq where i < {N}
|
|
43
|
+
)
|
|
44
|
+
select cast(i as signed) as id,
|
|
45
|
+
cast((i * 7 mod 100000) / 100.0 as decimal(12,2)) as amount,
|
|
46
|
+
(case when i mod 3 = 0 then 'paid'
|
|
47
|
+
when i mod 3 = 1 then 'open' else 'void' end) as status,
|
|
48
|
+
cast(i mod 11 = 0 as signed) as flag,
|
|
49
|
+
(case when i mod 13 = 0 then null
|
|
50
|
+
else concat('n', i) end) as note
|
|
51
|
+
from seq
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@pytest.fixture(scope="module")
|
|
56
|
+
def pg(pg_url):
|
|
57
|
+
"""Side A: PostgreSQL, its own schema, opened read-only."""
|
|
58
|
+
import psycopg
|
|
59
|
+
|
|
60
|
+
con = psycopg.connect(pg_url, autocommit=True)
|
|
61
|
+
try:
|
|
62
|
+
con.execute(f"drop schema if exists {SCHEMA} cascade")
|
|
63
|
+
con.execute(f"create schema {SCHEMA}")
|
|
64
|
+
con.execute(PG_BUILD)
|
|
65
|
+
finally:
|
|
66
|
+
con.close()
|
|
67
|
+
d = open_pg(pg_url, side="A")
|
|
68
|
+
yield d
|
|
69
|
+
d.close()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@pytest.fixture(scope="module")
|
|
73
|
+
def my(mysql_url):
|
|
74
|
+
"""Side B: MySQL, identical data, opened read-only."""
|
|
75
|
+
d = open_mysql(mysql_url, side="B")
|
|
76
|
+
cur = d._conn.cursor()
|
|
77
|
+
try:
|
|
78
|
+
cur.execute("set session cte_max_recursion_depth = 1000000")
|
|
79
|
+
cur.execute("drop table if exists xeng")
|
|
80
|
+
cur.execute("drop table if exists xeng_p")
|
|
81
|
+
cur.execute(MYSQL_BUILD)
|
|
82
|
+
# A perturbed copy for the planted-difference test, built once.
|
|
83
|
+
cur.execute("create table xeng_p as select * from xeng")
|
|
84
|
+
cur.execute("update xeng_p set amount = amount + 0.01 where id = 1234")
|
|
85
|
+
d._conn.commit()
|
|
86
|
+
finally:
|
|
87
|
+
cur.close()
|
|
88
|
+
yield d
|
|
89
|
+
d.close()
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_mysql_and_postgres_agree_on_identical_data(pg, my):
|
|
93
|
+
"""The same rows on each engine read identical, with zero download."""
|
|
94
|
+
result = diff(pg, my, f"{SCHEMA}.xeng", "xeng", "id")
|
|
95
|
+
assert result.identical, (
|
|
96
|
+
"MySQL and PostgreSQL disagreed on identical data: "
|
|
97
|
+
f"{[(d.key, d.columns, d.values_a, d.values_b) for d in result.diffs[:5]]}"
|
|
98
|
+
)
|
|
99
|
+
assert result.stats.rows_downloaded == 0
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def test_a_planted_change_between_mysql_and_postgres_is_found(pg, my):
|
|
103
|
+
"""One changed decimal is reported as exactly that row and column."""
|
|
104
|
+
result = diff(pg, my, f"{SCHEMA}.xeng", "xeng_p", "id")
|
|
105
|
+
assert [(d.key, d.kind) for d in result.diffs] == [(1234, "different")]
|
|
106
|
+
assert result.diffs[0].columns == ["amount"]
|
|
@@ -22,6 +22,7 @@ import pytest
|
|
|
22
22
|
# dependency must not look like a broken build.
|
|
23
23
|
pytest.importorskip("hypothesis")
|
|
24
24
|
|
|
25
|
+
from conftest import deep_examples
|
|
25
26
|
from fakes import DictTable, FakeDialect
|
|
26
27
|
from hypothesis import given, settings
|
|
27
28
|
from hypothesis import strategies as st
|
|
@@ -81,7 +82,7 @@ def _kinds(result) -> list[tuple[int, str]]:
|
|
|
81
82
|
# ---------------------------------------------------------------------------
|
|
82
83
|
|
|
83
84
|
|
|
84
|
-
@settings(max_examples=400)
|
|
85
|
+
@settings(max_examples=deep_examples(400))
|
|
85
86
|
@given(a=_table, b=_table)
|
|
86
87
|
def test_diff_matches_a_brute_force_oracle(a, b):
|
|
87
88
|
"""For any two tables, parity's result equals comparing every row directly.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|