parity-diff 0.2.1__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parity_diff-0.2.1 → parity_diff-0.2.3}/CONTRIBUTING.md +41 -0
- {parity_diff-0.2.1/src/parity_diff.egg-info → parity_diff-0.2.3}/PKG-INFO +11 -3
- {parity_diff-0.2.1 → parity_diff-0.2.3}/README.md +7 -2
- {parity_diff-0.2.1 → parity_diff-0.2.3}/pyproject.toml +8 -2
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/__init__.py +1 -1
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/base.py +5 -1
- parity_diff-0.2.3/src/parity/dialects/snowflake_dialect.py +231 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/engine.py +95 -38
- {parity_diff-0.2.1 → parity_diff-0.2.3/src/parity_diff.egg-info}/PKG-INFO +11 -3
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/SOURCES.txt +7 -1
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/requires.txt +4 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/conftest.py +49 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_engine.py +51 -0
- parity_diff-0.2.3/tests/test_identical.py +222 -0
- parity_diff-0.2.3/tests/test_identical_live.py +382 -0
- parity_diff-0.2.3/tests/test_mysql_postgres.py +106 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_properties.py +2 -1
- parity_diff-0.2.3/tests/test_snowflake.py +181 -0
- parity_diff-0.2.3/tests/test_snowflake_offline.py +113 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/LICENSE +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/MANIFEST.in +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/setup.cfg +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/cli.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/__init__.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/duckdb_dialect.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/mysql_dialect.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/postgres_dialect.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/types.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/dependency_links.txt +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/entry_points.txt +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/top_level.txt +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/fakes.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_cli.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_encoding.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_fuzz_encoding.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_integration.py +0 -0
- {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_mysql.py +0 -0
|
@@ -97,6 +97,34 @@ MySQL, added after v0.1.0, turned up two more that a warehouse dialect may hit:
|
|
|
97
97
|
sentinel silently became `N`. Build such bytes from `CHAR`/hex, not a
|
|
98
98
|
literal, and verify by hashing rather than by reading the SQL.
|
|
99
99
|
|
|
100
|
+
Snowflake, the first warehouse (v0.2.2), added three more — the sort a warehouse
|
|
101
|
+
is especially likely to spring:
|
|
102
|
+
|
|
103
|
+
9. **Your engine may have neither a bit-cast nor `CONV` to reach 60 bits.**
|
|
104
|
+
Snowflake had no `bit(60)::bigint` and no `conv(hex,16,10)`. What it did have
|
|
105
|
+
is `md5_number_upper64(x)`, the top 64 bits of the digest as a number, and
|
|
106
|
+
`floor(that / 16)` drops the low 4 to land on the same 60-bit prefix - the
|
|
107
|
+
fourth distinct path to `648541476951500027`. Find your engine's own route;
|
|
108
|
+
the constant is the contract, not the SQL that reaches it.
|
|
109
|
+
|
|
110
|
+
10. **Integers and decimals may share one type name.** Snowflake reports both as
|
|
111
|
+
`NUMBER` and only `numeric_scale` (0 = integer) tells them apart, so its
|
|
112
|
+
`columns()` reads the scale instead of trusting `data_type`. Trust the type
|
|
113
|
+
name and an integer key renders as `42.000000` and never matches another
|
|
114
|
+
engine's `42`. If your engine collapses numeric types like this, override
|
|
115
|
+
`columns()`.
|
|
116
|
+
|
|
117
|
+
11. **Identifier case-folding is the engine's, and it is not universal.**
|
|
118
|
+
Snowflake upper-cases unquoted identifiers where PostgreSQL and DuckDB
|
|
119
|
+
lower-case them. The engine matches keys and columns case-insensitively for
|
|
120
|
+
exactly this reason (`_fold_columns`), but the *table* name is looked up in
|
|
121
|
+
the case the engine stored, so `--a-table orders` against a Snowflake
|
|
122
|
+
`ORDERS` fails with a near-miss hint. And some warehouses (Snowflake among
|
|
123
|
+
them) offer only READ COMMITTED, so unlike PostgreSQL the walk cannot be
|
|
124
|
+
pinned to one snapshot - a real limitation to document, not hide. None of
|
|
125
|
+
these three showed up in the docs; they surfaced only against a live
|
|
126
|
+
account, which is why the rule below is not negotiable.
|
|
127
|
+
|
|
100
128
|
### Proving it works
|
|
101
129
|
|
|
102
130
|
A dialect is not done until `tests/test_encoding.py` passes against it. That
|
|
@@ -136,6 +164,19 @@ reason next to it rather than being switched off globally.
|
|
|
136
164
|
PostgreSQL-backed tests read `PARITY_TEST_PG` and skip cleanly when nothing is
|
|
137
165
|
listening, so the suite is useful with only DuckDB installed.
|
|
138
166
|
|
|
167
|
+
The generative suites hunt for the one failure that matters most — a real
|
|
168
|
+
difference reported as identical. To run that hunt deeper (many more generated
|
|
169
|
+
cases per property, at the cost of time), scale it with `PARITY_DEEP`:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
PARITY_DEEP=20 pytest tests/test_identical.py tests/test_properties.py
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
The default of `1` keeps the everyday suite fast; the deep run is the repeatable
|
|
176
|
+
"make sure there is no abnormality" check on the identical guarantee. A dozen
|
|
177
|
+
seeds at `PARITY_DEEP=6` have turned up nothing — but the point is that anyone
|
|
178
|
+
can re-run it.
|
|
179
|
+
|
|
139
180
|
## Scope
|
|
140
181
|
|
|
141
182
|
Before proposing a feature, check it against the question the tool exists to
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parity-diff
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
|
|
5
5
|
Author: Alessio Sorio
|
|
6
6
|
License-Expression: MIT
|
|
@@ -30,10 +30,13 @@ Provides-Extra: postgres
|
|
|
30
30
|
Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
|
|
31
31
|
Provides-Extra: mysql
|
|
32
32
|
Requires-Dist: mysql-connector-python>=8.0; extra == "mysql"
|
|
33
|
+
Provides-Extra: snowflake
|
|
34
|
+
Requires-Dist: snowflake-connector-python>=3.0; extra == "snowflake"
|
|
33
35
|
Provides-Extra: all
|
|
34
36
|
Requires-Dist: duckdb>=1.0; extra == "all"
|
|
35
37
|
Requires-Dist: psycopg[binary]>=3.1; extra == "all"
|
|
36
38
|
Requires-Dist: mysql-connector-python>=8.0; extra == "all"
|
|
39
|
+
Requires-Dist: snowflake-connector-python>=3.0; extra == "all"
|
|
37
40
|
Provides-Extra: test
|
|
38
41
|
Requires-Dist: hypothesis>=6.0; extra == "test"
|
|
39
42
|
Dynamic: license-file
|
|
@@ -127,6 +130,7 @@ pip install "parity-diff[all]" # every engine
|
|
|
127
130
|
pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
|
|
128
131
|
pip install "parity-diff[postgres]" # PostgreSQL only
|
|
129
132
|
pip install "parity-diff[mysql]" # MySQL only
|
|
133
|
+
pip install "parity-diff[snowflake]" # Snowflake only
|
|
130
134
|
```
|
|
131
135
|
|
|
132
136
|
> The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
|
|
@@ -288,7 +292,8 @@ stated plainly rather than buried.
|
|
|
288
292
|
| PostgreSQL | supported (tested against 16 and 18) |
|
|
289
293
|
| DuckDB | supported (tested against 1.5) |
|
|
290
294
|
| MySQL | supported (tested against 8.0) |
|
|
291
|
-
| Snowflake
|
|
295
|
+
| Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
|
|
296
|
+
| BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
|
|
292
297
|
|
|
293
298
|
## Scope
|
|
294
299
|
|
|
@@ -336,7 +341,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
|
|
|
336
341
|
|
|
337
342
|
## Changelog
|
|
338
343
|
|
|
339
|
-
|
|
344
|
+
Latest release **0.2.1** — a generative test suite (property-based, oracle, and
|
|
345
|
+
cross-engine fuzzing) and the correctness fix it surfaced: a same-content
|
|
346
|
+
insert-plus-delete falling in one bucket that could previously read as
|
|
347
|
+
identical. See `CHANGELOG.md` for the full history.
|
|
340
348
|
|
|
341
349
|
## License
|
|
342
350
|
|
|
@@ -87,6 +87,7 @@ pip install "parity-diff[all]" # every engine
|
|
|
87
87
|
pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
|
|
88
88
|
pip install "parity-diff[postgres]" # PostgreSQL only
|
|
89
89
|
pip install "parity-diff[mysql]" # MySQL only
|
|
90
|
+
pip install "parity-diff[snowflake]" # Snowflake only
|
|
90
91
|
```
|
|
91
92
|
|
|
92
93
|
> The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
|
|
@@ -248,7 +249,8 @@ stated plainly rather than buried.
|
|
|
248
249
|
| PostgreSQL | supported (tested against 16 and 18) |
|
|
249
250
|
| DuckDB | supported (tested against 1.5) |
|
|
250
251
|
| MySQL | supported (tested against 8.0) |
|
|
251
|
-
| Snowflake
|
|
252
|
+
| Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
|
|
253
|
+
| BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
|
|
252
254
|
|
|
253
255
|
## Scope
|
|
254
256
|
|
|
@@ -296,7 +298,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
|
|
|
296
298
|
|
|
297
299
|
## Changelog
|
|
298
300
|
|
|
299
|
-
|
|
301
|
+
Latest release **0.2.1** — a generative test suite (property-based, oracle, and
|
|
302
|
+
cross-engine fuzzing) and the correctness fix it surfaced: a same-content
|
|
303
|
+
insert-plus-delete falling in one bucket that could previously read as
|
|
304
|
+
identical. See `CHANGELOG.md` for the full history.
|
|
300
305
|
|
|
301
306
|
## License
|
|
302
307
|
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
# PyPI by an empty project. The import name, the CLI command and the repo are
|
|
4
4
|
# all still `parity`: `pip install parity-diff` gives you `parity ...`.
|
|
5
5
|
name = "parity-diff"
|
|
6
|
-
version = "0.2.
|
|
6
|
+
version = "0.2.3"
|
|
7
7
|
description = "Prove two tables in two different database engines hold the same data - without moving the data out of either engine."
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
license = "MIT"
|
|
@@ -37,7 +37,11 @@ dependencies = []
|
|
|
37
37
|
duckdb = ["duckdb>=1.0"]
|
|
38
38
|
postgres = ["psycopg[binary]>=3.1"]
|
|
39
39
|
mysql = ["mysql-connector-python>=8.0"]
|
|
40
|
-
|
|
40
|
+
snowflake = ["snowflake-connector-python>=3.0"]
|
|
41
|
+
all = [
|
|
42
|
+
"duckdb>=1.0", "psycopg[binary]>=3.1",
|
|
43
|
+
"mysql-connector-python>=8.0", "snowflake-connector-python>=3.0",
|
|
44
|
+
]
|
|
41
45
|
# Test-only dependencies, kept out of every runtime extra. Hypothesis drives
|
|
42
46
|
# the generative suite (tests/test_properties.py); the core stays stdlib-only.
|
|
43
47
|
test = ["hypothesis>=6.0"]
|
|
@@ -64,6 +68,7 @@ markers = [
|
|
|
64
68
|
"postgres: requires a reachable PostgreSQL server",
|
|
65
69
|
"mysql: requires a reachable MySQL server",
|
|
66
70
|
"duckdb: requires the duckdb driver",
|
|
71
|
+
"snowflake: requires a reachable Snowflake account (PARITY_TEST_SNOWFLAKE)",
|
|
67
72
|
]
|
|
68
73
|
|
|
69
74
|
[tool.ruff]
|
|
@@ -105,6 +110,7 @@ ignore = [
|
|
|
105
110
|
"src/parity/dialects/base.py" = ["S608"]
|
|
106
111
|
"src/parity/dialects/postgres_dialect.py" = ["S608"]
|
|
107
112
|
"src/parity/dialects/mysql_dialect.py" = ["S608"]
|
|
113
|
+
"src/parity/dialects/snowflake_dialect.py" = ["S608"]
|
|
108
114
|
# proof.py is kept close to the original author's script so it stays readable
|
|
109
115
|
# next to the findings it produced; its terse one-line style is deliberate.
|
|
110
116
|
"demo/proof.py" = ["E401", "E402", "E701", "E702", "B007", "S311", "S608"]
|
|
@@ -551,10 +551,14 @@ def get_dialect(
|
|
|
551
551
|
from parity.dialects.mysql_dialect import MySQLDialect
|
|
552
552
|
|
|
553
553
|
dialect = MySQLDialect(float_scale=float_scale, side=side)
|
|
554
|
+
elif scheme in ("snowflake",):
|
|
555
|
+
from parity.dialects.snowflake_dialect import SnowflakeDialect
|
|
556
|
+
|
|
557
|
+
dialect = SnowflakeDialect(float_scale=float_scale, side=side)
|
|
554
558
|
else:
|
|
555
559
|
raise ValueError(
|
|
556
560
|
f"[side {side}] no dialect for scheme {scheme!r}. "
|
|
557
|
-
f"Supported: duckdb, postgres, mysql."
|
|
561
|
+
f"Supported: duckdb, postgres, mysql, snowflake."
|
|
558
562
|
)
|
|
559
563
|
try:
|
|
560
564
|
dialect.connect(connection_string)
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Snowflake dialect.
|
|
2
|
+
|
|
3
|
+
Verified against a live Snowflake account (AWS eu-central-2, 2026-09-07): the
|
|
4
|
+
hash constant agrees, a full-table checksum matches DuckDB byte-for-byte over
|
|
5
|
+
5,000 mixed-type rows, and `tests/test_snowflake.py` passes end to end - the
|
|
6
|
+
identical check, a changed decimal, a deleted row, and the NULL-versus-empty
|
|
7
|
+
trap. It was first written against Snowflake's documentation; the live run then
|
|
8
|
+
turned up the one thing docs could not: Snowflake upper-cases unquoted
|
|
9
|
+
identifiers, so the engine had to match keys and columns case-insensitively
|
|
10
|
+
(see `_fold_columns` in engine.py) for a Snowflake table to diff against a
|
|
11
|
+
lower-casing engine at all.
|
|
12
|
+
|
|
13
|
+
The Snowflake-specific decisions, each confirmed by that run:
|
|
14
|
+
|
|
15
|
+
- **No `CONV` and no bit-cast for the hash.** Snowflake has neither, but
|
|
16
|
+
`MD5_NUMBER_UPPER64(x)` returns the top 64 bits of the digest as an unsigned
|
|
17
|
+
number, and the top 60 bits - the first 15 hex characters the other engines
|
|
18
|
+
fold - are `FLOOR(that / 16)`. That should equal 648541476951500027 for
|
|
19
|
+
`'abc'`; a test must pin it.
|
|
20
|
+
- **Integer and decimal both report as `NUMBER`.** `information_schema` tells
|
|
21
|
+
them apart only by `numeric_scale` (0 = integer), so `columns()` reads the
|
|
22
|
+
scale rather than trusting `data_type` - otherwise an integer key would be
|
|
23
|
+
rendered as `42.000000` and never match another engine's `42`.
|
|
24
|
+
- **The NULL sentinel is built from `CHR(92)`.** Snowflake interprets
|
|
25
|
+
backslash escapes in string literals, so a literal `'\\N'` is unsafe the same
|
|
26
|
+
way it was on MySQL.
|
|
27
|
+
- **One-snapshot isolation is not available.** Snowflake offers only READ
|
|
28
|
+
COMMITTED, so unlike PostgreSQL the walk cannot be pinned to a single
|
|
29
|
+
snapshot; a source table mutating mid-diff can produce an inconsistent
|
|
30
|
+
result. This is a genuine limitation, documented rather than hidden.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from typing import Any
|
|
36
|
+
from urllib.parse import parse_qs, unquote, urlparse
|
|
37
|
+
|
|
38
|
+
from parity.dialects.base import Dialect, sql_literal
|
|
39
|
+
from parity.types import Column, LogicalType
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class SnowflakeDialect(Dialect):
|
|
43
|
+
name = "snowflake"
|
|
44
|
+
#: Set from the connection's schema segment. Snowflake folds unquoted names
|
|
45
|
+
#: to upper case, so an unqualified table is looked up in the connected
|
|
46
|
+
#: schema as stored.
|
|
47
|
+
default_schema = "PUBLIC"
|
|
48
|
+
|
|
49
|
+
def connect(self, connection_string: str) -> None: # pragma: no cover
|
|
50
|
+
"""Open a connection, pinned to UTC.
|
|
51
|
+
|
|
52
|
+
URL grammar mirrors SQLAlchemy's Snowflake dialect:
|
|
53
|
+
``snowflake://user:password@account/database/schema?warehouse=wh&role=r``
|
|
54
|
+
|
|
55
|
+
Note the honest gap: Snowflake supports only READ COMMITTED, so there
|
|
56
|
+
is no equivalent of PostgreSQL's REPEATABLE READ - the diff cannot hold
|
|
57
|
+
one snapshot across the walk. UTC is still pinned so timestamps render
|
|
58
|
+
deterministically.
|
|
59
|
+
"""
|
|
60
|
+
import snowflake.connector
|
|
61
|
+
|
|
62
|
+
url = urlparse(connection_string)
|
|
63
|
+
parts = [p for p in url.path.split("/") if p]
|
|
64
|
+
database = parts[0] if parts else None
|
|
65
|
+
schema = parts[1] if len(parts) > 1 else None
|
|
66
|
+
params = parse_qs(url.query)
|
|
67
|
+
|
|
68
|
+
def opt(key: str) -> str | None:
|
|
69
|
+
"""First value of a query parameter, or None."""
|
|
70
|
+
values = params.get(key)
|
|
71
|
+
return values[0] if values else None
|
|
72
|
+
if schema:
|
|
73
|
+
self.default_schema = schema
|
|
74
|
+
|
|
75
|
+
self._conn = snowflake.connector.connect(
|
|
76
|
+
account=url.hostname,
|
|
77
|
+
user=unquote(url.username) if url.username else None,
|
|
78
|
+
password=unquote(url.password) if url.password else None,
|
|
79
|
+
database=database,
|
|
80
|
+
schema=schema,
|
|
81
|
+
warehouse=opt("warehouse"),
|
|
82
|
+
role=opt("role"),
|
|
83
|
+
autocommit=True,
|
|
84
|
+
)
|
|
85
|
+
cur = self._conn.cursor()
|
|
86
|
+
try:
|
|
87
|
+
# Timestamps render through this; pin it so two accounts in
|
|
88
|
+
# different regions cannot disagree on the same instant.
|
|
89
|
+
cur.execute("alter session set timezone = 'UTC'")
|
|
90
|
+
finally:
|
|
91
|
+
cur.close()
|
|
92
|
+
|
|
93
|
+
def close(self) -> None: # pragma: no cover
|
|
94
|
+
"""Close the connection."""
|
|
95
|
+
self._conn.close()
|
|
96
|
+
|
|
97
|
+
def query(self, sql: str) -> list[tuple[Any, ...]]: # pragma: no cover
|
|
98
|
+
"""Run `sql` and return every row as a list of tuples."""
|
|
99
|
+
cur = self._conn.cursor()
|
|
100
|
+
try:
|
|
101
|
+
cur.execute(sql)
|
|
102
|
+
return list(cur.fetchall())
|
|
103
|
+
finally:
|
|
104
|
+
cur.close()
|
|
105
|
+
|
|
106
|
+
def columns(self, table: str) -> list[Column]:
|
|
107
|
+
"""Introspect columns, reading numeric_scale to split NUMBER.
|
|
108
|
+
|
|
109
|
+
Snowflake reports every integer and decimal as `NUMBER`; only the scale
|
|
110
|
+
distinguishes them, so this cannot use the shared `columns()`.
|
|
111
|
+
"""
|
|
112
|
+
schema, name = self.split_table(table, self.default_schema)
|
|
113
|
+
rows = self.query(
|
|
114
|
+
"select column_name, data_type, numeric_scale "
|
|
115
|
+
"from information_schema.columns "
|
|
116
|
+
f"where table_schema = {sql_literal(schema)} "
|
|
117
|
+
f"and table_name = {sql_literal(name)} "
|
|
118
|
+
"order by ordinal_position"
|
|
119
|
+
)
|
|
120
|
+
if not rows:
|
|
121
|
+
raise self._err(self._not_found(table, schema, name))
|
|
122
|
+
return [
|
|
123
|
+
Column(str(r[0]), map_type_snowflake(str(r[1]), r[2]), str(r[1]))
|
|
124
|
+
for r in rows
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
def quote(self, identifier: str) -> str:
|
|
128
|
+
"""Wrap an identifier in double quotes, doubling any it contains.
|
|
129
|
+
|
|
130
|
+
The injection boundary - names arrive from the command line. Snowflake
|
|
131
|
+
stores unquoted names upper-cased, so a lower-case `"id"` will not match
|
|
132
|
+
a column created as `ID`; the tool sidesteps this by quoting the exact
|
|
133
|
+
names `columns()` read back.
|
|
134
|
+
"""
|
|
135
|
+
return '"' + identifier.replace('"', '""') + '"'
|
|
136
|
+
|
|
137
|
+
# ----------------------------------------------------------- rendering
|
|
138
|
+
|
|
139
|
+
def null_sentinel_sql(self) -> str:
|
|
140
|
+
r"""Build the sentinel from CHR(92), not a `\N` literal.
|
|
141
|
+
|
|
142
|
+
Snowflake processes backslash escapes in string literals, so a literal
|
|
143
|
+
is unsafe; CHR(92) is a backslash unconditionally and CHR returns a
|
|
144
|
+
varchar (not binary), so the coalesce stays text.
|
|
145
|
+
"""
|
|
146
|
+
return "(chr(92) || 'N')"
|
|
147
|
+
|
|
148
|
+
def normalize(self, column: Column) -> str:
|
|
149
|
+
"""Render one column as canonical text, null-safe.
|
|
150
|
+
|
|
151
|
+
DECIMAL/FLOAT match DuckDB's path - cast to NUMBER(38, scale) then to
|
|
152
|
+
text - so `1.5` becomes `'1.500000'`. Non-finite floats are a known
|
|
153
|
+
unverified gap: Snowflake's Inf/NaN detection differs from the other
|
|
154
|
+
engines and must be checked against a real account before this is
|
|
155
|
+
trusted for FLOAT columns holding them.
|
|
156
|
+
"""
|
|
157
|
+
c = self.quote(column.name)
|
|
158
|
+
t = column.logical_type
|
|
159
|
+
if t is LogicalType.INTEGER:
|
|
160
|
+
expr = f"cast({c} as varchar)"
|
|
161
|
+
elif t in (LogicalType.DECIMAL, LogicalType.FLOAT):
|
|
162
|
+
expr = f"cast(cast({c} as number(38,{self.float_scale})) as varchar)"
|
|
163
|
+
elif t is LogicalType.BOOLEAN:
|
|
164
|
+
expr = f"case when {c} then 'true' when not {c} then 'false' end"
|
|
165
|
+
elif t is LogicalType.DATE:
|
|
166
|
+
expr = f"to_char({c}, 'YYYY-MM-DD')"
|
|
167
|
+
elif t is LogicalType.TIMESTAMP:
|
|
168
|
+
# FF6 is microseconds, matching the other engines' six digits.
|
|
169
|
+
expr = f"to_char({c}, 'YYYY-MM-DD HH24:MI:SS.FF6')"
|
|
170
|
+
else:
|
|
171
|
+
expr = f"cast({c} as varchar)"
|
|
172
|
+
return f"coalesce({expr}, {self.null_sentinel_sql()})"
|
|
173
|
+
|
|
174
|
+
def hash_expr(self, text_expr: str) -> str:
|
|
175
|
+
"""Fold canonical text into a positive 60-bit integer.
|
|
176
|
+
|
|
177
|
+
`MD5_NUMBER_UPPER64` is the top 64 bits of the digest as an unsigned
|
|
178
|
+
number; the top 60 bits - the first 15 hex characters the other engines
|
|
179
|
+
take - are that floor-divided by 16.
|
|
180
|
+
"""
|
|
181
|
+
return f"floor(md5_number_upper64({text_expr}) / 16)"
|
|
182
|
+
|
|
183
|
+
def int_div(self, numerator: str, denominator: str) -> str:
|
|
184
|
+
"""Truncating integer division.
|
|
185
|
+
|
|
186
|
+
Operands are NUMBER (via `wide_int`), so `/` is exact decimal and
|
|
187
|
+
`floor` truncates exactly for the non-negative operands used here -
|
|
188
|
+
unlike a float `/`, which CLAUDE.md 4.5 warns loses precision.
|
|
189
|
+
"""
|
|
190
|
+
return f"floor(({numerator}) / ({denominator}))"
|
|
191
|
+
|
|
192
|
+
def wide_int(self, expr: str) -> str:
|
|
193
|
+
"""Widen past 64 bits before arithmetic. NUMBER(38,0) holds 38 digits,
|
|
194
|
+
which the key offset cannot overflow."""
|
|
195
|
+
return f"cast(({expr}) as number(38,0))"
|
|
196
|
+
|
|
197
|
+
def sum_wide(self, expr: str) -> str:
|
|
198
|
+
"""Sum row hashes without overflowing, and return 0 for an empty group.
|
|
199
|
+
|
|
200
|
+
NUMBER(38,0) is far beyond any row count times 2^60.
|
|
201
|
+
"""
|
|
202
|
+
return f"coalesce(sum(cast(({expr}) as number(38,0))), 0)"
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def map_type_snowflake(raw: str, numeric_scale: Any) -> LogicalType:
|
|
206
|
+
"""Map a Snowflake type onto a logical category.
|
|
207
|
+
|
|
208
|
+
Snowflake reports both integers and decimals as `NUMBER`; only the scale
|
|
209
|
+
tells them apart, so it is passed in. A NULL scale (non-numeric type) is
|
|
210
|
+
treated as not-an-integer.
|
|
211
|
+
"""
|
|
212
|
+
t = raw.upper().split("(")[0].strip()
|
|
213
|
+
if t in {"NUMBER", "DECIMAL", "NUMERIC"}:
|
|
214
|
+
try:
|
|
215
|
+
scale = int(numeric_scale) if numeric_scale is not None else 6
|
|
216
|
+
except (TypeError, ValueError):
|
|
217
|
+
scale = 6
|
|
218
|
+
return LogicalType.INTEGER if scale == 0 else LogicalType.DECIMAL
|
|
219
|
+
if t in {"INT", "INTEGER", "BIGINT", "SMALLINT", "TINYINT", "BYTEINT"}:
|
|
220
|
+
return LogicalType.INTEGER # aliases that may appear via DATA_TYPE_ALIAS
|
|
221
|
+
if t in {"FLOAT", "FLOAT4", "FLOAT8", "DOUBLE", "DOUBLE PRECISION", "REAL"}:
|
|
222
|
+
return LogicalType.FLOAT
|
|
223
|
+
if t == "BOOLEAN":
|
|
224
|
+
return LogicalType.BOOLEAN
|
|
225
|
+
if t == "DATE":
|
|
226
|
+
return LogicalType.DATE
|
|
227
|
+
if t.startswith("TIMESTAMP") or t == "DATETIME":
|
|
228
|
+
return LogicalType.TIMESTAMP
|
|
229
|
+
if t in {"TEXT", "VARCHAR", "CHAR", "CHARACTER", "STRING"}:
|
|
230
|
+
return LogicalType.STRING
|
|
231
|
+
return LogicalType.UNKNOWN
|
|
@@ -132,6 +132,34 @@ def _key_order(key: int | str) -> tuple[int, int | str]:
|
|
|
132
132
|
return (1, key) if isinstance(key, str) else (0, key)
|
|
133
133
|
|
|
134
134
|
|
|
135
|
+
def _fold_columns(
|
|
136
|
+
columns: list[Column], side: str, table: str
|
|
137
|
+
) -> dict[str, Column]:
|
|
138
|
+
"""Index a side's columns by case-folded name, for cross-engine matching.
|
|
139
|
+
|
|
140
|
+
Two engines fold unquoted identifiers to different cases, so a column is
|
|
141
|
+
the same column on both sides when its *folded* name matches. Each side
|
|
142
|
+
still holds its own `Column`, whose real stored name is what gets quoted
|
|
143
|
+
into SQL - only the matching is case-insensitive, never the rendering.
|
|
144
|
+
|
|
145
|
+
A table with two columns that differ only in case (possible only through
|
|
146
|
+
quoted identifiers) cannot be folded unambiguously, so it is refused with
|
|
147
|
+
a clear message rather than silently dropping one of them.
|
|
148
|
+
"""
|
|
149
|
+
out: dict[str, Column] = {}
|
|
150
|
+
for c in columns:
|
|
151
|
+
fold = c.name.casefold()
|
|
152
|
+
if fold in out:
|
|
153
|
+
raise ValueError(
|
|
154
|
+
f"[side {side}] {table} has two columns that differ only in "
|
|
155
|
+
f"case: {out[fold].name!r} and {c.name!r}. parity matches "
|
|
156
|
+
f"columns case-insensitively across engines and cannot tell "
|
|
157
|
+
f"these apart - rename or quote one, or diff a view that does."
|
|
158
|
+
)
|
|
159
|
+
out[fold] = c
|
|
160
|
+
return out
|
|
161
|
+
|
|
162
|
+
|
|
135
163
|
def _resolve_key(
|
|
136
164
|
key: str | Sequence[str],
|
|
137
165
|
cols_a: dict[str, Column],
|
|
@@ -149,22 +177,27 @@ def _resolve_key(
|
|
|
149
177
|
|
|
150
178
|
Returns one spec per side. They agree on shape but hold each side's own
|
|
151
179
|
`Column` objects, because a column can be `text` on one side and
|
|
152
|
-
`varchar` on the other and each dialect renders its own
|
|
180
|
+
`varchar` on the other and each dialect renders its own - and the key can
|
|
181
|
+
be `ID` on one side and `id` on the other, since `--key` is one name that
|
|
182
|
+
has to resolve against whatever case each engine stored.
|
|
183
|
+
|
|
184
|
+
`cols_a` and `cols_b` are keyed by case-folded name (see `_fold_columns`),
|
|
185
|
+
so the same `--key id` finds `id` on PostgreSQL and `ID` on Snowflake.
|
|
153
186
|
"""
|
|
154
187
|
names = [key] if isinstance(key, str) else list(dict.fromkeys(key))
|
|
155
188
|
if not names:
|
|
156
189
|
raise ValueError("--key needs at least one column")
|
|
157
190
|
|
|
158
191
|
for side, table, cols in (("A", a_table, cols_a), ("B", b_table, cols_b)):
|
|
159
|
-
missing = [n for n in names if n not in cols]
|
|
192
|
+
missing = [n for n in names if n.casefold() not in cols]
|
|
160
193
|
if missing:
|
|
161
194
|
raise ValueError(
|
|
162
195
|
f"[side {side}] key column(s) {missing} not in {table}. "
|
|
163
|
-
f"Columns are: {sorted(cols)}"
|
|
196
|
+
f"Columns are: {sorted(c.name for c in cols.values())}"
|
|
164
197
|
)
|
|
165
198
|
|
|
166
|
-
a_key = tuple(cols_a[n] for n in names)
|
|
167
|
-
b_key = tuple(cols_b[n] for n in names)
|
|
199
|
+
a_key = tuple(cols_a[n.casefold()] for n in names)
|
|
200
|
+
b_key = tuple(cols_b[n.casefold()] for n in names)
|
|
168
201
|
|
|
169
202
|
# Hash unless it is one integer column on both sides. A single-column key
|
|
170
203
|
# that is integer on one side and text on the other has to be hashed too,
|
|
@@ -194,21 +227,29 @@ def _resolve_key(
|
|
|
194
227
|
def _select_columns(
|
|
195
228
|
cols_a: dict[str, Column],
|
|
196
229
|
cols_b: dict[str, Column],
|
|
197
|
-
|
|
230
|
+
key_folds: set[str],
|
|
198
231
|
columns: Sequence[str] | None,
|
|
199
232
|
exclude: Sequence[str],
|
|
200
233
|
warnings: list[str],
|
|
201
234
|
) -> list[str]:
|
|
202
235
|
"""Decide which columns to compare, explaining anything dropped.
|
|
203
236
|
|
|
237
|
+
`cols_a`/`cols_b` are keyed by case-folded name and the returned list is
|
|
238
|
+
folded names too, so matching is case-insensitive across engines (see
|
|
239
|
+
`_fold_columns`); the caller maps each folded name back to that side's real
|
|
240
|
+
`Column`. Messages echo the user's own tokens, or side A's stored name, so
|
|
241
|
+
a folded lookup never leaks a lower-cased identifier back at the reader.
|
|
242
|
+
|
|
204
243
|
Key columns are never compared: they are how rows are matched up, not
|
|
205
244
|
something compared between them.
|
|
206
245
|
"""
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
excluded = set(exclude)
|
|
246
|
+
both = (set(cols_a) & set(cols_b)) - key_folds
|
|
247
|
+
excluded = {e.casefold() for e in exclude}
|
|
210
248
|
|
|
211
|
-
unknown_exclude =
|
|
249
|
+
unknown_exclude = [
|
|
250
|
+
e for e in dict.fromkeys(exclude)
|
|
251
|
+
if e.casefold() not in cols_a and e.casefold() not in cols_b
|
|
252
|
+
]
|
|
212
253
|
if unknown_exclude:
|
|
213
254
|
warnings.append(
|
|
214
255
|
f"--exclude named columns that exist on neither side: "
|
|
@@ -217,13 +258,19 @@ def _select_columns(
|
|
|
217
258
|
|
|
218
259
|
shared = sorted(both - excluded)
|
|
219
260
|
if columns:
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
261
|
+
# Fold for matching, but keep the first spelling the user gave each
|
|
262
|
+
# column so error messages read back their own words, not a fold.
|
|
263
|
+
token: dict[str, str] = {}
|
|
264
|
+
for c in columns:
|
|
265
|
+
token.setdefault(c.casefold(), c)
|
|
266
|
+
req = list(token) # folded, de-duplicated, order kept
|
|
267
|
+
nowhere_folds = {f for f in req if f not in cols_a and f not in cols_b}
|
|
268
|
+
nowhere = [token[f] for f in req if f in nowhere_folds]
|
|
269
|
+
one_side = [token[f] for f in req if f not in nowhere_folds and f not in both]
|
|
270
|
+
dropped = [token[f] for f in req if f in excluded]
|
|
224
271
|
# Order matters: key columns are present on both sides but excluded
|
|
225
272
|
# from `both`, so they would otherwise be misreported as one-sided.
|
|
226
|
-
if named_keys := [
|
|
273
|
+
if named_keys := [token[f] for f in req if f in key_folds]:
|
|
227
274
|
raise ValueError(
|
|
228
275
|
f"--columns named the key column(s) {named_keys}; the key is "
|
|
229
276
|
f"how rows are matched up, not something compared between them"
|
|
@@ -238,26 +285,28 @@ def _select_columns(
|
|
|
238
285
|
raise ValueError(
|
|
239
286
|
f"--columns and --exclude both name: {dropped}"
|
|
240
287
|
)
|
|
241
|
-
shared = [
|
|
288
|
+
shared = [f for f in shared if f in set(req)]
|
|
242
289
|
|
|
243
|
-
for side, only in (
|
|
244
|
-
("A", sorted(set(cols_a) - set(cols_b) -
|
|
245
|
-
("B", sorted(set(cols_b) - set(cols_a) -
|
|
290
|
+
for side, cols, only in (
|
|
291
|
+
("A", cols_a, sorted(set(cols_a) - set(cols_b) - key_folds)),
|
|
292
|
+
("B", cols_b, sorted(set(cols_b) - set(cols_a) - key_folds)),
|
|
246
293
|
):
|
|
247
294
|
if only:
|
|
248
|
-
|
|
295
|
+
names = sorted(cols[f].name for f in only)
|
|
296
|
+
warnings.append(f"not compared, present only on side {side}: {names}")
|
|
249
297
|
|
|
250
298
|
# A column whose logical type differs between sides renders through a
|
|
251
299
|
# different canonical encoding, so every row would report as changed. That
|
|
252
300
|
# looks like a catastrophic data difference but is really a schema
|
|
253
301
|
# difference, so name it explicitly.
|
|
254
|
-
for
|
|
255
|
-
|
|
302
|
+
for f in shared:
|
|
303
|
+
name = cols_a[f].name
|
|
304
|
+
ta, tb = cols_a[f].logical_type, cols_b[f].logical_type
|
|
256
305
|
if ta is tb or {ta, tb} <= _NUMERIC_EQUIVALENT:
|
|
257
306
|
continue
|
|
258
307
|
warnings.append(
|
|
259
|
-
f"column {name!r} is {cols_a[
|
|
260
|
-
f"but {cols_b[
|
|
308
|
+
f"column {name!r} is {cols_a[f].raw_type or ta.value} on side A "
|
|
309
|
+
f"but {cols_b[f].raw_type or tb.value} on side B; values are "
|
|
261
310
|
f"compared as text and will very likely all differ"
|
|
262
311
|
)
|
|
263
312
|
|
|
@@ -267,21 +316,21 @@ def _select_columns(
|
|
|
267
316
|
# are pinned to UTC, so a migration that stored UTC compares clean. One
|
|
268
317
|
# that stored local wall-clock reports *every* row as different, and
|
|
269
318
|
# without this line there is nothing pointing at which axis to look along.
|
|
270
|
-
for
|
|
271
|
-
aware_a = _is_tz_aware(cols_a[
|
|
272
|
-
if aware_a is _is_tz_aware(cols_b[
|
|
319
|
+
for f in shared:
|
|
320
|
+
aware_a = _is_tz_aware(cols_a[f])
|
|
321
|
+
if aware_a is _is_tz_aware(cols_b[f]):
|
|
273
322
|
continue
|
|
274
323
|
aware, naive = ("A", "B") if aware_a else ("B", "A")
|
|
275
324
|
warnings.append(
|
|
276
|
-
f"column {name!r} is timezone-aware on side {aware} but
|
|
277
|
-
f"side {naive}; both are read in UTC, so a migration that
|
|
278
|
-
f"local wall-clock time rather than UTC will show every row
|
|
279
|
-
f"different"
|
|
325
|
+
f"column {cols_a[f].name!r} is timezone-aware on side {aware} but "
|
|
326
|
+
f"not on side {naive}; both are read in UTC, so a migration that "
|
|
327
|
+
f"stored local wall-clock time rather than UTC will show every row "
|
|
328
|
+
f"as different"
|
|
280
329
|
)
|
|
281
330
|
|
|
282
331
|
unknown = sorted(
|
|
283
|
-
{
|
|
284
|
-
| {
|
|
332
|
+
{cols_a[f].name for f in shared if cols_a[f].logical_type is LogicalType.UNKNOWN}
|
|
333
|
+
| {cols_a[f].name for f in shared if cols_b[f].logical_type is LogicalType.UNKNOWN}
|
|
285
334
|
)
|
|
286
335
|
if unknown:
|
|
287
336
|
warnings.append(
|
|
@@ -343,8 +392,15 @@ def diff(
|
|
|
343
392
|
return _gather(fa, fb)
|
|
344
393
|
|
|
345
394
|
cols_a_list, cols_b_list = both("columns")
|
|
346
|
-
|
|
347
|
-
|
|
395
|
+
# Match identifiers case-insensitively across the two sides. Engines
|
|
396
|
+
# fold unquoted names differently - Snowflake upper-cases, PostgreSQL
|
|
397
|
+
# and DuckDB lower-case - so a Postgres `amount` and its Snowflake
|
|
398
|
+
# `AMOUNT` are the same column and must line up, or the headline use
|
|
399
|
+
# case (diffing a table against its migration) finds no shared columns
|
|
400
|
+
# and no usable key. The map is keyed by the folded name; each side
|
|
401
|
+
# keeps its own Column, with its real stored name, for quoting in SQL.
|
|
402
|
+
cols_a = _fold_columns(cols_a_list, "A", a_table)
|
|
403
|
+
cols_b = _fold_columns(cols_b_list, "B", b_table)
|
|
348
404
|
# Introspection is metadata, not a scan; do not inflate the query count
|
|
349
405
|
# users read as "how much work did this cost".
|
|
350
406
|
stats.queries -= 2
|
|
@@ -352,12 +408,13 @@ def diff(
|
|
|
352
408
|
key_a, key_b = _resolve_key(
|
|
353
409
|
key, cols_a, cols_b, a_table, b_table, warnings
|
|
354
410
|
)
|
|
411
|
+
key_folds = {c.name.casefold() for c in key_a.columns}
|
|
355
412
|
|
|
356
413
|
shared = _select_columns(
|
|
357
|
-
cols_a, cols_b,
|
|
414
|
+
cols_a, cols_b, key_folds, columns, exclude, warnings
|
|
358
415
|
)
|
|
359
|
-
a_cols = [cols_a[
|
|
360
|
-
b_cols = [cols_b[
|
|
416
|
+
a_cols = [cols_a[f] for f in shared]
|
|
417
|
+
b_cols = [cols_b[f] for f in shared]
|
|
361
418
|
|
|
362
419
|
ks_a, ks_b = both("key_stats", key_a, _args_b=(key_b,))
|
|
363
420
|
|