parity-diff 0.2.1__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {parity_diff-0.2.1 → parity_diff-0.2.3}/CONTRIBUTING.md +41 -0
  2. {parity_diff-0.2.1/src/parity_diff.egg-info → parity_diff-0.2.3}/PKG-INFO +11 -3
  3. {parity_diff-0.2.1 → parity_diff-0.2.3}/README.md +7 -2
  4. {parity_diff-0.2.1 → parity_diff-0.2.3}/pyproject.toml +8 -2
  5. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/__init__.py +1 -1
  6. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/base.py +5 -1
  7. parity_diff-0.2.3/src/parity/dialects/snowflake_dialect.py +231 -0
  8. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/engine.py +95 -38
  9. {parity_diff-0.2.1 → parity_diff-0.2.3/src/parity_diff.egg-info}/PKG-INFO +11 -3
  10. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/SOURCES.txt +7 -1
  11. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/requires.txt +4 -0
  12. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/conftest.py +49 -0
  13. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_engine.py +51 -0
  14. parity_diff-0.2.3/tests/test_identical.py +222 -0
  15. parity_diff-0.2.3/tests/test_identical_live.py +382 -0
  16. parity_diff-0.2.3/tests/test_mysql_postgres.py +106 -0
  17. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_properties.py +2 -1
  18. parity_diff-0.2.3/tests/test_snowflake.py +181 -0
  19. parity_diff-0.2.3/tests/test_snowflake_offline.py +113 -0
  20. {parity_diff-0.2.1 → parity_diff-0.2.3}/LICENSE +0 -0
  21. {parity_diff-0.2.1 → parity_diff-0.2.3}/MANIFEST.in +0 -0
  22. {parity_diff-0.2.1 → parity_diff-0.2.3}/setup.cfg +0 -0
  23. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/cli.py +0 -0
  24. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/__init__.py +0 -0
  25. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/duckdb_dialect.py +0 -0
  26. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/mysql_dialect.py +0 -0
  27. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/dialects/postgres_dialect.py +0 -0
  28. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity/types.py +0 -0
  29. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/dependency_links.txt +0 -0
  30. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/entry_points.txt +0 -0
  31. {parity_diff-0.2.1 → parity_diff-0.2.3}/src/parity_diff.egg-info/top_level.txt +0 -0
  32. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/fakes.py +0 -0
  33. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_cli.py +0 -0
  34. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_encoding.py +0 -0
  35. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_fuzz_encoding.py +0 -0
  36. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_integration.py +0 -0
  37. {parity_diff-0.2.1 → parity_diff-0.2.3}/tests/test_mysql.py +0 -0
@@ -97,6 +97,34 @@ MySQL, added after v0.1.0, turned up two more that a warehouse dialect may hit:
97
97
  sentinel silently became `N`. Build such bytes from `CHAR`/hex, not a
98
98
  literal, and verify by hashing rather than by reading the SQL.
99
99
 
100
+ Snowflake, the first warehouse (v0.2.2), added three more — the sort a warehouse
101
+ is especially likely to spring:
102
+
103
+ 9. **Your engine may have neither a bit-cast nor `CONV` to reach 60 bits.**
104
+ Snowflake had no `bit(60)::bigint` and no `conv(hex,16,10)`. What it did have
105
+ is `md5_number_upper64(x)`, the top 64 bits of the digest as a number, and
106
+ `floor(that / 16)` drops the low 4 to land on the same 60-bit prefix - the
107
+ fourth distinct path to `648541476951500027`. Find your engine's own route;
108
+ the constant is the contract, not the SQL that reaches it.
109
+
110
+ 10. **Integers and decimals may share one type name.** Snowflake reports both as
111
+ `NUMBER` and only `numeric_scale` (0 = integer) tells them apart, so its
112
+ `columns()` reads the scale instead of trusting `data_type`. Trust the type
113
+ name and an integer key renders as `42.000000` and never matches another
114
+ engine's `42`. If your engine collapses numeric types like this, override
115
+ `columns()`.
116
+
117
+ 11. **Identifier case-folding is the engine's, and it is not universal.**
118
+ Snowflake upper-cases unquoted identifiers where PostgreSQL and DuckDB
119
+ lower-case them. The engine matches keys and columns case-insensitively for
120
+ exactly this reason (`_fold_columns`), but the *table* name is looked up in
121
+ the case the engine stored, so `--a-table orders` against a Snowflake
122
+ `ORDERS` fails with a near-miss hint. And some warehouses (Snowflake among
123
+ them) offer only READ COMMITTED, so unlike PostgreSQL the walk cannot be
124
+ pinned to one snapshot - a real limitation to document, not hide. None of
125
+ these three showed up in the docs; they surfaced only against a live
126
+ account, which is why the rule below is not negotiable.
127
+
100
128
  ### Proving it works
101
129
 
102
130
  A dialect is not done until `tests/test_encoding.py` passes against it. That
@@ -136,6 +164,19 @@ reason next to it rather than being switched off globally.
136
164
  PostgreSQL-backed tests read `PARITY_TEST_PG` and skip cleanly when nothing is
137
165
  listening, so the suite is useful with only DuckDB installed.
138
166
 
167
+ The generative suites hunt for the one failure that matters most — a real
168
+ difference reported as identical. To run that hunt deeper (many more generated
169
+ cases per property, at the cost of time), scale it with `PARITY_DEEP`:
170
+
171
+ ```bash
172
+ PARITY_DEEP=20 pytest tests/test_identical.py tests/test_properties.py
173
+ ```
174
+
175
+ The default of `1` keeps the everyday suite fast; the deep run is the repeatable
176
+ "make sure there is no abnormality" check on the identical guarantee. A dozen
177
+ seeds at `PARITY_DEEP=6` have turned up nothing — but the point is that anyone
178
+ can re-run it.
179
+
139
180
  ## Scope
140
181
 
141
182
  Before proposing a feature, check it against the question the tool exists to
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parity-diff
3
- Version: 0.2.1
3
+ Version: 0.2.3
4
4
  Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
5
5
  Author: Alessio Sorio
6
6
  License-Expression: MIT
@@ -30,10 +30,13 @@ Provides-Extra: postgres
30
30
  Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
31
31
  Provides-Extra: mysql
32
32
  Requires-Dist: mysql-connector-python>=8.0; extra == "mysql"
33
+ Provides-Extra: snowflake
34
+ Requires-Dist: snowflake-connector-python>=3.0; extra == "snowflake"
33
35
  Provides-Extra: all
34
36
  Requires-Dist: duckdb>=1.0; extra == "all"
35
37
  Requires-Dist: psycopg[binary]>=3.1; extra == "all"
36
38
  Requires-Dist: mysql-connector-python>=8.0; extra == "all"
39
+ Requires-Dist: snowflake-connector-python>=3.0; extra == "all"
37
40
  Provides-Extra: test
38
41
  Requires-Dist: hypothesis>=6.0; extra == "test"
39
42
  Dynamic: license-file
@@ -127,6 +130,7 @@ pip install "parity-diff[all]" # every engine
127
130
  pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
128
131
  pip install "parity-diff[postgres]" # PostgreSQL only
129
132
  pip install "parity-diff[mysql]" # MySQL only
133
+ pip install "parity-diff[snowflake]" # Snowflake only
130
134
  ```
131
135
 
132
136
  > The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
@@ -288,7 +292,8 @@ stated plainly rather than buried.
288
292
  | PostgreSQL | supported (tested against 16 and 18) |
289
293
  | DuckDB | supported (tested against 1.5) |
290
294
  | MySQL | supported (tested against 8.0) |
291
- | Snowflake, BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
295
+ | Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
296
+ | BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
292
297
 
293
298
  ## Scope
294
299
 
@@ -336,7 +341,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
336
341
 
337
342
  ## Changelog
338
343
 
339
- See `CHANGELOG.md`.
344
+ Latest release **0.2.1** — a generative test suite (property-based, oracle, and
345
+ cross-engine fuzzing) and the correctness fix it surfaced: a same-content
346
+ insert-plus-delete falling in one bucket that could previously read as
347
+ identical. See `CHANGELOG.md` for the full history.
340
348
 
341
349
  ## License
342
350
 
@@ -87,6 +87,7 @@ pip install "parity-diff[all]" # every engine
87
87
  pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
88
88
  pip install "parity-diff[postgres]" # PostgreSQL only
89
89
  pip install "parity-diff[mysql]" # MySQL only
90
+ pip install "parity-diff[snowflake]" # Snowflake only
90
91
  ```
91
92
 
92
93
  > The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
@@ -248,7 +249,8 @@ stated plainly rather than buried.
248
249
  | PostgreSQL | supported (tested against 16 and 18) |
249
250
  | DuckDB | supported (tested against 1.5) |
250
251
  | MySQL | supported (tested against 8.0) |
251
- | Snowflake, BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
252
+ | Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
253
+ | BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
252
254
 
253
255
  ## Scope
254
256
 
@@ -296,7 +298,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
296
298
 
297
299
  ## Changelog
298
300
 
299
- See `CHANGELOG.md`.
301
+ Latest release **0.2.1** — a generative test suite (property-based, oracle, and
302
+ cross-engine fuzzing) and the correctness fix it surfaced: a same-content
303
+ insert-plus-delete falling in one bucket that could previously read as
304
+ identical. See `CHANGELOG.md` for the full history.
300
305
 
301
306
  ## License
302
307
 
@@ -3,7 +3,7 @@
3
3
  # PyPI by an empty project. The import name, the CLI command and the repo are
4
4
  # all still `parity`: `pip install parity-diff` gives you `parity ...`.
5
5
  name = "parity-diff"
6
- version = "0.2.1"
6
+ version = "0.2.3"
7
7
  description = "Prove two tables in two different database engines hold the same data - without moving the data out of either engine."
8
8
  readme = "README.md"
9
9
  license = "MIT"
@@ -37,7 +37,11 @@ dependencies = []
37
37
  duckdb = ["duckdb>=1.0"]
38
38
  postgres = ["psycopg[binary]>=3.1"]
39
39
  mysql = ["mysql-connector-python>=8.0"]
40
- all = ["duckdb>=1.0", "psycopg[binary]>=3.1", "mysql-connector-python>=8.0"]
40
+ snowflake = ["snowflake-connector-python>=3.0"]
41
+ all = [
42
+ "duckdb>=1.0", "psycopg[binary]>=3.1",
43
+ "mysql-connector-python>=8.0", "snowflake-connector-python>=3.0",
44
+ ]
41
45
  # Test-only dependencies, kept out of every runtime extra. Hypothesis drives
42
46
  # the generative suite (tests/test_properties.py); the core stays stdlib-only.
43
47
  test = ["hypothesis>=6.0"]
@@ -64,6 +68,7 @@ markers = [
64
68
  "postgres: requires a reachable PostgreSQL server",
65
69
  "mysql: requires a reachable MySQL server",
66
70
  "duckdb: requires the duckdb driver",
71
+ "snowflake: requires a reachable Snowflake account (PARITY_TEST_SNOWFLAKE)",
67
72
  ]
68
73
 
69
74
  [tool.ruff]
@@ -105,6 +110,7 @@ ignore = [
105
110
  "src/parity/dialects/base.py" = ["S608"]
106
111
  "src/parity/dialects/postgres_dialect.py" = ["S608"]
107
112
  "src/parity/dialects/mysql_dialect.py" = ["S608"]
113
+ "src/parity/dialects/snowflake_dialect.py" = ["S608"]
108
114
  # proof.py is kept close to the original author's script so it stays readable
109
115
  # next to the findings it produced; its terse one-line style is deliberate.
110
116
  "demo/proof.py" = ["E401", "E402", "E701", "E702", "B007", "S311", "S608"]
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  from typing import Any
15
15
 
16
- __version__ = "0.2.1"
16
+ __version__ = "0.2.3"
17
17
 
18
18
  __all__ = ["__version__", "diff", "get_dialect"]
19
19
 
@@ -551,10 +551,14 @@ def get_dialect(
551
551
  from parity.dialects.mysql_dialect import MySQLDialect
552
552
 
553
553
  dialect = MySQLDialect(float_scale=float_scale, side=side)
554
+ elif scheme in ("snowflake",):
555
+ from parity.dialects.snowflake_dialect import SnowflakeDialect
556
+
557
+ dialect = SnowflakeDialect(float_scale=float_scale, side=side)
554
558
  else:
555
559
  raise ValueError(
556
560
  f"[side {side}] no dialect for scheme {scheme!r}. "
557
- f"Supported: duckdb, postgres, mysql."
561
+ f"Supported: duckdb, postgres, mysql, snowflake."
558
562
  )
559
563
  try:
560
564
  dialect.connect(connection_string)
@@ -0,0 +1,231 @@
1
+ """Snowflake dialect.
2
+
3
+ Verified against a live Snowflake account (AWS eu-central-2, 2026-09-07): the
4
+ hash constant agrees, a full-table checksum matches DuckDB byte-for-byte over
5
+ 5,000 mixed-type rows, and `tests/test_snowflake.py` passes end to end - the
6
+ identical check, a changed decimal, a deleted row, and the NULL-versus-empty
7
+ trap. It was first written against Snowflake's documentation; the live run then
8
+ turned up the one thing docs could not: Snowflake upper-cases unquoted
9
+ identifiers, so the engine had to match keys and columns case-insensitively
10
+ (see `_fold_columns` in engine.py) for a Snowflake table to diff against a
11
+ lower-casing engine at all.
12
+
13
+ The Snowflake-specific decisions, each confirmed by that run:
14
+
15
+ - **No `CONV` and no bit-cast for the hash.** Snowflake has neither, but
16
+ `MD5_NUMBER_UPPER64(x)` returns the top 64 bits of the digest as an unsigned
17
+ number, and the top 60 bits - the first 15 hex characters the other engines
18
+ fold - are `FLOOR(that / 16)`. That should equal 648541476951500027 for
19
+ `'abc'`; a test must pin it.
20
+ - **Integer and decimal both report as `NUMBER`.** `information_schema` tells
21
+ them apart only by `numeric_scale` (0 = integer), so `columns()` reads the
22
+ scale rather than trusting `data_type` - otherwise an integer key would be
23
+ rendered as `42.000000` and never match another engine's `42`.
24
+ - **The NULL sentinel is built from `CHR(92)`.** Snowflake interprets
25
+ backslash escapes in string literals, so a literal `'\\N'` is unsafe the same
26
+ way it was on MySQL.
27
+ - **One-snapshot isolation is not available.** Snowflake offers only READ
28
+ COMMITTED, so unlike PostgreSQL the walk cannot be pinned to a single
29
+ snapshot; a source table mutating mid-diff can produce an inconsistent
30
+ result. This is a genuine limitation, documented rather than hidden.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ from typing import Any
36
+ from urllib.parse import parse_qs, unquote, urlparse
37
+
38
+ from parity.dialects.base import Dialect, sql_literal
39
+ from parity.types import Column, LogicalType
40
+
41
+
42
+ class SnowflakeDialect(Dialect):
43
+ name = "snowflake"
44
+ #: Set from the connection's schema segment. Snowflake folds unquoted names
45
+ #: to upper case, so an unqualified table is looked up in the connected
46
+ #: schema as stored.
47
+ default_schema = "PUBLIC"
48
+
49
+ def connect(self, connection_string: str) -> None: # pragma: no cover
50
+ """Open a connection, pinned to UTC.
51
+
52
+ URL grammar mirrors SQLAlchemy's Snowflake dialect:
53
+ ``snowflake://user:password@account/database/schema?warehouse=wh&role=r``
54
+
55
+ Note the honest gap: Snowflake supports only READ COMMITTED, so there
56
+ is no equivalent of PostgreSQL's REPEATABLE READ - the diff cannot hold
57
+ one snapshot across the walk. UTC is still pinned so timestamps render
58
+ deterministically.
59
+ """
60
+ import snowflake.connector
61
+
62
+ url = urlparse(connection_string)
63
+ parts = [p for p in url.path.split("/") if p]
64
+ database = parts[0] if parts else None
65
+ schema = parts[1] if len(parts) > 1 else None
66
+ params = parse_qs(url.query)
67
+
68
+ def opt(key: str) -> str | None:
69
+ """First value of a query parameter, or None."""
70
+ values = params.get(key)
71
+ return values[0] if values else None
72
+ if schema:
73
+ self.default_schema = schema
74
+
75
+ self._conn = snowflake.connector.connect(
76
+ account=url.hostname,
77
+ user=unquote(url.username) if url.username else None,
78
+ password=unquote(url.password) if url.password else None,
79
+ database=database,
80
+ schema=schema,
81
+ warehouse=opt("warehouse"),
82
+ role=opt("role"),
83
+ autocommit=True,
84
+ )
85
+ cur = self._conn.cursor()
86
+ try:
87
+ # Timestamps render through this; pin it so two accounts in
88
+ # different regions cannot disagree on the same instant.
89
+ cur.execute("alter session set timezone = 'UTC'")
90
+ finally:
91
+ cur.close()
92
+
93
+ def close(self) -> None: # pragma: no cover
94
+ """Close the connection."""
95
+ self._conn.close()
96
+
97
+ def query(self, sql: str) -> list[tuple[Any, ...]]: # pragma: no cover
98
+ """Run `sql` and return every row as a list of tuples."""
99
+ cur = self._conn.cursor()
100
+ try:
101
+ cur.execute(sql)
102
+ return list(cur.fetchall())
103
+ finally:
104
+ cur.close()
105
+
106
+ def columns(self, table: str) -> list[Column]:
107
+ """Introspect columns, reading numeric_scale to split NUMBER.
108
+
109
+ Snowflake reports every integer and decimal as `NUMBER`; only the scale
110
+ distinguishes them, so this cannot use the shared `columns()`.
111
+ """
112
+ schema, name = self.split_table(table, self.default_schema)
113
+ rows = self.query(
114
+ "select column_name, data_type, numeric_scale "
115
+ "from information_schema.columns "
116
+ f"where table_schema = {sql_literal(schema)} "
117
+ f"and table_name = {sql_literal(name)} "
118
+ "order by ordinal_position"
119
+ )
120
+ if not rows:
121
+ raise self._err(self._not_found(table, schema, name))
122
+ return [
123
+ Column(str(r[0]), map_type_snowflake(str(r[1]), r[2]), str(r[1]))
124
+ for r in rows
125
+ ]
126
+
127
+ def quote(self, identifier: str) -> str:
128
+ """Wrap an identifier in double quotes, doubling any it contains.
129
+
130
+ The injection boundary - names arrive from the command line. Snowflake
131
+ stores unquoted names upper-cased, so a lower-case `"id"` will not match
132
+ a column created as `ID`; the tool sidesteps this by quoting the exact
133
+ names `columns()` read back.
134
+ """
135
+ return '"' + identifier.replace('"', '""') + '"'
136
+
137
+ # ----------------------------------------------------------- rendering
138
+
139
+ def null_sentinel_sql(self) -> str:
140
+ r"""Build the sentinel from CHR(92), not a `\N` literal.
141
+
142
+ Snowflake processes backslash escapes in string literals, so a literal
143
+ is unsafe; CHR(92) is a backslash unconditionally and CHR returns a
144
+ varchar (not binary), so the coalesce stays text.
145
+ """
146
+ return "(chr(92) || 'N')"
147
+
148
+ def normalize(self, column: Column) -> str:
149
+ """Render one column as canonical text, null-safe.
150
+
151
+ DECIMAL/FLOAT match DuckDB's path - cast to NUMBER(38, scale) then to
152
+ text - so `1.5` becomes `'1.500000'`. Non-finite floats are a known
153
+ unverified gap: Snowflake's Inf/NaN detection differs from the other
154
+ engines and must be checked against a real account before this is
155
+ trusted for FLOAT columns holding them.
156
+ """
157
+ c = self.quote(column.name)
158
+ t = column.logical_type
159
+ if t is LogicalType.INTEGER:
160
+ expr = f"cast({c} as varchar)"
161
+ elif t in (LogicalType.DECIMAL, LogicalType.FLOAT):
162
+ expr = f"cast(cast({c} as number(38,{self.float_scale})) as varchar)"
163
+ elif t is LogicalType.BOOLEAN:
164
+ expr = f"case when {c} then 'true' when not {c} then 'false' end"
165
+ elif t is LogicalType.DATE:
166
+ expr = f"to_char({c}, 'YYYY-MM-DD')"
167
+ elif t is LogicalType.TIMESTAMP:
168
+ # FF6 is microseconds, matching the other engines' six digits.
169
+ expr = f"to_char({c}, 'YYYY-MM-DD HH24:MI:SS.FF6')"
170
+ else:
171
+ expr = f"cast({c} as varchar)"
172
+ return f"coalesce({expr}, {self.null_sentinel_sql()})"
173
+
174
+ def hash_expr(self, text_expr: str) -> str:
175
+ """Fold canonical text into a positive 60-bit integer.
176
+
177
+ `MD5_NUMBER_UPPER64` is the top 64 bits of the digest as an unsigned
178
+ number; the top 60 bits - the first 15 hex characters the other engines
179
+ take - are that floor-divided by 16.
180
+ """
181
+ return f"floor(md5_number_upper64({text_expr}) / 16)"
182
+
183
+ def int_div(self, numerator: str, denominator: str) -> str:
184
+ """Truncating integer division.
185
+
186
+ Operands are NUMBER (via `wide_int`), so `/` is exact decimal and
187
+ `floor` truncates exactly for the non-negative operands used here -
188
+ unlike a float `/`, which CLAUDE.md 4.5 warns loses precision.
189
+ """
190
+ return f"floor(({numerator}) / ({denominator}))"
191
+
192
+ def wide_int(self, expr: str) -> str:
193
+ """Widen past 64 bits before arithmetic. NUMBER(38,0) holds 38 digits,
194
+ which the key offset cannot overflow."""
195
+ return f"cast(({expr}) as number(38,0))"
196
+
197
+ def sum_wide(self, expr: str) -> str:
198
+ """Sum row hashes without overflowing, and return 0 for an empty group.
199
+
200
+ NUMBER(38,0) is far beyond any row count times 2^60.
201
+ """
202
+ return f"coalesce(sum(cast(({expr}) as number(38,0))), 0)"
203
+
204
+
205
+ def map_type_snowflake(raw: str, numeric_scale: Any) -> LogicalType:
206
+ """Map a Snowflake type onto a logical category.
207
+
208
+ Snowflake reports both integers and decimals as `NUMBER`; only the scale
209
+ tells them apart, so it is passed in. A NULL scale (non-numeric type) is
210
+ treated as not-an-integer.
211
+ """
212
+ t = raw.upper().split("(")[0].strip()
213
+ if t in {"NUMBER", "DECIMAL", "NUMERIC"}:
214
+ try:
215
+ scale = int(numeric_scale) if numeric_scale is not None else 6
216
+ except (TypeError, ValueError):
217
+ scale = 6
218
+ return LogicalType.INTEGER if scale == 0 else LogicalType.DECIMAL
219
+ if t in {"INT", "INTEGER", "BIGINT", "SMALLINT", "TINYINT", "BYTEINT"}:
220
+ return LogicalType.INTEGER # aliases that may appear via DATA_TYPE_ALIAS
221
+ if t in {"FLOAT", "FLOAT4", "FLOAT8", "DOUBLE", "DOUBLE PRECISION", "REAL"}:
222
+ return LogicalType.FLOAT
223
+ if t == "BOOLEAN":
224
+ return LogicalType.BOOLEAN
225
+ if t == "DATE":
226
+ return LogicalType.DATE
227
+ if t.startswith("TIMESTAMP") or t == "DATETIME":
228
+ return LogicalType.TIMESTAMP
229
+ if t in {"TEXT", "VARCHAR", "CHAR", "CHARACTER", "STRING"}:
230
+ return LogicalType.STRING
231
+ return LogicalType.UNKNOWN
@@ -132,6 +132,34 @@ def _key_order(key: int | str) -> tuple[int, int | str]:
132
132
  return (1, key) if isinstance(key, str) else (0, key)
133
133
 
134
134
 
135
+ def _fold_columns(
136
+ columns: list[Column], side: str, table: str
137
+ ) -> dict[str, Column]:
138
+ """Index a side's columns by case-folded name, for cross-engine matching.
139
+
140
+ Two engines fold unquoted identifiers to different cases, so a column is
141
+ the same column on both sides when its *folded* name matches. Each side
142
+ still holds its own `Column`, whose real stored name is what gets quoted
143
+ into SQL - only the matching is case-insensitive, never the rendering.
144
+
145
+ A table with two columns that differ only in case (possible only through
146
+ quoted identifiers) cannot be folded unambiguously, so it is refused with
147
+ a clear message rather than silently dropping one of them.
148
+ """
149
+ out: dict[str, Column] = {}
150
+ for c in columns:
151
+ fold = c.name.casefold()
152
+ if fold in out:
153
+ raise ValueError(
154
+ f"[side {side}] {table} has two columns that differ only in "
155
+ f"case: {out[fold].name!r} and {c.name!r}. parity matches "
156
+ f"columns case-insensitively across engines and cannot tell "
157
+ f"these apart - rename or quote one, or diff a view that does."
158
+ )
159
+ out[fold] = c
160
+ return out
161
+
162
+
135
163
  def _resolve_key(
136
164
  key: str | Sequence[str],
137
165
  cols_a: dict[str, Column],
@@ -149,22 +177,27 @@ def _resolve_key(
149
177
 
150
178
  Returns one spec per side. They agree on shape but hold each side's own
151
179
  `Column` objects, because a column can be `text` on one side and
152
- `varchar` on the other and each dialect renders its own.
180
+ `varchar` on the other and each dialect renders its own - and the key can
181
+ be `ID` on one side and `id` on the other, since `--key` is one name that
182
+ has to resolve against whatever case each engine stored.
183
+
184
+ `cols_a` and `cols_b` are keyed by case-folded name (see `_fold_columns`),
185
+ so the same `--key id` finds `id` on PostgreSQL and `ID` on Snowflake.
153
186
  """
154
187
  names = [key] if isinstance(key, str) else list(dict.fromkeys(key))
155
188
  if not names:
156
189
  raise ValueError("--key needs at least one column")
157
190
 
158
191
  for side, table, cols in (("A", a_table, cols_a), ("B", b_table, cols_b)):
159
- missing = [n for n in names if n not in cols]
192
+ missing = [n for n in names if n.casefold() not in cols]
160
193
  if missing:
161
194
  raise ValueError(
162
195
  f"[side {side}] key column(s) {missing} not in {table}. "
163
- f"Columns are: {sorted(cols)}"
196
+ f"Columns are: {sorted(c.name for c in cols.values())}"
164
197
  )
165
198
 
166
- a_key = tuple(cols_a[n] for n in names)
167
- b_key = tuple(cols_b[n] for n in names)
199
+ a_key = tuple(cols_a[n.casefold()] for n in names)
200
+ b_key = tuple(cols_b[n.casefold()] for n in names)
168
201
 
169
202
  # Hash unless it is one integer column on both sides. A single-column key
170
203
  # that is integer on one side and text on the other has to be hashed too,
@@ -194,21 +227,29 @@ def _resolve_key(
194
227
  def _select_columns(
195
228
  cols_a: dict[str, Column],
196
229
  cols_b: dict[str, Column],
197
- key_names: Sequence[str],
230
+ key_folds: set[str],
198
231
  columns: Sequence[str] | None,
199
232
  exclude: Sequence[str],
200
233
  warnings: list[str],
201
234
  ) -> list[str]:
202
235
  """Decide which columns to compare, explaining anything dropped.
203
236
 
237
+ `cols_a`/`cols_b` are keyed by case-folded name and the returned list is
238
+ folded names too, so matching is case-insensitive across engines (see
239
+ `_fold_columns`); the caller maps each folded name back to that side's real
240
+ `Column`. Messages echo the user's own tokens, or side A's stored name, so
241
+ a folded lookup never leaks a lower-cased identifier back at the reader.
242
+
204
243
  Key columns are never compared: they are how rows are matched up, not
205
244
  something compared between them.
206
245
  """
207
- keys = set(key_names)
208
- both = (set(cols_a) & set(cols_b)) - keys
209
- excluded = set(exclude)
246
+ both = (set(cols_a) & set(cols_b)) - key_folds
247
+ excluded = {e.casefold() for e in exclude}
210
248
 
211
- unknown_exclude = excluded - set(cols_a) - set(cols_b)
249
+ unknown_exclude = [
250
+ e for e in dict.fromkeys(exclude)
251
+ if e.casefold() not in cols_a and e.casefold() not in cols_b
252
+ ]
212
253
  if unknown_exclude:
213
254
  warnings.append(
214
255
  f"--exclude named columns that exist on neither side: "
@@ -217,13 +258,19 @@ def _select_columns(
217
258
 
218
259
  shared = sorted(both - excluded)
219
260
  if columns:
220
- requested = list(dict.fromkeys(columns)) # de-duplicate, keep order
221
- nowhere = [c for c in requested if c not in cols_a and c not in cols_b]
222
- one_side = [c for c in requested if c not in nowhere and c not in both]
223
- dropped = [c for c in requested if c in excluded]
261
+ # Fold for matching, but keep the first spelling the user gave each
262
+ # column so error messages read back their own words, not a fold.
263
+ token: dict[str, str] = {}
264
+ for c in columns:
265
+ token.setdefault(c.casefold(), c)
266
+ req = list(token) # folded, de-duplicated, order kept
267
+ nowhere_folds = {f for f in req if f not in cols_a and f not in cols_b}
268
+ nowhere = [token[f] for f in req if f in nowhere_folds]
269
+ one_side = [token[f] for f in req if f not in nowhere_folds and f not in both]
270
+ dropped = [token[f] for f in req if f in excluded]
224
271
  # Order matters: key columns are present on both sides but excluded
225
272
  # from `both`, so they would otherwise be misreported as one-sided.
226
- if named_keys := [c for c in requested if c in keys]:
273
+ if named_keys := [token[f] for f in req if f in key_folds]:
227
274
  raise ValueError(
228
275
  f"--columns named the key column(s) {named_keys}; the key is "
229
276
  f"how rows are matched up, not something compared between them"
@@ -238,26 +285,28 @@ def _select_columns(
238
285
  raise ValueError(
239
286
  f"--columns and --exclude both name: {dropped}"
240
287
  )
241
- shared = [c for c in shared if c in set(requested)]
288
+ shared = [f for f in shared if f in set(req)]
242
289
 
243
- for side, only in (
244
- ("A", sorted(set(cols_a) - set(cols_b) - keys)),
245
- ("B", sorted(set(cols_b) - set(cols_a) - keys)),
290
+ for side, cols, only in (
291
+ ("A", cols_a, sorted(set(cols_a) - set(cols_b) - key_folds)),
292
+ ("B", cols_b, sorted(set(cols_b) - set(cols_a) - key_folds)),
246
293
  ):
247
294
  if only:
248
- warnings.append(f"not compared, present only on side {side}: {only}")
295
+ names = sorted(cols[f].name for f in only)
296
+ warnings.append(f"not compared, present only on side {side}: {names}")
249
297
 
250
298
  # A column whose logical type differs between sides renders through a
251
299
  # different canonical encoding, so every row would report as changed. That
252
300
  # looks like a catastrophic data difference but is really a schema
253
301
  # difference, so name it explicitly.
254
- for name in shared:
255
- ta, tb = cols_a[name].logical_type, cols_b[name].logical_type
302
+ for f in shared:
303
+ name = cols_a[f].name
304
+ ta, tb = cols_a[f].logical_type, cols_b[f].logical_type
256
305
  if ta is tb or {ta, tb} <= _NUMERIC_EQUIVALENT:
257
306
  continue
258
307
  warnings.append(
259
- f"column {name!r} is {cols_a[name].raw_type or ta.value} on side A "
260
- f"but {cols_b[name].raw_type or tb.value} on side B; values are "
308
+ f"column {name!r} is {cols_a[f].raw_type or ta.value} on side A "
309
+ f"but {cols_b[f].raw_type or tb.value} on side B; values are "
261
310
  f"compared as text and will very likely all differ"
262
311
  )
263
312
 
@@ -267,21 +316,21 @@ def _select_columns(
267
316
  # are pinned to UTC, so a migration that stored UTC compares clean. One
268
317
  # that stored local wall-clock reports *every* row as different, and
269
318
  # without this line there is nothing pointing at which axis to look along.
270
- for name in shared:
271
- aware_a = _is_tz_aware(cols_a[name])
272
- if aware_a is _is_tz_aware(cols_b[name]):
319
+ for f in shared:
320
+ aware_a = _is_tz_aware(cols_a[f])
321
+ if aware_a is _is_tz_aware(cols_b[f]):
273
322
  continue
274
323
  aware, naive = ("A", "B") if aware_a else ("B", "A")
275
324
  warnings.append(
276
- f"column {name!r} is timezone-aware on side {aware} but not on "
277
- f"side {naive}; both are read in UTC, so a migration that stored "
278
- f"local wall-clock time rather than UTC will show every row as "
279
- f"different"
325
+ f"column {cols_a[f].name!r} is timezone-aware on side {aware} but "
326
+ f"not on side {naive}; both are read in UTC, so a migration that "
327
+ f"stored local wall-clock time rather than UTC will show every row "
328
+ f"as different"
280
329
  )
281
330
 
282
331
  unknown = sorted(
283
- {n for n in shared if cols_a[n].logical_type is LogicalType.UNKNOWN}
284
- | {n for n in shared if cols_b[n].logical_type is LogicalType.UNKNOWN}
332
+ {cols_a[f].name for f in shared if cols_a[f].logical_type is LogicalType.UNKNOWN}
333
+ | {cols_a[f].name for f in shared if cols_b[f].logical_type is LogicalType.UNKNOWN}
285
334
  )
286
335
  if unknown:
287
336
  warnings.append(
@@ -343,8 +392,15 @@ def diff(
343
392
  return _gather(fa, fb)
344
393
 
345
394
  cols_a_list, cols_b_list = both("columns")
346
- cols_a = {c.name: c for c in cols_a_list}
347
- cols_b = {c.name: c for c in cols_b_list}
395
+ # Match identifiers case-insensitively across the two sides. Engines
396
+ # fold unquoted names differently - Snowflake upper-cases, PostgreSQL
397
+ # and DuckDB lower-case - so a Postgres `amount` and its Snowflake
398
+ # `AMOUNT` are the same column and must line up, or the headline use
399
+ # case (diffing a table against its migration) finds no shared columns
400
+ # and no usable key. The map is keyed by the folded name; each side
401
+ # keeps its own Column, with its real stored name, for quoting in SQL.
402
+ cols_a = _fold_columns(cols_a_list, "A", a_table)
403
+ cols_b = _fold_columns(cols_b_list, "B", b_table)
348
404
  # Introspection is metadata, not a scan; do not inflate the query count
349
405
  # users read as "how much work did this cost".
350
406
  stats.queries -= 2
@@ -352,12 +408,13 @@ def diff(
352
408
  key_a, key_b = _resolve_key(
353
409
  key, cols_a, cols_b, a_table, b_table, warnings
354
410
  )
411
+ key_folds = {c.name.casefold() for c in key_a.columns}
355
412
 
356
413
  shared = _select_columns(
357
- cols_a, cols_b, key_a.names, columns, exclude, warnings
414
+ cols_a, cols_b, key_folds, columns, exclude, warnings
358
415
  )
359
- a_cols = [cols_a[c] for c in shared]
360
- b_cols = [cols_b[c] for c in shared]
416
+ a_cols = [cols_a[f] for f in shared]
417
+ b_cols = [cols_b[f] for f in shared]
361
418
 
362
419
  ks_a, ks_b = both("key_stats", key_a, _args_b=(key_b,))
363
420