parity-diff 0.2.1__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {parity_diff-0.2.1/src/parity_diff.egg-info → parity_diff-0.2.2}/PKG-INFO +11 -3
  2. {parity_diff-0.2.1 → parity_diff-0.2.2}/README.md +7 -2
  3. {parity_diff-0.2.1 → parity_diff-0.2.2}/pyproject.toml +8 -2
  4. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/__init__.py +1 -1
  5. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/dialects/base.py +5 -1
  6. parity_diff-0.2.2/src/parity/dialects/snowflake_dialect.py +231 -0
  7. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/engine.py +95 -38
  8. {parity_diff-0.2.1 → parity_diff-0.2.2/src/parity_diff.egg-info}/PKG-INFO +11 -3
  9. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity_diff.egg-info/SOURCES.txt +4 -1
  10. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity_diff.egg-info/requires.txt +4 -0
  11. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/conftest.py +39 -0
  12. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_engine.py +51 -0
  13. parity_diff-0.2.2/tests/test_snowflake.py +181 -0
  14. parity_diff-0.2.2/tests/test_snowflake_offline.py +113 -0
  15. {parity_diff-0.2.1 → parity_diff-0.2.2}/CONTRIBUTING.md +0 -0
  16. {parity_diff-0.2.1 → parity_diff-0.2.2}/LICENSE +0 -0
  17. {parity_diff-0.2.1 → parity_diff-0.2.2}/MANIFEST.in +0 -0
  18. {parity_diff-0.2.1 → parity_diff-0.2.2}/setup.cfg +0 -0
  19. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/cli.py +0 -0
  20. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/dialects/__init__.py +0 -0
  21. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/dialects/duckdb_dialect.py +0 -0
  22. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/dialects/mysql_dialect.py +0 -0
  23. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/dialects/postgres_dialect.py +0 -0
  24. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity/types.py +0 -0
  25. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity_diff.egg-info/dependency_links.txt +0 -0
  26. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity_diff.egg-info/entry_points.txt +0 -0
  27. {parity_diff-0.2.1 → parity_diff-0.2.2}/src/parity_diff.egg-info/top_level.txt +0 -0
  28. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/fakes.py +0 -0
  29. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_cli.py +0 -0
  30. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_encoding.py +0 -0
  31. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_fuzz_encoding.py +0 -0
  32. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_integration.py +0 -0
  33. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_mysql.py +0 -0
  34. {parity_diff-0.2.1 → parity_diff-0.2.2}/tests/test_properties.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parity-diff
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
5
5
  Author: Alessio Sorio
6
6
  License-Expression: MIT
@@ -30,10 +30,13 @@ Provides-Extra: postgres
30
30
  Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
31
31
  Provides-Extra: mysql
32
32
  Requires-Dist: mysql-connector-python>=8.0; extra == "mysql"
33
+ Provides-Extra: snowflake
34
+ Requires-Dist: snowflake-connector-python>=3.0; extra == "snowflake"
33
35
  Provides-Extra: all
34
36
  Requires-Dist: duckdb>=1.0; extra == "all"
35
37
  Requires-Dist: psycopg[binary]>=3.1; extra == "all"
36
38
  Requires-Dist: mysql-connector-python>=8.0; extra == "all"
39
+ Requires-Dist: snowflake-connector-python>=3.0; extra == "all"
37
40
  Provides-Extra: test
38
41
  Requires-Dist: hypothesis>=6.0; extra == "test"
39
42
  Dynamic: license-file
@@ -127,6 +130,7 @@ pip install "parity-diff[all]" # every engine
127
130
  pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
128
131
  pip install "parity-diff[postgres]" # PostgreSQL only
129
132
  pip install "parity-diff[mysql]" # MySQL only
133
+ pip install "parity-diff[snowflake]" # Snowflake only
130
134
  ```
131
135
 
132
136
  > The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
@@ -288,7 +292,8 @@ stated plainly rather than buried.
288
292
  | PostgreSQL | supported (tested against 16 and 18) |
289
293
  | DuckDB | supported (tested against 1.5) |
290
294
  | MySQL | supported (tested against 8.0) |
291
- | Snowflake, BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
295
+ | Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
296
+ | BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
292
297
 
293
298
  ## Scope
294
299
 
@@ -336,7 +341,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
336
341
 
337
342
  ## Changelog
338
343
 
339
- See `CHANGELOG.md`.
344
+ Latest release **0.2.1** — a generative test suite (property-based, oracle, and
345
+ cross-engine fuzzing) and the correctness fix it surfaced: a same-content
346
+ insert-plus-delete falling in one bucket that could previously read as
347
+ identical. See `CHANGELOG.md` for the full history.
340
348
 
341
349
  ## License
342
350
 
@@ -87,6 +87,7 @@ pip install "parity-diff[all]" # every engine
87
87
  pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
88
88
  pip install "parity-diff[postgres]" # PostgreSQL only
89
89
  pip install "parity-diff[mysql]" # MySQL only
90
+ pip install "parity-diff[snowflake]" # Snowflake only
90
91
  ```
91
92
 
92
93
  > The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
@@ -248,7 +249,8 @@ stated plainly rather than buried.
248
249
  | PostgreSQL | supported (tested against 16 and 18) |
249
250
  | DuckDB | supported (tested against 1.5) |
250
251
  | MySQL | supported (tested against 8.0) |
251
- | Snowflake, BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
252
+ | Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
253
+ | BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
252
254
 
253
255
  ## Scope
254
256
 
@@ -296,7 +298,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
296
298
 
297
299
  ## Changelog
298
300
 
299
- See `CHANGELOG.md`.
301
+ Latest release **0.2.1** — a generative test suite (property-based, oracle, and
302
+ cross-engine fuzzing) and the correctness fix it surfaced: a same-content
303
+ insert-plus-delete falling in one bucket that could previously read as
304
+ identical. See `CHANGELOG.md` for the full history.
300
305
 
301
306
  ## License
302
307
 
@@ -3,7 +3,7 @@
3
3
  # PyPI by an empty project. The import name, the CLI command and the repo are
4
4
  # all still `parity`: `pip install parity-diff` gives you `parity ...`.
5
5
  name = "parity-diff"
6
- version = "0.2.1"
6
+ version = "0.2.2"
7
7
  description = "Prove two tables in two different database engines hold the same data - without moving the data out of either engine."
8
8
  readme = "README.md"
9
9
  license = "MIT"
@@ -37,7 +37,11 @@ dependencies = []
37
37
  duckdb = ["duckdb>=1.0"]
38
38
  postgres = ["psycopg[binary]>=3.1"]
39
39
  mysql = ["mysql-connector-python>=8.0"]
40
- all = ["duckdb>=1.0", "psycopg[binary]>=3.1", "mysql-connector-python>=8.0"]
40
+ snowflake = ["snowflake-connector-python>=3.0"]
41
+ all = [
42
+ "duckdb>=1.0", "psycopg[binary]>=3.1",
43
+ "mysql-connector-python>=8.0", "snowflake-connector-python>=3.0",
44
+ ]
41
45
  # Test-only dependencies, kept out of every runtime extra. Hypothesis drives
42
46
  # the generative suite (tests/test_properties.py); the core stays stdlib-only.
43
47
  test = ["hypothesis>=6.0"]
@@ -64,6 +68,7 @@ markers = [
64
68
  "postgres: requires a reachable PostgreSQL server",
65
69
  "mysql: requires a reachable MySQL server",
66
70
  "duckdb: requires the duckdb driver",
71
+ "snowflake: requires a reachable Snowflake account (PARITY_TEST_SNOWFLAKE)",
67
72
  ]
68
73
 
69
74
  [tool.ruff]
@@ -105,6 +110,7 @@ ignore = [
105
110
  "src/parity/dialects/base.py" = ["S608"]
106
111
  "src/parity/dialects/postgres_dialect.py" = ["S608"]
107
112
  "src/parity/dialects/mysql_dialect.py" = ["S608"]
113
+ "src/parity/dialects/snowflake_dialect.py" = ["S608"]
108
114
  # proof.py is kept close to the original author's script so it stays readable
109
115
  # next to the findings it produced; its terse one-line style is deliberate.
110
116
  "demo/proof.py" = ["E401", "E402", "E701", "E702", "B007", "S311", "S608"]
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  from typing import Any
15
15
 
16
- __version__ = "0.2.1"
16
+ __version__ = "0.2.2"
17
17
 
18
18
  __all__ = ["__version__", "diff", "get_dialect"]
19
19
 
@@ -551,10 +551,14 @@ def get_dialect(
551
551
  from parity.dialects.mysql_dialect import MySQLDialect
552
552
 
553
553
  dialect = MySQLDialect(float_scale=float_scale, side=side)
554
+ elif scheme in ("snowflake",):
555
+ from parity.dialects.snowflake_dialect import SnowflakeDialect
556
+
557
+ dialect = SnowflakeDialect(float_scale=float_scale, side=side)
554
558
  else:
555
559
  raise ValueError(
556
560
  f"[side {side}] no dialect for scheme {scheme!r}. "
557
- f"Supported: duckdb, postgres, mysql."
561
+ f"Supported: duckdb, postgres, mysql, snowflake."
558
562
  )
559
563
  try:
560
564
  dialect.connect(connection_string)
@@ -0,0 +1,231 @@
1
+ """Snowflake dialect.
2
+
3
+ Verified against a live Snowflake account (AWS eu-central-2, 2026-09-07): the
4
+ hash constant agrees, a full-table checksum matches DuckDB byte-for-byte over
5
+ 5,000 mixed-type rows, and `tests/test_snowflake.py` passes end to end - the
6
+ identical check, a changed decimal, a deleted row, and the NULL-versus-empty
7
+ trap. It was first written against Snowflake's documentation; the live run then
8
+ turned up the one thing docs could not: Snowflake upper-cases unquoted
9
+ identifiers, so the engine had to match keys and columns case-insensitively
10
+ (see `_fold_columns` in engine.py) for a Snowflake table to diff against a
11
+ lower-casing engine at all.
12
+
13
+ The Snowflake-specific decisions, each confirmed by that run:
14
+
15
+ - **No `CONV` and no bit-cast for the hash.** Snowflake has neither, but
16
+ `MD5_NUMBER_UPPER64(x)` returns the top 64 bits of the digest as an unsigned
17
+ number, and the top 60 bits - the first 15 hex characters the other engines
18
+ fold - are `FLOOR(that / 16)`. That should equal 648541476951500027 for
19
+ `'abc'`; a test must pin it.
20
+ - **Integer and decimal both report as `NUMBER`.** `information_schema` tells
21
+ them apart only by `numeric_scale` (0 = integer), so `columns()` reads the
22
+ scale rather than trusting `data_type` - otherwise an integer key would be
23
+ rendered as `42.000000` and never match another engine's `42`.
24
+ - **The NULL sentinel is built from `CHR(92)`.** Snowflake interprets
25
+ backslash escapes in string literals, so a literal `'\\N'` is unsafe the same
26
+ way it was on MySQL.
27
+ - **One-snapshot isolation is not available.** Snowflake offers only READ
28
+ COMMITTED, so unlike PostgreSQL the walk cannot be pinned to a single
29
+ snapshot; a source table mutating mid-diff can produce an inconsistent
30
+ result. This is a genuine limitation, documented rather than hidden.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ from typing import Any
36
+ from urllib.parse import parse_qs, unquote, urlparse
37
+
38
+ from parity.dialects.base import Dialect, sql_literal
39
+ from parity.types import Column, LogicalType
40
+
41
+
42
+ class SnowflakeDialect(Dialect):
43
+ name = "snowflake"
44
+ #: Set from the connection's schema segment. Snowflake folds unquoted names
45
+ #: to upper case, so an unqualified table is looked up in the connected
46
+ #: schema as stored.
47
+ default_schema = "PUBLIC"
48
+
49
+ def connect(self, connection_string: str) -> None: # pragma: no cover
50
+ """Open a connection, pinned to UTC.
51
+
52
+ URL grammar mirrors SQLAlchemy's Snowflake dialect:
53
+ ``snowflake://user:password@account/database/schema?warehouse=wh&role=r``
54
+
55
+ Note the honest gap: Snowflake supports only READ COMMITTED, so there
56
+ is no equivalent of PostgreSQL's REPEATABLE READ - the diff cannot hold
57
+ one snapshot across the walk. UTC is still pinned so timestamps render
58
+ deterministically.
59
+ """
60
+ import snowflake.connector
61
+
62
+ url = urlparse(connection_string)
63
+ parts = [p for p in url.path.split("/") if p]
64
+ database = parts[0] if parts else None
65
+ schema = parts[1] if len(parts) > 1 else None
66
+ params = parse_qs(url.query)
67
+
68
+ def opt(key: str) -> str | None:
69
+ """First value of a query parameter, or None."""
70
+ values = params.get(key)
71
+ return values[0] if values else None
72
+ if schema:
73
+ self.default_schema = schema
74
+
75
+ self._conn = snowflake.connector.connect(
76
+ account=url.hostname,
77
+ user=unquote(url.username) if url.username else None,
78
+ password=unquote(url.password) if url.password else None,
79
+ database=database,
80
+ schema=schema,
81
+ warehouse=opt("warehouse"),
82
+ role=opt("role"),
83
+ autocommit=True,
84
+ )
85
+ cur = self._conn.cursor()
86
+ try:
87
+ # Timestamps render through this; pin it so two accounts in
88
+ # different regions cannot disagree on the same instant.
89
+ cur.execute("alter session set timezone = 'UTC'")
90
+ finally:
91
+ cur.close()
92
+
93
+ def close(self) -> None: # pragma: no cover
94
+ """Close the connection."""
95
+ self._conn.close()
96
+
97
+ def query(self, sql: str) -> list[tuple[Any, ...]]: # pragma: no cover
98
+ """Run `sql` and return every row as a list of tuples."""
99
+ cur = self._conn.cursor()
100
+ try:
101
+ cur.execute(sql)
102
+ return list(cur.fetchall())
103
+ finally:
104
+ cur.close()
105
+
106
+ def columns(self, table: str) -> list[Column]:
107
+ """Introspect columns, reading numeric_scale to split NUMBER.
108
+
109
+ Snowflake reports every integer and decimal as `NUMBER`; only the scale
110
+ distinguishes them, so this cannot use the shared `columns()`.
111
+ """
112
+ schema, name = self.split_table(table, self.default_schema)
113
+ rows = self.query(
114
+ "select column_name, data_type, numeric_scale "
115
+ "from information_schema.columns "
116
+ f"where table_schema = {sql_literal(schema)} "
117
+ f"and table_name = {sql_literal(name)} "
118
+ "order by ordinal_position"
119
+ )
120
+ if not rows:
121
+ raise self._err(self._not_found(table, schema, name))
122
+ return [
123
+ Column(str(r[0]), map_type_snowflake(str(r[1]), r[2]), str(r[1]))
124
+ for r in rows
125
+ ]
126
+
127
+ def quote(self, identifier: str) -> str:
128
+ """Wrap an identifier in double quotes, doubling any it contains.
129
+
130
+ The injection boundary - names arrive from the command line. Snowflake
131
+ stores unquoted names upper-cased, so a lower-case `"id"` will not match
132
+ a column created as `ID`; the tool sidesteps this by quoting the exact
133
+ names `columns()` read back.
134
+ """
135
+ return '"' + identifier.replace('"', '""') + '"'
136
+
137
+ # ----------------------------------------------------------- rendering
138
+
139
+ def null_sentinel_sql(self) -> str:
140
+ r"""Build the sentinel from CHR(92), not a `\N` literal.
141
+
142
+ Snowflake processes backslash escapes in string literals, so a literal
143
+ is unsafe; CHR(92) is a backslash unconditionally and CHR returns a
144
+ varchar (not binary), so the coalesce stays text.
145
+ """
146
+ return "(chr(92) || 'N')"
147
+
148
+ def normalize(self, column: Column) -> str:
149
+ """Render one column as canonical text, null-safe.
150
+
151
+ DECIMAL/FLOAT match DuckDB's path - cast to NUMBER(38, scale) then to
152
+ text - so `1.5` becomes `'1.500000'`. Non-finite floats are a known
153
+ unverified gap: Snowflake's Inf/NaN detection differs from the other
154
+ engines and must be checked against a real account before this is
155
+ trusted for FLOAT columns holding them.
156
+ """
157
+ c = self.quote(column.name)
158
+ t = column.logical_type
159
+ if t is LogicalType.INTEGER:
160
+ expr = f"cast({c} as varchar)"
161
+ elif t in (LogicalType.DECIMAL, LogicalType.FLOAT):
162
+ expr = f"cast(cast({c} as number(38,{self.float_scale})) as varchar)"
163
+ elif t is LogicalType.BOOLEAN:
164
+ expr = f"case when {c} then 'true' when not {c} then 'false' end"
165
+ elif t is LogicalType.DATE:
166
+ expr = f"to_char({c}, 'YYYY-MM-DD')"
167
+ elif t is LogicalType.TIMESTAMP:
168
+ # FF6 is microseconds, matching the other engines' six digits.
169
+ expr = f"to_char({c}, 'YYYY-MM-DD HH24:MI:SS.FF6')"
170
+ else:
171
+ expr = f"cast({c} as varchar)"
172
+ return f"coalesce({expr}, {self.null_sentinel_sql()})"
173
+
174
+ def hash_expr(self, text_expr: str) -> str:
175
+ """Fold canonical text into a positive 60-bit integer.
176
+
177
+ `MD5_NUMBER_UPPER64` is the top 64 bits of the digest as an unsigned
178
+ number; the top 60 bits - the first 15 hex characters the other engines
179
+ take - are that floor-divided by 16.
180
+ """
181
+ return f"floor(md5_number_upper64({text_expr}) / 16)"
182
+
183
+ def int_div(self, numerator: str, denominator: str) -> str:
184
+ """Truncating integer division.
185
+
186
+ Operands are NUMBER (via `wide_int`), so `/` is exact decimal and
187
+ `floor` truncates exactly for the non-negative operands used here -
188
+ unlike a float `/`, which CLAUDE.md 4.5 warns loses precision.
189
+ """
190
+ return f"floor(({numerator}) / ({denominator}))"
191
+
192
+ def wide_int(self, expr: str) -> str:
193
+ """Widen past 64 bits before arithmetic. NUMBER(38,0) holds 38 digits,
194
+ which the key offset cannot overflow."""
195
+ return f"cast(({expr}) as number(38,0))"
196
+
197
+ def sum_wide(self, expr: str) -> str:
198
+ """Sum row hashes without overflowing, and return 0 for an empty group.
199
+
200
+ NUMBER(38,0) is far beyond any row count times 2^60.
201
+ """
202
+ return f"coalesce(sum(cast(({expr}) as number(38,0))), 0)"
203
+
204
+
205
+ def map_type_snowflake(raw: str, numeric_scale: Any) -> LogicalType:
206
+ """Map a Snowflake type onto a logical category.
207
+
208
+ Snowflake reports both integers and decimals as `NUMBER`; only the scale
209
+ tells them apart, so it is passed in. A NULL scale (non-numeric type) is
210
+ treated as not-an-integer.
211
+ """
212
+ t = raw.upper().split("(")[0].strip()
213
+ if t in {"NUMBER", "DECIMAL", "NUMERIC"}:
214
+ try:
215
+ scale = int(numeric_scale) if numeric_scale is not None else 6
216
+ except (TypeError, ValueError):
217
+ scale = 6
218
+ return LogicalType.INTEGER if scale == 0 else LogicalType.DECIMAL
219
+ if t in {"INT", "INTEGER", "BIGINT", "SMALLINT", "TINYINT", "BYTEINT"}:
220
+ return LogicalType.INTEGER # aliases that may appear via DATA_TYPE_ALIAS
221
+ if t in {"FLOAT", "FLOAT4", "FLOAT8", "DOUBLE", "DOUBLE PRECISION", "REAL"}:
222
+ return LogicalType.FLOAT
223
+ if t == "BOOLEAN":
224
+ return LogicalType.BOOLEAN
225
+ if t == "DATE":
226
+ return LogicalType.DATE
227
+ if t.startswith("TIMESTAMP") or t == "DATETIME":
228
+ return LogicalType.TIMESTAMP
229
+ if t in {"TEXT", "VARCHAR", "CHAR", "CHARACTER", "STRING"}:
230
+ return LogicalType.STRING
231
+ return LogicalType.UNKNOWN
@@ -132,6 +132,34 @@ def _key_order(key: int | str) -> tuple[int, int | str]:
132
132
  return (1, key) if isinstance(key, str) else (0, key)
133
133
 
134
134
 
135
+ def _fold_columns(
136
+ columns: list[Column], side: str, table: str
137
+ ) -> dict[str, Column]:
138
+ """Index a side's columns by case-folded name, for cross-engine matching.
139
+
140
+ Two engines fold unquoted identifiers to different cases, so a column is
141
+ the same column on both sides when its *folded* name matches. Each side
142
+ still holds its own `Column`, whose real stored name is what gets quoted
143
+ into SQL - only the matching is case-insensitive, never the rendering.
144
+
145
+ A table with two columns that differ only in case (possible only through
146
+ quoted identifiers) cannot be folded unambiguously, so it is refused with
147
+ a clear message rather than silently dropping one of them.
148
+ """
149
+ out: dict[str, Column] = {}
150
+ for c in columns:
151
+ fold = c.name.casefold()
152
+ if fold in out:
153
+ raise ValueError(
154
+ f"[side {side}] {table} has two columns that differ only in "
155
+ f"case: {out[fold].name!r} and {c.name!r}. parity matches "
156
+ f"columns case-insensitively across engines and cannot tell "
157
+ f"these apart - rename or quote one, or diff a view that does."
158
+ )
159
+ out[fold] = c
160
+ return out
161
+
162
+
135
163
  def _resolve_key(
136
164
  key: str | Sequence[str],
137
165
  cols_a: dict[str, Column],
@@ -149,22 +177,27 @@ def _resolve_key(
149
177
 
150
178
  Returns one spec per side. They agree on shape but hold each side's own
151
179
  `Column` objects, because a column can be `text` on one side and
152
- `varchar` on the other and each dialect renders its own.
180
+ `varchar` on the other and each dialect renders its own - and the key can
181
+ be `ID` on one side and `id` on the other, since `--key` is one name that
182
+ has to resolve against whatever case each engine stored.
183
+
184
+ `cols_a` and `cols_b` are keyed by case-folded name (see `_fold_columns`),
185
+ so the same `--key id` finds `id` on PostgreSQL and `ID` on Snowflake.
153
186
  """
154
187
  names = [key] if isinstance(key, str) else list(dict.fromkeys(key))
155
188
  if not names:
156
189
  raise ValueError("--key needs at least one column")
157
190
 
158
191
  for side, table, cols in (("A", a_table, cols_a), ("B", b_table, cols_b)):
159
- missing = [n for n in names if n not in cols]
192
+ missing = [n for n in names if n.casefold() not in cols]
160
193
  if missing:
161
194
  raise ValueError(
162
195
  f"[side {side}] key column(s) {missing} not in {table}. "
163
- f"Columns are: {sorted(cols)}"
196
+ f"Columns are: {sorted(c.name for c in cols.values())}"
164
197
  )
165
198
 
166
- a_key = tuple(cols_a[n] for n in names)
167
- b_key = tuple(cols_b[n] for n in names)
199
+ a_key = tuple(cols_a[n.casefold()] for n in names)
200
+ b_key = tuple(cols_b[n.casefold()] for n in names)
168
201
 
169
202
  # Hash unless it is one integer column on both sides. A single-column key
170
203
  # that is integer on one side and text on the other has to be hashed too,
@@ -194,21 +227,29 @@ def _resolve_key(
194
227
  def _select_columns(
195
228
  cols_a: dict[str, Column],
196
229
  cols_b: dict[str, Column],
197
- key_names: Sequence[str],
230
+ key_folds: set[str],
198
231
  columns: Sequence[str] | None,
199
232
  exclude: Sequence[str],
200
233
  warnings: list[str],
201
234
  ) -> list[str]:
202
235
  """Decide which columns to compare, explaining anything dropped.
203
236
 
237
+ `cols_a`/`cols_b` are keyed by case-folded name and the returned list is
238
+ folded names too, so matching is case-insensitive across engines (see
239
+ `_fold_columns`); the caller maps each folded name back to that side's real
240
+ `Column`. Messages echo the user's own tokens, or side A's stored name, so
241
+ a folded lookup never leaks a lower-cased identifier back at the reader.
242
+
204
243
  Key columns are never compared: they are how rows are matched up, not
205
244
  something compared between them.
206
245
  """
207
- keys = set(key_names)
208
- both = (set(cols_a) & set(cols_b)) - keys
209
- excluded = set(exclude)
246
+ both = (set(cols_a) & set(cols_b)) - key_folds
247
+ excluded = {e.casefold() for e in exclude}
210
248
 
211
- unknown_exclude = excluded - set(cols_a) - set(cols_b)
249
+ unknown_exclude = [
250
+ e for e in dict.fromkeys(exclude)
251
+ if e.casefold() not in cols_a and e.casefold() not in cols_b
252
+ ]
212
253
  if unknown_exclude:
213
254
  warnings.append(
214
255
  f"--exclude named columns that exist on neither side: "
@@ -217,13 +258,19 @@ def _select_columns(
217
258
 
218
259
  shared = sorted(both - excluded)
219
260
  if columns:
220
- requested = list(dict.fromkeys(columns)) # de-duplicate, keep order
221
- nowhere = [c for c in requested if c not in cols_a and c not in cols_b]
222
- one_side = [c for c in requested if c not in nowhere and c not in both]
223
- dropped = [c for c in requested if c in excluded]
261
+ # Fold for matching, but keep the first spelling the user gave each
262
+ # column so error messages read back their own words, not a fold.
263
+ token: dict[str, str] = {}
264
+ for c in columns:
265
+ token.setdefault(c.casefold(), c)
266
+ req = list(token) # folded, de-duplicated, order kept
267
+ nowhere_folds = {f for f in req if f not in cols_a and f not in cols_b}
268
+ nowhere = [token[f] for f in req if f in nowhere_folds]
269
+ one_side = [token[f] for f in req if f not in nowhere_folds and f not in both]
270
+ dropped = [token[f] for f in req if f in excluded]
224
271
  # Order matters: key columns are present on both sides but excluded
225
272
  # from `both`, so they would otherwise be misreported as one-sided.
226
- if named_keys := [c for c in requested if c in keys]:
273
+ if named_keys := [token[f] for f in req if f in key_folds]:
227
274
  raise ValueError(
228
275
  f"--columns named the key column(s) {named_keys}; the key is "
229
276
  f"how rows are matched up, not something compared between them"
@@ -238,26 +285,28 @@ def _select_columns(
238
285
  raise ValueError(
239
286
  f"--columns and --exclude both name: {dropped}"
240
287
  )
241
- shared = [c for c in shared if c in set(requested)]
288
+ shared = [f for f in shared if f in set(req)]
242
289
 
243
- for side, only in (
244
- ("A", sorted(set(cols_a) - set(cols_b) - keys)),
245
- ("B", sorted(set(cols_b) - set(cols_a) - keys)),
290
+ for side, cols, only in (
291
+ ("A", cols_a, sorted(set(cols_a) - set(cols_b) - key_folds)),
292
+ ("B", cols_b, sorted(set(cols_b) - set(cols_a) - key_folds)),
246
293
  ):
247
294
  if only:
248
- warnings.append(f"not compared, present only on side {side}: {only}")
295
+ names = sorted(cols[f].name for f in only)
296
+ warnings.append(f"not compared, present only on side {side}: {names}")
249
297
 
250
298
  # A column whose logical type differs between sides renders through a
251
299
  # different canonical encoding, so every row would report as changed. That
252
300
  # looks like a catastrophic data difference but is really a schema
253
301
  # difference, so name it explicitly.
254
- for name in shared:
255
- ta, tb = cols_a[name].logical_type, cols_b[name].logical_type
302
+ for f in shared:
303
+ name = cols_a[f].name
304
+ ta, tb = cols_a[f].logical_type, cols_b[f].logical_type
256
305
  if ta is tb or {ta, tb} <= _NUMERIC_EQUIVALENT:
257
306
  continue
258
307
  warnings.append(
259
- f"column {name!r} is {cols_a[name].raw_type or ta.value} on side A "
260
- f"but {cols_b[name].raw_type or tb.value} on side B; values are "
308
+ f"column {name!r} is {cols_a[f].raw_type or ta.value} on side A "
309
+ f"but {cols_b[f].raw_type or tb.value} on side B; values are "
261
310
  f"compared as text and will very likely all differ"
262
311
  )
263
312
 
@@ -267,21 +316,21 @@ def _select_columns(
267
316
  # are pinned to UTC, so a migration that stored UTC compares clean. One
268
317
  # that stored local wall-clock reports *every* row as different, and
269
318
  # without this line there is nothing pointing at which axis to look along.
270
- for name in shared:
271
- aware_a = _is_tz_aware(cols_a[name])
272
- if aware_a is _is_tz_aware(cols_b[name]):
319
+ for f in shared:
320
+ aware_a = _is_tz_aware(cols_a[f])
321
+ if aware_a is _is_tz_aware(cols_b[f]):
273
322
  continue
274
323
  aware, naive = ("A", "B") if aware_a else ("B", "A")
275
324
  warnings.append(
276
- f"column {name!r} is timezone-aware on side {aware} but not on "
277
- f"side {naive}; both are read in UTC, so a migration that stored "
278
- f"local wall-clock time rather than UTC will show every row as "
279
- f"different"
325
+ f"column {cols_a[f].name!r} is timezone-aware on side {aware} but "
326
+ f"not on side {naive}; both are read in UTC, so a migration that "
327
+ f"stored local wall-clock time rather than UTC will show every row "
328
+ f"as different"
280
329
  )
281
330
 
282
331
  unknown = sorted(
283
- {n for n in shared if cols_a[n].logical_type is LogicalType.UNKNOWN}
284
- | {n for n in shared if cols_b[n].logical_type is LogicalType.UNKNOWN}
332
+ {cols_a[f].name for f in shared if cols_a[f].logical_type is LogicalType.UNKNOWN}
333
+ | {cols_a[f].name for f in shared if cols_b[f].logical_type is LogicalType.UNKNOWN}
285
334
  )
286
335
  if unknown:
287
336
  warnings.append(
@@ -343,8 +392,15 @@ def diff(
343
392
  return _gather(fa, fb)
344
393
 
345
394
  cols_a_list, cols_b_list = both("columns")
346
- cols_a = {c.name: c for c in cols_a_list}
347
- cols_b = {c.name: c for c in cols_b_list}
395
+ # Match identifiers case-insensitively across the two sides. Engines
396
+ # fold unquoted names differently - Snowflake upper-cases, PostgreSQL
397
+ # and DuckDB lower-case - so a Postgres `amount` and its Snowflake
398
+ # `AMOUNT` are the same column and must line up, or the headline use
399
+ # case (diffing a table against its migration) finds no shared columns
400
+ # and no usable key. The map is keyed by the folded name; each side
401
+ # keeps its own Column, with its real stored name, for quoting in SQL.
402
+ cols_a = _fold_columns(cols_a_list, "A", a_table)
403
+ cols_b = _fold_columns(cols_b_list, "B", b_table)
348
404
  # Introspection is metadata, not a scan; do not inflate the query count
349
405
  # users read as "how much work did this cost".
350
406
  stats.queries -= 2
@@ -352,12 +408,13 @@ def diff(
352
408
  key_a, key_b = _resolve_key(
353
409
  key, cols_a, cols_b, a_table, b_table, warnings
354
410
  )
411
+ key_folds = {c.name.casefold() for c in key_a.columns}
355
412
 
356
413
  shared = _select_columns(
357
- cols_a, cols_b, key_a.names, columns, exclude, warnings
414
+ cols_a, cols_b, key_folds, columns, exclude, warnings
358
415
  )
359
- a_cols = [cols_a[c] for c in shared]
360
- b_cols = [cols_b[c] for c in shared]
416
+ a_cols = [cols_a[f] for f in shared]
417
+ b_cols = [cols_b[f] for f in shared]
361
418
 
362
419
  ks_a, ks_b = both("key_stats", key_a, _args_b=(key_b,))
363
420
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parity-diff
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
5
5
  Author: Alessio Sorio
6
6
  License-Expression: MIT
@@ -30,10 +30,13 @@ Provides-Extra: postgres
30
30
  Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
31
31
  Provides-Extra: mysql
32
32
  Requires-Dist: mysql-connector-python>=8.0; extra == "mysql"
33
+ Provides-Extra: snowflake
34
+ Requires-Dist: snowflake-connector-python>=3.0; extra == "snowflake"
33
35
  Provides-Extra: all
34
36
  Requires-Dist: duckdb>=1.0; extra == "all"
35
37
  Requires-Dist: psycopg[binary]>=3.1; extra == "all"
36
38
  Requires-Dist: mysql-connector-python>=8.0; extra == "all"
39
+ Requires-Dist: snowflake-connector-python>=3.0; extra == "all"
37
40
  Provides-Extra: test
38
41
  Requires-Dist: hypothesis>=6.0; extra == "test"
39
42
  Dynamic: license-file
@@ -127,6 +130,7 @@ pip install "parity-diff[all]" # every engine
127
130
  pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
128
131
  pip install "parity-diff[postgres]" # PostgreSQL only
129
132
  pip install "parity-diff[mysql]" # MySQL only
133
+ pip install "parity-diff[snowflake]" # Snowflake only
130
134
  ```
131
135
 
132
136
  > The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
@@ -288,7 +292,8 @@ stated plainly rather than buried.
288
292
  | PostgreSQL | supported (tested against 16 and 18) |
289
293
  | DuckDB | supported (tested against 1.5) |
290
294
  | MySQL | supported (tested against 8.0) |
291
- | Snowflake, BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
295
+ | Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
296
+ | BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
292
297
 
293
298
  ## Scope
294
299
 
@@ -336,7 +341,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
336
341
 
337
342
  ## Changelog
338
343
 
339
- See `CHANGELOG.md`.
344
+ Latest release **0.2.1** — a generative test suite (property-based, oracle, and
345
+ cross-engine fuzzing) and the correctness fix it surfaced: a same-content
346
+ insert-plus-delete falling in one bucket that could previously read as
347
+ identical. See `CHANGELOG.md` for the full history.
340
348
 
341
349
  ## License
342
350
 
@@ -12,6 +12,7 @@ src/parity/dialects/base.py
12
12
  src/parity/dialects/duckdb_dialect.py
13
13
  src/parity/dialects/mysql_dialect.py
14
14
  src/parity/dialects/postgres_dialect.py
15
+ src/parity/dialects/snowflake_dialect.py
15
16
  src/parity_diff.egg-info/PKG-INFO
16
17
  src/parity_diff.egg-info/SOURCES.txt
17
18
  src/parity_diff.egg-info/dependency_links.txt
@@ -26,4 +27,6 @@ tests/test_engine.py
26
27
  tests/test_fuzz_encoding.py
27
28
  tests/test_integration.py
28
29
  tests/test_mysql.py
29
- tests/test_properties.py
30
+ tests/test_properties.py
31
+ tests/test_snowflake.py
32
+ tests/test_snowflake_offline.py
@@ -3,6 +3,7 @@
3
3
  duckdb>=1.0
4
4
  psycopg[binary]>=3.1
5
5
  mysql-connector-python>=8.0
6
+ snowflake-connector-python>=3.0
6
7
 
7
8
  [duckdb]
8
9
  duckdb>=1.0
@@ -13,5 +14,8 @@ mysql-connector-python>=8.0
13
14
  [postgres]
14
15
  psycopg[binary]>=3.1
15
16
 
17
+ [snowflake]
18
+ snowflake-connector-python>=3.0
19
+
16
20
  [test]
17
21
  hypothesis>=6.0
@@ -67,6 +67,31 @@ def _mysql_available() -> tuple[bool, str]:
67
67
  return True, ""
68
68
 
69
69
 
70
+ #: Snowflake, if one is configured. There is no local Snowflake and no free
71
+ #: service container, so this is opt-in via an env var and skips otherwise -
72
+ #: it never runs in CI, only when someone points it at a real account.
73
+ SNOWFLAKE_URL = os.environ.get("PARITY_TEST_SNOWFLAKE", "")
74
+
75
+
76
+ def _snowflake_available() -> tuple[bool, str]:
77
+ """Can we reach Snowflake? Returns (yes/no, why not).
78
+
79
+ Gated on the env var first so an unconfigured run skips instantly without
80
+ importing a driver or opening a network connection.
81
+ """
82
+ if not SNOWFLAKE_URL:
83
+ return False, "PARITY_TEST_SNOWFLAKE is not set"
84
+ try:
85
+ import snowflake.connector # noqa: F401 - importing it is the probe
86
+ except ImportError: # pragma: no cover - depends on install extras
87
+ return False, "snowflake-connector-python is not installed"
88
+ try:
89
+ get_dialect(SNOWFLAKE_URL, side="A").close()
90
+ except Exception as exc: # pragma: no cover - depends on the account
91
+ return False, f"no Snowflake at the configured URL: {type(exc).__name__}"
92
+ return True, ""
93
+
94
+
70
95
  def _duckdb_available() -> tuple[bool, str]:
71
96
  """Is the duckdb driver installed? Returns (yes/no, why not)."""
72
97
  try:
@@ -94,6 +119,15 @@ def mysql_url() -> str:
94
119
  return MYSQL_URL
95
120
 
96
121
 
122
+ @pytest.fixture(scope="session")
123
+ def snowflake_url() -> str:
124
+ """The Snowflake endpoint, or skip if none is configured/reachable."""
125
+ ok, why = _snowflake_available()
126
+ if not ok:
127
+ pytest.skip(why)
128
+ return SNOWFLAKE_URL
129
+
130
+
97
131
  @pytest.fixture(scope="session")
98
132
  def duckdb_path(tmp_path_factory: pytest.TempPathFactory) -> str:
99
133
  """A scratch path for a DuckDB file, unique to this test session."""
@@ -128,3 +162,8 @@ def open_pg(url: str, side: str = "A", float_scale: int = 6):
128
162
  def open_mysql(url: str, side: str = "A", float_scale: int = 6):
129
163
  """Open a read-only, UTC-pinned MySQL dialect."""
130
164
  return get_dialect(url, side=side, float_scale=float_scale)
165
+
166
+
167
+ def open_snowflake(url: str, side: str = "A", float_scale: int = 6):
168
+ """Open a UTC-pinned Snowflake dialect (DRAFT - see snowflake_dialect.py)."""
169
+ return get_dialect(url, side=side, float_scale=float_scale)
@@ -884,3 +884,54 @@ def test_identity_and_bucket_are_different_expressions_when_hashed():
884
884
  assert d.key_bucket(text) != d.key_identity(text)
885
885
  assert "md5" in d.key_bucket(text)
886
886
  assert "md5" not in d.key_identity(text)
887
+
888
+
889
+ # ---------------------------------------------------------------------------
890
+ # Case-insensitive identifier matching across engines.
891
+ # ---------------------------------------------------------------------------
892
+
893
+
894
+ def _cased(names, rows, key_name, side):
895
+ """A fake side whose key and columns carry the given (cased) names."""
896
+ cols = [Column(n, LogicalType.STRING, "varchar") for n in names]
897
+ return FakeDialect(
898
+ DictTable(cols, dict(rows)),
899
+ side=side,
900
+ key_type_override=Column(key_name, LogicalType.INTEGER, "bigint"),
901
+ )
902
+
903
+
904
+ def test_columns_and_key_match_across_engines_that_fold_case_differently():
905
+ """Snowflake upper-cases unquoted identifiers; PostgreSQL and DuckDB
906
+ lower-case them. The same column and key must line up across the two sides
907
+ regardless of case, or a table and its migration would share no columns and
908
+ no usable key - the headline comparison this tool exists for.
909
+ """
910
+ rows = {1: ("10", "ok"), 2: ("20", "paid"), 3: ("30", "void")}
911
+ upper = _cased(["AMOUNT", "STATUS"], rows, "ID", "A") # Snowflake-style
912
+ lower = _cased(["amount", "status"], rows, "id", "B") # Postgres-style
913
+
914
+ # Identical data, differently-cased identifiers -> identical, zero download.
915
+ same = diff(upper, lower, "t", "t", "id")
916
+ assert same.identical, same.diffs
917
+ assert same.stats.rows_downloaded == 0
918
+
919
+ # A planted change is still found, and the key resolves even when --key is
920
+ # given in yet another case than either side stores.
921
+ changed_lower = _cased(["amount", "status"], {**rows, 2: ("999", "paid")}, "id", "B")
922
+ changed = diff(upper, changed_lower, "t", "t", "Id")
923
+ assert [(d.key, d.kind) for d in changed.diffs] == [(2, "different")]
924
+ # The differing column is named as side A stores it.
925
+ assert changed.diffs[0].columns == ["AMOUNT"]
926
+
927
+
928
+ def test_two_columns_differing_only_in_case_are_refused():
929
+ """A table with `Col` and `col` cannot be folded unambiguously, so parity
930
+ refuses it with a clear message rather than silently dropping one.
931
+ """
932
+ cols = [Column("Amount", LogicalType.STRING, "varchar"),
933
+ Column("amount", LogicalType.STRING, "varchar")]
934
+ a = FakeDialect(DictTable(cols, {1: ("a", "b")}), side="A")
935
+ b = FakeDialect(DictTable(list(COLS), {1: ("a", "b")}), side="B")
936
+ with pytest.raises(ValueError, match="differ only in case"):
937
+ diff(a, b, "t", "t", "id")
@@ -0,0 +1,181 @@
1
+ """Snowflake against DuckDB, end to end - the verification the draft needs.
2
+
3
+ DRAFT dialect (see src/parity/dialects/snowflake_dialect.py): Snowflake earns
4
+ the word "supported" only once this file passes against a live account, byte
5
+ for byte with DuckDB. It is the Snowflake twin of test_mysql.py, and the only
6
+ Snowflake-specific content is the fixture DDL and the three decisions the draft
7
+ made: the row hash goes through FLOOR(MD5_NUMBER_UPPER64(x)/16), integer and
8
+ decimal are split by numeric_scale, and the NULL sentinel is CHR(92)||'N'. Each
9
+ is verified below by the values agreeing with DuckDB, not by reading the SQL.
10
+
11
+ Skips cleanly unless PARITY_TEST_SNOWFLAKE points at a reachable account, so it
12
+ never runs in CI and never spends warehouse credits on its own. Run it with:
13
+
14
+ pip install -e ".[duckdb,snowflake]" pytest
15
+ set PARITY_TEST_SNOWFLAKE=snowflake://user:pw@account/PARITY_TEST/ENC?warehouse=PARITY_WH&role=PARITY_RO
16
+ pytest tests/test_snowflake.py -v
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import pytest
22
+ from conftest import duckdb_write, open_duckdb, open_snowflake
23
+
24
+ from parity.engine import diff
25
+
26
+ pytestmark = pytest.mark.snowflake
27
+
28
+ N = 5_000
29
+
30
+ #: Side B: the DuckDB reference. Types line up with the Snowflake table below -
31
+ #: a real boolean on both sides (Snowflake has one, unlike MySQL), a decimal,
32
+ #: a naive timestamp - so any difference the tool reports is a real one, not a
33
+ #: type mismatch.
34
+ DUCKDB_TABLE = f"""
35
+ create table orders as
36
+ select i::bigint as id,
37
+ (i % 97)::integer as customer_id,
38
+ ((i * 7 % 100000) / 100.0)::decimal(12,2) as amount,
39
+ case when i % 3 = 0 then 'paid'
40
+ when i % 3 = 1 then 'open' else 'void' end as status,
41
+ (i % 11 = 0) as is_refunded,
42
+ (timestamp '2024-01-01 00:00:00'
43
+ + (i % 86400) * interval '1 second') as created_at,
44
+ case when i % 13 = 0 then null
45
+ else 'note ' || i::varchar end as note
46
+ from generate_series(1, {N}) as s(i)
47
+ """
48
+
49
+ #: Side A: the Snowflake table. NUMBER(38,0) reports as an integer only because
50
+ #: its scale is 0 - the draft's columns() reads numeric_scale to tell it from
51
+ #: the decimal `amount`, and a bug there would render the key as `1.000000`.
52
+ SNOWFLAKE_TABLE = """
53
+ create or replace table orders (
54
+ id number(38,0),
55
+ customer_id number(38,0),
56
+ amount number(12,2),
57
+ status varchar,
58
+ is_refunded boolean,
59
+ created_at timestamp_ntz,
60
+ note varchar
61
+ )
62
+ """
63
+
64
+ #: GENERATOR makes N rows; row_number() turns them into a stable 1..N key. id
65
+ #: is materialised in the subquery so each derived column reads one fixed value
66
+ #: rather than calling the sequence generator again mid-row.
67
+ SNOWFLAKE_FILL = f"""
68
+ insert into orders
69
+ select id,
70
+ mod(id, 97),
71
+ (mod(id * 7, 100000) / 100.0),
72
+ case when mod(id, 3) = 0 then 'paid'
73
+ when mod(id, 3) = 1 then 'open' else 'void' end,
74
+ (mod(id, 11) = 0),
75
+ dateadd(second, mod(id, 86400), '2024-01-01 00:00:00'::timestamp_ntz),
76
+ case when mod(id, 13) = 0 then null else 'note ' || id::varchar end
77
+ from (
78
+ select row_number() over (order by seq4()) as id
79
+ from table(generator(rowcount => {N}))
80
+ )
81
+ """
82
+
83
+
84
+ @pytest.fixture(scope="module")
85
+ def duck_path(tmp_path_factory) -> str:
86
+ """Side B: the DuckDB reference table, built once and opened read-only."""
87
+ path = str(tmp_path_factory.mktemp("snowflake_it") / "b.duckdb")
88
+ con = duckdb_write(path)
89
+ try:
90
+ con.execute(DUCKDB_TABLE)
91
+ finally:
92
+ con.close()
93
+ return path
94
+
95
+
96
+ def _build_snowflake(snowflake_url: str, plant: str | None) -> None:
97
+ """Create the Snowflake `orders` table, optionally with one planted defect.
98
+
99
+ Uses its own connection and commits (autocommit is on), because the dialect
100
+ opened later must see the data. This is the *only* place the tests write to
101
+ Snowflake; the parity dialect itself issues SELECT only.
102
+ """
103
+ a = open_snowflake(snowflake_url, side="A")
104
+ cur = a._conn.cursor()
105
+ try:
106
+ cur.execute(SNOWFLAKE_TABLE)
107
+ cur.execute("truncate table orders")
108
+ cur.execute(SNOWFLAKE_FILL)
109
+ if plant == "changed":
110
+ cur.execute("update orders set amount = amount + 0.01 where id = 1234")
111
+ elif plant == "deleted":
112
+ cur.execute("delete from orders where id = 777")
113
+ elif plant == "null_trap":
114
+ # id 13 is a multiple of 13, so `note` is NULL on the DuckDB side;
115
+ # setting it to '' here plants the NULL-versus-empty-string trap.
116
+ cur.execute("update orders set note = '' where id = 13")
117
+ finally:
118
+ cur.close()
119
+ a.close()
120
+
121
+
122
+ def _diff(snowflake_url: str, duck_path: str, **kwargs):
123
+ """Diff the Snowflake orders table against the DuckDB one."""
124
+ a = open_snowflake(snowflake_url, side="A")
125
+ b = open_duckdb(duck_path, side="B")
126
+ try:
127
+ # Side A names the table as Snowflake stored it (unquoted -> ORDERS);
128
+ # the key is given lower-case on purpose, to exercise the engine's
129
+ # case-insensitive key/column matching against Snowflake's ID/AMOUNT.
130
+ return diff(a, b, "ORDERS", "main.orders", "id", **kwargs)
131
+ finally:
132
+ a.close()
133
+ b.close()
134
+
135
+
136
+ def test_the_hash_constant_agrees_with_the_other_engines(snowflake_url):
137
+ """The whole cross-engine contract in one number.
138
+
139
+ Snowflake reaches it through FLOOR(MD5_NUMBER_UPPER64(x)/16), a fourth
140
+ distinct path after PostgreSQL's bit-cast, DuckDB's hex-cast and MySQL's
141
+ CONV. If this disagrees, nothing else can be trusted.
142
+ """
143
+ a = open_snowflake(snowflake_url, side="A")
144
+ try:
145
+ got = a.query(f"select {a.hash_expr(chr(39) + 'abc' + chr(39))}")[0][0]
146
+ assert int(got) == 648541476951500027
147
+ finally:
148
+ a.close()
149
+
150
+
151
+ def test_identical_tables_match_and_download_nothing(snowflake_url, duck_path):
152
+ """The headline claim, cross-engine: agreement moves zero rows."""
153
+ _build_snowflake(snowflake_url, plant=None)
154
+ result = _diff(snowflake_url, duck_path)
155
+ assert result.identical
156
+ assert result.diffs == []
157
+ assert result.stats.rows_downloaded == 0
158
+
159
+
160
+ def test_a_changed_decimal_is_found_on_exactly_that_row(snowflake_url, duck_path):
161
+ """A one-cent change on one row is reported as that row, that column."""
162
+ _build_snowflake(snowflake_url, plant="changed")
163
+ result = _diff(snowflake_url, duck_path)
164
+ assert [(d.key, d.kind) for d in result.diffs] == [(1234, "different")]
165
+ # Columns are reported as side A (Snowflake) stores them - upper-cased.
166
+ assert result.diffs[0].columns == ["AMOUNT"]
167
+
168
+
169
+ def test_a_deleted_row_is_reported_only_in_b(snowflake_url, duck_path):
170
+ """A row missing from Snowflake is only_in_b, not an error."""
171
+ _build_snowflake(snowflake_url, plant="deleted")
172
+ result = _diff(snowflake_url, duck_path)
173
+ assert [(d.key, d.kind) for d in result.diffs] == [(777, "only_in_b")]
174
+
175
+
176
+ def test_null_versus_empty_string_is_caught(snowflake_url, duck_path):
177
+ """The trap naive tools miss: NULL on one side, '' on the other."""
178
+ _build_snowflake(snowflake_url, plant="null_trap")
179
+ result = _diff(snowflake_url, duck_path)
180
+ assert [(d.key, d.kind) for d in result.diffs] == [(13, "different")]
181
+ assert result.diffs[0].columns == ["NOTE"]
@@ -0,0 +1,113 @@
1
+ """Offline unit tests for the Snowflake dialect's SQL rendering.
2
+
3
+ The live end-to-end proof is `test_snowflake.py`, which needs a real account and
4
+ so cannot run in CI. These tests pin the *SQL the dialect generates* - every
5
+ rendering decision from CLAUDE.md section 4, in pure string form - so the logic
6
+ stays covered on every push without a database. Only `connect`/`close`/`query`,
7
+ which touch the live driver, are left to the credential-gated live test (they
8
+ carry `# pragma: no cover`); `columns()` is exercised here through a stubbed
9
+ `query`.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from parity.dialects.snowflake_dialect import SnowflakeDialect, map_type_snowflake
15
+ from parity.types import Column, LogicalType
16
+
17
+
18
+ def _d() -> SnowflakeDialect:
19
+ """A dialect instance; no connection is opened."""
20
+ return SnowflakeDialect(side="A")
21
+
22
+
23
+ def _col(name: str, t: LogicalType) -> Column:
24
+ """A column of the given logical type, for rendering."""
25
+ return Column(name, t, name)
26
+
27
+
28
+ def test_hash_expr_is_the_top_60_bits_of_the_md5_number():
29
+ """The whole cross-engine contract: 15 hex chars = top 64 bits, drop 4."""
30
+ assert _d().hash_expr("'abc'") == "floor(md5_number_upper64('abc') / 16)"
31
+
32
+
33
+ def test_the_null_sentinel_is_built_from_chr_92():
34
+ r"""A literal '\N' is unsafe on Snowflake (it processes backslash escapes)."""
35
+ assert _d().null_sentinel_sql() == "(chr(92) || 'N')"
36
+
37
+
38
+ def test_quote_doubles_embedded_quotes():
39
+ """The injection boundary - names arrive from the command line."""
40
+ assert _d().quote('a"b') == '"a""b"'
41
+
42
+
43
+ def test_integer_division_and_widening():
44
+ """`/` is exact on NUMBER and `floor` truncates; NUMBER(38,0) cannot overflow."""
45
+ d = _d()
46
+ assert d.int_div("x", "y") == "floor((x) / (y))"
47
+ assert d.wide_int("k") == "cast((k) as number(38,0))"
48
+ assert d.sum_wide("h") == "coalesce(sum(cast((h) as number(38,0))), 0)"
49
+
50
+
51
+ def test_normalize_renders_each_logical_type():
52
+ """Every branch of the canonical-text encoding, null-safe."""
53
+ d = _d()
54
+ sentinel = "(chr(92) || 'N')"
55
+ cases = {
56
+ LogicalType.INTEGER: 'cast("c" as varchar)',
57
+ LogicalType.DECIMAL: 'cast(cast("c" as number(38,6)) as varchar)',
58
+ LogicalType.FLOAT: 'cast(cast("c" as number(38,6)) as varchar)',
59
+ LogicalType.BOOLEAN: 'case when "c" then \'true\' when not "c" then \'false\' end',
60
+ LogicalType.DATE: 'to_char("c", \'YYYY-MM-DD\')',
61
+ LogicalType.TIMESTAMP: 'to_char("c", \'YYYY-MM-DD HH24:MI:SS.FF6\')',
62
+ LogicalType.STRING: 'cast("c" as varchar)',
63
+ LogicalType.UNKNOWN: 'cast("c" as varchar)',
64
+ }
65
+ for t, inner in cases.items():
66
+ assert d.normalize(_col("c", t)) == f"coalesce({inner}, {sentinel})", t
67
+
68
+
69
+ def test_map_type_splits_number_by_scale_and_maps_the_rest():
70
+ """Snowflake reports integer and decimal both as NUMBER; only scale tells
71
+ them apart, and a wrong split would render an integer key as `42.000000`."""
72
+ assert map_type_snowflake("NUMBER", 0) is LogicalType.INTEGER
73
+ assert map_type_snowflake("NUMBER(38,0)", 0) is LogicalType.INTEGER
74
+ assert map_type_snowflake("NUMBER", 2) is LogicalType.DECIMAL
75
+ assert map_type_snowflake("NUMBER", None) is LogicalType.DECIMAL # no scale -> not integer
76
+ assert map_type_snowflake("DECIMAL", 6) is LogicalType.DECIMAL
77
+ assert map_type_snowflake("INT", 0) is LogicalType.INTEGER
78
+ assert map_type_snowflake("FLOAT", None) is LogicalType.FLOAT
79
+ assert map_type_snowflake("DOUBLE", None) is LogicalType.FLOAT
80
+ assert map_type_snowflake("BOOLEAN", None) is LogicalType.BOOLEAN
81
+ assert map_type_snowflake("DATE", None) is LogicalType.DATE
82
+ assert map_type_snowflake("TIMESTAMP_NTZ", None) is LogicalType.TIMESTAMP
83
+ assert map_type_snowflake("DATETIME", None) is LogicalType.TIMESTAMP
84
+ assert map_type_snowflake("TEXT", None) is LogicalType.STRING
85
+ assert map_type_snowflake("VARCHAR", None) is LogicalType.STRING
86
+ assert map_type_snowflake("VARIANT", None) is LogicalType.UNKNOWN
87
+ # A non-integer scale value is treated defensively as decimal, not a crash.
88
+ assert map_type_snowflake("NUMBER", "oops") is LogicalType.DECIMAL
89
+
90
+
91
+ def test_columns_reads_numeric_scale_to_split_number():
92
+ """`columns()` turns information_schema rows into typed Columns, splitting
93
+ NUMBER on its scale - checked here through a stubbed query, no connection."""
94
+
95
+ class Stubbed(SnowflakeDialect):
96
+ def query(self, sql): # type: ignore[override]
97
+ """Return canned information_schema rows instead of hitting a DB."""
98
+ assert "information_schema.columns" in sql
99
+ assert "PARITY_TEST" in sql or "ENC" in sql or "ORDERS" in sql
100
+ return [
101
+ ("ID", "NUMBER", 0),
102
+ ("AMOUNT", "NUMBER", 2),
103
+ ("STATUS", "TEXT", None),
104
+ ("CREATED_AT", "TIMESTAMP_NTZ", None),
105
+ ]
106
+
107
+ cols = Stubbed(side="A").columns("ENC.ORDERS")
108
+ assert [(c.name, c.logical_type) for c in cols] == [
109
+ ("ID", LogicalType.INTEGER),
110
+ ("AMOUNT", LogicalType.DECIMAL),
111
+ ("STATUS", LogicalType.STRING),
112
+ ("CREATED_AT", LogicalType.TIMESTAMP),
113
+ ]
File without changes
File without changes
File without changes
File without changes
File without changes