parity-diff 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parity_diff-0.2.0/src/parity_diff.egg-info → parity_diff-0.2.2}/PKG-INFO +13 -3
- {parity_diff-0.2.0 → parity_diff-0.2.2}/README.md +7 -2
- {parity_diff-0.2.0 → parity_diff-0.2.2}/pyproject.toml +11 -2
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/__init__.py +1 -1
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/dialects/base.py +24 -5
- parity_diff-0.2.2/src/parity/dialects/snowflake_dialect.py +231 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/engine.py +114 -41
- {parity_diff-0.2.0 → parity_diff-0.2.2/src/parity_diff.egg-info}/PKG-INFO +13 -3
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity_diff.egg-info/SOURCES.txt +6 -1
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity_diff.egg-info/requires.txt +7 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/conftest.py +39 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/fakes.py +7 -1
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/test_engine.py +51 -0
- parity_diff-0.2.2/tests/test_fuzz_encoding.py +347 -0
- parity_diff-0.2.2/tests/test_properties.py +257 -0
- parity_diff-0.2.2/tests/test_snowflake.py +181 -0
- parity_diff-0.2.2/tests/test_snowflake_offline.py +113 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/CONTRIBUTING.md +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/LICENSE +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/MANIFEST.in +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/setup.cfg +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/cli.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/dialects/__init__.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/dialects/duckdb_dialect.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/dialects/mysql_dialect.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/dialects/postgres_dialect.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity/types.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity_diff.egg-info/dependency_links.txt +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity_diff.egg-info/entry_points.txt +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/src/parity_diff.egg-info/top_level.txt +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/test_cli.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/test_encoding.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/test_integration.py +0 -0
- {parity_diff-0.2.0 → parity_diff-0.2.2}/tests/test_mysql.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parity-diff
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Prove two tables in two different database engines hold the same data - without moving the data out of either engine.
|
|
5
5
|
Author: Alessio Sorio
|
|
6
6
|
License-Expression: MIT
|
|
@@ -30,10 +30,15 @@ Provides-Extra: postgres
|
|
|
30
30
|
Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
|
|
31
31
|
Provides-Extra: mysql
|
|
32
32
|
Requires-Dist: mysql-connector-python>=8.0; extra == "mysql"
|
|
33
|
+
Provides-Extra: snowflake
|
|
34
|
+
Requires-Dist: snowflake-connector-python>=3.0; extra == "snowflake"
|
|
33
35
|
Provides-Extra: all
|
|
34
36
|
Requires-Dist: duckdb>=1.0; extra == "all"
|
|
35
37
|
Requires-Dist: psycopg[binary]>=3.1; extra == "all"
|
|
36
38
|
Requires-Dist: mysql-connector-python>=8.0; extra == "all"
|
|
39
|
+
Requires-Dist: snowflake-connector-python>=3.0; extra == "all"
|
|
40
|
+
Provides-Extra: test
|
|
41
|
+
Requires-Dist: hypothesis>=6.0; extra == "test"
|
|
37
42
|
Dynamic: license-file
|
|
38
43
|
|
|
39
44
|
# parity
|
|
@@ -125,6 +130,7 @@ pip install "parity-diff[all]" # every engine
|
|
|
125
130
|
pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
|
|
126
131
|
pip install "parity-diff[postgres]" # PostgreSQL only
|
|
127
132
|
pip install "parity-diff[mysql]" # MySQL only
|
|
133
|
+
pip install "parity-diff[snowflake]" # Snowflake only
|
|
128
134
|
```
|
|
129
135
|
|
|
130
136
|
> The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
|
|
@@ -286,7 +292,8 @@ stated plainly rather than buried.
|
|
|
286
292
|
| PostgreSQL | supported (tested against 16 and 18) |
|
|
287
293
|
| DuckDB | supported (tested against 1.5) |
|
|
288
294
|
| MySQL | supported (tested against 8.0) |
|
|
289
|
-
| Snowflake
|
|
295
|
+
| Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
|
|
296
|
+
| BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
|
|
290
297
|
|
|
291
298
|
## Scope
|
|
292
299
|
|
|
@@ -334,7 +341,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
|
|
|
334
341
|
|
|
335
342
|
## Changelog
|
|
336
343
|
|
|
337
|
-
|
|
344
|
+
Latest release **0.2.1** — a generative test suite (property-based, oracle, and
|
|
345
|
+
cross-engine fuzzing) and the correctness fix it surfaced: a same-content
|
|
346
|
+
insert-plus-delete falling in one bucket that could previously read as
|
|
347
|
+
identical. See `CHANGELOG.md` for the full history.
|
|
338
348
|
|
|
339
349
|
## License
|
|
340
350
|
|
|
@@ -87,6 +87,7 @@ pip install "parity-diff[all]" # every engine
|
|
|
87
87
|
pip install "parity-diff[duckdb]" # DuckDB only — no other driver pulled in
|
|
88
88
|
pip install "parity-diff[postgres]" # PostgreSQL only
|
|
89
89
|
pip install "parity-diff[mysql]" # MySQL only
|
|
90
|
+
pip install "parity-diff[snowflake]" # Snowflake only
|
|
90
91
|
```
|
|
91
92
|
|
|
92
93
|
> The PyPI distribution is `parity-diff` — plain `parity` is squatted by an
|
|
@@ -248,7 +249,8 @@ stated plainly rather than buried.
|
|
|
248
249
|
| PostgreSQL | supported (tested against 16 and 18) |
|
|
249
250
|
| DuckDB | supported (tested against 1.5) |
|
|
250
251
|
| MySQL | supported (tested against 8.0) |
|
|
251
|
-
| Snowflake
|
|
252
|
+
| Snowflake | supported (verified live on AWS; `pip install "parity-diff[snowflake]"`) |
|
|
253
|
+
| BigQuery, Redshift | not yet — see `CONTRIBUTING.md`, a dialect is ~80 lines |
|
|
252
254
|
|
|
253
255
|
## Scope
|
|
254
256
|
|
|
@@ -296,7 +298,10 @@ way it is. `ROADMAP.md` is what is done and what comes next.
|
|
|
296
298
|
|
|
297
299
|
## Changelog
|
|
298
300
|
|
|
299
|
-
|
|
301
|
+
Latest release **0.2.1** — a generative test suite (property-based, oracle, and
|
|
302
|
+
cross-engine fuzzing) and the correctness fix it surfaced: a same-content
|
|
303
|
+
insert-plus-delete falling in one bucket that could previously read as
|
|
304
|
+
identical. See `CHANGELOG.md` for the full history.
|
|
300
305
|
|
|
301
306
|
## License
|
|
302
307
|
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
# PyPI by an empty project. The import name, the CLI command and the repo are
|
|
4
4
|
# all still `parity`: `pip install parity-diff` gives you `parity ...`.
|
|
5
5
|
name = "parity-diff"
|
|
6
|
-
version = "0.2.
|
|
6
|
+
version = "0.2.2"
|
|
7
7
|
description = "Prove two tables in two different database engines hold the same data - without moving the data out of either engine."
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
license = "MIT"
|
|
@@ -37,7 +37,14 @@ dependencies = []
|
|
|
37
37
|
duckdb = ["duckdb>=1.0"]
|
|
38
38
|
postgres = ["psycopg[binary]>=3.1"]
|
|
39
39
|
mysql = ["mysql-connector-python>=8.0"]
|
|
40
|
-
|
|
40
|
+
snowflake = ["snowflake-connector-python>=3.0"]
|
|
41
|
+
all = [
|
|
42
|
+
"duckdb>=1.0", "psycopg[binary]>=3.1",
|
|
43
|
+
"mysql-connector-python>=8.0", "snowflake-connector-python>=3.0",
|
|
44
|
+
]
|
|
45
|
+
# Test-only dependencies, kept out of every runtime extra. Hypothesis drives
|
|
46
|
+
# the generative suite (tests/test_properties.py); the core stays stdlib-only.
|
|
47
|
+
test = ["hypothesis>=6.0"]
|
|
41
48
|
|
|
42
49
|
[project.urls]
|
|
43
50
|
Homepage = "https://github.com/Aleixiou/parity-diff"
|
|
@@ -61,6 +68,7 @@ markers = [
|
|
|
61
68
|
"postgres: requires a reachable PostgreSQL server",
|
|
62
69
|
"mysql: requires a reachable MySQL server",
|
|
63
70
|
"duckdb: requires the duckdb driver",
|
|
71
|
+
"snowflake: requires a reachable Snowflake account (PARITY_TEST_SNOWFLAKE)",
|
|
64
72
|
]
|
|
65
73
|
|
|
66
74
|
[tool.ruff]
|
|
@@ -102,6 +110,7 @@ ignore = [
|
|
|
102
110
|
"src/parity/dialects/base.py" = ["S608"]
|
|
103
111
|
"src/parity/dialects/postgres_dialect.py" = ["S608"]
|
|
104
112
|
"src/parity/dialects/mysql_dialect.py" = ["S608"]
|
|
113
|
+
"src/parity/dialects/snowflake_dialect.py" = ["S608"]
|
|
105
114
|
# proof.py is kept close to the original author's script so it stays readable
|
|
106
115
|
# next to the findings it produced; its terse one-line style is deliberate.
|
|
107
116
|
"demo/proof.py" = ["E401", "E402", "E701", "E702", "B007", "S311", "S608"]
|
|
@@ -275,9 +275,10 @@ class Dialect(ABC):
|
|
|
275
275
|
# Two tables can legitimately share only their key - after
|
|
276
276
|
# `--columns`/`--exclude`, or when the schemas have diverged
|
|
277
277
|
# entirely. `concat_ws(chr(31), )` is a syntax error, so render a
|
|
278
|
-
# constant instead.
|
|
279
|
-
# `
|
|
280
|
-
#
|
|
278
|
+
# constant instead. The checksum query never passes an empty column
|
|
279
|
+
# set here: `segment_checksums` always folds the key in, so in that
|
|
280
|
+
# mode the key itself is what gets hashed. This branch stays as a
|
|
281
|
+
# defensive fallback for any other caller.
|
|
281
282
|
return "''"
|
|
282
283
|
return self._concat([self.normalize(c) for c in columns])
|
|
283
284
|
|
|
@@ -459,9 +460,23 @@ class Dialect(ABC):
|
|
|
459
460
|
# the one function whose off-by-one would make the walker skip rows.
|
|
460
461
|
offset = f"({self.wide_int(k)} - ({lo}))"
|
|
461
462
|
bucket = self.int_div(f"{offset} * {n_segments}", f"({hi - lo})")
|
|
463
|
+
# Fold the *key* into every row's hash, not just the comparable columns.
|
|
464
|
+
# A count plus a content-only sum cannot see a same-content insert and
|
|
465
|
+
# delete in one bucket: the counts balance (one in, one out) and equal
|
|
466
|
+
# content sums to the same value, so the bucket reads clean - a false
|
|
467
|
+
# "identical", on data as ordinary as two rows sharing a status or an
|
|
468
|
+
# empty string. Including the key makes an inserted key and a deleted
|
|
469
|
+
# key hash to different values, so the bucket sum changes and the walker
|
|
470
|
+
# recurses in. This only ever *adds* sensitivity: a checksum that
|
|
471
|
+
# differs is always re-checked by downloading and comparing the real
|
|
472
|
+
# rows, so an incidental mismatch costs a query, never a wrong verdict.
|
|
473
|
+
# It also subsumes the no-comparable-columns mode, where the key becomes
|
|
474
|
+
# the only thing hashed. `columns` never contains the key, so the key is
|
|
475
|
+
# hashed exactly once.
|
|
476
|
+
content = self.row_hash([*key.columns, *columns])
|
|
462
477
|
return (
|
|
463
478
|
f"select {bucket} as seg, count(*), "
|
|
464
|
-
f"{self.sum_wide(
|
|
479
|
+
f"{self.sum_wide(content)} "
|
|
465
480
|
f"from {self.qualify(table)} "
|
|
466
481
|
f"where {k} >= {lo} and {k} <= {hi - 1} "
|
|
467
482
|
f"group by 1"
|
|
@@ -536,10 +551,14 @@ def get_dialect(
|
|
|
536
551
|
from parity.dialects.mysql_dialect import MySQLDialect
|
|
537
552
|
|
|
538
553
|
dialect = MySQLDialect(float_scale=float_scale, side=side)
|
|
554
|
+
elif scheme in ("snowflake",):
|
|
555
|
+
from parity.dialects.snowflake_dialect import SnowflakeDialect
|
|
556
|
+
|
|
557
|
+
dialect = SnowflakeDialect(float_scale=float_scale, side=side)
|
|
539
558
|
else:
|
|
540
559
|
raise ValueError(
|
|
541
560
|
f"[side {side}] no dialect for scheme {scheme!r}. "
|
|
542
|
-
f"Supported: duckdb, postgres, mysql."
|
|
561
|
+
f"Supported: duckdb, postgres, mysql, snowflake."
|
|
543
562
|
)
|
|
544
563
|
try:
|
|
545
564
|
dialect.connect(connection_string)
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Snowflake dialect.
|
|
2
|
+
|
|
3
|
+
Verified against a live Snowflake account (AWS eu-central-2, 2026-09-07): the
|
|
4
|
+
hash constant agrees, a full-table checksum matches DuckDB byte-for-byte over
|
|
5
|
+
5,000 mixed-type rows, and `tests/test_snowflake.py` passes end to end - the
|
|
6
|
+
identical check, a changed decimal, a deleted row, and the NULL-versus-empty
|
|
7
|
+
trap. It was first written against Snowflake's documentation; the live run then
|
|
8
|
+
turned up the one thing docs could not: Snowflake upper-cases unquoted
|
|
9
|
+
identifiers, so the engine had to match keys and columns case-insensitively
|
|
10
|
+
(see `_fold_columns` in engine.py) for a Snowflake table to diff against a
|
|
11
|
+
lower-casing engine at all.
|
|
12
|
+
|
|
13
|
+
The Snowflake-specific decisions, each confirmed by that run:
|
|
14
|
+
|
|
15
|
+
- **No `CONV` and no bit-cast for the hash.** Snowflake has neither, but
|
|
16
|
+
`MD5_NUMBER_UPPER64(x)` returns the top 64 bits of the digest as an unsigned
|
|
17
|
+
number, and the top 60 bits - the first 15 hex characters the other engines
|
|
18
|
+
fold - are `FLOOR(that / 16)`. That should equal 648541476951500027 for
|
|
19
|
+
`'abc'`; a test must pin it.
|
|
20
|
+
- **Integer and decimal both report as `NUMBER`.** `information_schema` tells
|
|
21
|
+
them apart only by `numeric_scale` (0 = integer), so `columns()` reads the
|
|
22
|
+
scale rather than trusting `data_type` - otherwise an integer key would be
|
|
23
|
+
rendered as `42.000000` and never match another engine's `42`.
|
|
24
|
+
- **The NULL sentinel is built from `CHR(92)`.** Snowflake interprets
|
|
25
|
+
backslash escapes in string literals, so a literal `'\\N'` is unsafe the same
|
|
26
|
+
way it was on MySQL.
|
|
27
|
+
- **One-snapshot isolation is not available.** Snowflake offers only READ
|
|
28
|
+
COMMITTED, so unlike PostgreSQL the walk cannot be pinned to a single
|
|
29
|
+
snapshot; a source table mutating mid-diff can produce an inconsistent
|
|
30
|
+
result. This is a genuine limitation, documented rather than hidden.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from typing import Any
|
|
36
|
+
from urllib.parse import parse_qs, unquote, urlparse
|
|
37
|
+
|
|
38
|
+
from parity.dialects.base import Dialect, sql_literal
|
|
39
|
+
from parity.types import Column, LogicalType
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class SnowflakeDialect(Dialect):
|
|
43
|
+
name = "snowflake"
|
|
44
|
+
#: Set from the connection's schema segment. Snowflake folds unquoted names
|
|
45
|
+
#: to upper case, so an unqualified table is looked up in the connected
|
|
46
|
+
#: schema as stored.
|
|
47
|
+
default_schema = "PUBLIC"
|
|
48
|
+
|
|
49
|
+
def connect(self, connection_string: str) -> None: # pragma: no cover
|
|
50
|
+
"""Open a connection, pinned to UTC.
|
|
51
|
+
|
|
52
|
+
URL grammar mirrors SQLAlchemy's Snowflake dialect:
|
|
53
|
+
``snowflake://user:password@account/database/schema?warehouse=wh&role=r``
|
|
54
|
+
|
|
55
|
+
Note the honest gap: Snowflake supports only READ COMMITTED, so there
|
|
56
|
+
is no equivalent of PostgreSQL's REPEATABLE READ - the diff cannot hold
|
|
57
|
+
one snapshot across the walk. UTC is still pinned so timestamps render
|
|
58
|
+
deterministically.
|
|
59
|
+
"""
|
|
60
|
+
import snowflake.connector
|
|
61
|
+
|
|
62
|
+
url = urlparse(connection_string)
|
|
63
|
+
parts = [p for p in url.path.split("/") if p]
|
|
64
|
+
database = parts[0] if parts else None
|
|
65
|
+
schema = parts[1] if len(parts) > 1 else None
|
|
66
|
+
params = parse_qs(url.query)
|
|
67
|
+
|
|
68
|
+
def opt(key: str) -> str | None:
|
|
69
|
+
"""First value of a query parameter, or None."""
|
|
70
|
+
values = params.get(key)
|
|
71
|
+
return values[0] if values else None
|
|
72
|
+
if schema:
|
|
73
|
+
self.default_schema = schema
|
|
74
|
+
|
|
75
|
+
self._conn = snowflake.connector.connect(
|
|
76
|
+
account=url.hostname,
|
|
77
|
+
user=unquote(url.username) if url.username else None,
|
|
78
|
+
password=unquote(url.password) if url.password else None,
|
|
79
|
+
database=database,
|
|
80
|
+
schema=schema,
|
|
81
|
+
warehouse=opt("warehouse"),
|
|
82
|
+
role=opt("role"),
|
|
83
|
+
autocommit=True,
|
|
84
|
+
)
|
|
85
|
+
cur = self._conn.cursor()
|
|
86
|
+
try:
|
|
87
|
+
# Timestamps render through this; pin it so two accounts in
|
|
88
|
+
# different regions cannot disagree on the same instant.
|
|
89
|
+
cur.execute("alter session set timezone = 'UTC'")
|
|
90
|
+
finally:
|
|
91
|
+
cur.close()
|
|
92
|
+
|
|
93
|
+
def close(self) -> None: # pragma: no cover
|
|
94
|
+
"""Close the connection."""
|
|
95
|
+
self._conn.close()
|
|
96
|
+
|
|
97
|
+
def query(self, sql: str) -> list[tuple[Any, ...]]: # pragma: no cover
|
|
98
|
+
"""Run `sql` and return every row as a list of tuples."""
|
|
99
|
+
cur = self._conn.cursor()
|
|
100
|
+
try:
|
|
101
|
+
cur.execute(sql)
|
|
102
|
+
return list(cur.fetchall())
|
|
103
|
+
finally:
|
|
104
|
+
cur.close()
|
|
105
|
+
|
|
106
|
+
def columns(self, table: str) -> list[Column]:
|
|
107
|
+
"""Introspect columns, reading numeric_scale to split NUMBER.
|
|
108
|
+
|
|
109
|
+
Snowflake reports every integer and decimal as `NUMBER`; only the scale
|
|
110
|
+
distinguishes them, so this cannot use the shared `columns()`.
|
|
111
|
+
"""
|
|
112
|
+
schema, name = self.split_table(table, self.default_schema)
|
|
113
|
+
rows = self.query(
|
|
114
|
+
"select column_name, data_type, numeric_scale "
|
|
115
|
+
"from information_schema.columns "
|
|
116
|
+
f"where table_schema = {sql_literal(schema)} "
|
|
117
|
+
f"and table_name = {sql_literal(name)} "
|
|
118
|
+
"order by ordinal_position"
|
|
119
|
+
)
|
|
120
|
+
if not rows:
|
|
121
|
+
raise self._err(self._not_found(table, schema, name))
|
|
122
|
+
return [
|
|
123
|
+
Column(str(r[0]), map_type_snowflake(str(r[1]), r[2]), str(r[1]))
|
|
124
|
+
for r in rows
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
def quote(self, identifier: str) -> str:
|
|
128
|
+
"""Wrap an identifier in double quotes, doubling any it contains.
|
|
129
|
+
|
|
130
|
+
The injection boundary - names arrive from the command line. Snowflake
|
|
131
|
+
stores unquoted names upper-cased, so a lower-case `"id"` will not match
|
|
132
|
+
a column created as `ID`; the tool sidesteps this by quoting the exact
|
|
133
|
+
names `columns()` read back.
|
|
134
|
+
"""
|
|
135
|
+
return '"' + identifier.replace('"', '""') + '"'
|
|
136
|
+
|
|
137
|
+
# ----------------------------------------------------------- rendering
|
|
138
|
+
|
|
139
|
+
def null_sentinel_sql(self) -> str:
|
|
140
|
+
r"""Build the sentinel from CHR(92), not a `\N` literal.
|
|
141
|
+
|
|
142
|
+
Snowflake processes backslash escapes in string literals, so a literal
|
|
143
|
+
is unsafe; CHR(92) is a backslash unconditionally and CHR returns a
|
|
144
|
+
varchar (not binary), so the coalesce stays text.
|
|
145
|
+
"""
|
|
146
|
+
return "(chr(92) || 'N')"
|
|
147
|
+
|
|
148
|
+
def normalize(self, column: Column) -> str:
|
|
149
|
+
"""Render one column as canonical text, null-safe.
|
|
150
|
+
|
|
151
|
+
DECIMAL/FLOAT match DuckDB's path - cast to NUMBER(38, scale) then to
|
|
152
|
+
text - so `1.5` becomes `'1.500000'`. Non-finite floats are a known
|
|
153
|
+
unverified gap: Snowflake's Inf/NaN detection differs from the other
|
|
154
|
+
engines and must be checked against a real account before this is
|
|
155
|
+
trusted for FLOAT columns holding them.
|
|
156
|
+
"""
|
|
157
|
+
c = self.quote(column.name)
|
|
158
|
+
t = column.logical_type
|
|
159
|
+
if t is LogicalType.INTEGER:
|
|
160
|
+
expr = f"cast({c} as varchar)"
|
|
161
|
+
elif t in (LogicalType.DECIMAL, LogicalType.FLOAT):
|
|
162
|
+
expr = f"cast(cast({c} as number(38,{self.float_scale})) as varchar)"
|
|
163
|
+
elif t is LogicalType.BOOLEAN:
|
|
164
|
+
expr = f"case when {c} then 'true' when not {c} then 'false' end"
|
|
165
|
+
elif t is LogicalType.DATE:
|
|
166
|
+
expr = f"to_char({c}, 'YYYY-MM-DD')"
|
|
167
|
+
elif t is LogicalType.TIMESTAMP:
|
|
168
|
+
# FF6 is microseconds, matching the other engines' six digits.
|
|
169
|
+
expr = f"to_char({c}, 'YYYY-MM-DD HH24:MI:SS.FF6')"
|
|
170
|
+
else:
|
|
171
|
+
expr = f"cast({c} as varchar)"
|
|
172
|
+
return f"coalesce({expr}, {self.null_sentinel_sql()})"
|
|
173
|
+
|
|
174
|
+
def hash_expr(self, text_expr: str) -> str:
|
|
175
|
+
"""Fold canonical text into a positive 60-bit integer.
|
|
176
|
+
|
|
177
|
+
`MD5_NUMBER_UPPER64` is the top 64 bits of the digest as an unsigned
|
|
178
|
+
number; the top 60 bits - the first 15 hex characters the other engines
|
|
179
|
+
take - are that floor-divided by 16.
|
|
180
|
+
"""
|
|
181
|
+
return f"floor(md5_number_upper64({text_expr}) / 16)"
|
|
182
|
+
|
|
183
|
+
def int_div(self, numerator: str, denominator: str) -> str:
|
|
184
|
+
"""Truncating integer division.
|
|
185
|
+
|
|
186
|
+
Operands are NUMBER (via `wide_int`), so `/` is exact decimal and
|
|
187
|
+
`floor` truncates exactly for the non-negative operands used here -
|
|
188
|
+
unlike a float `/`, which CLAUDE.md 4.5 warns loses precision.
|
|
189
|
+
"""
|
|
190
|
+
return f"floor(({numerator}) / ({denominator}))"
|
|
191
|
+
|
|
192
|
+
def wide_int(self, expr: str) -> str:
|
|
193
|
+
"""Widen past 64 bits before arithmetic. NUMBER(38,0) holds 38 digits,
|
|
194
|
+
which the key offset cannot overflow."""
|
|
195
|
+
return f"cast(({expr}) as number(38,0))"
|
|
196
|
+
|
|
197
|
+
def sum_wide(self, expr: str) -> str:
|
|
198
|
+
"""Sum row hashes without overflowing, and return 0 for an empty group.
|
|
199
|
+
|
|
200
|
+
NUMBER(38,0) is far beyond any row count times 2^60.
|
|
201
|
+
"""
|
|
202
|
+
return f"coalesce(sum(cast(({expr}) as number(38,0))), 0)"
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def map_type_snowflake(raw: str, numeric_scale: Any) -> LogicalType:
|
|
206
|
+
"""Map a Snowflake type onto a logical category.
|
|
207
|
+
|
|
208
|
+
Snowflake reports both integers and decimals as `NUMBER`; only the scale
|
|
209
|
+
tells them apart, so it is passed in. A NULL scale (non-numeric type) is
|
|
210
|
+
treated as not-an-integer.
|
|
211
|
+
"""
|
|
212
|
+
t = raw.upper().split("(")[0].strip()
|
|
213
|
+
if t in {"NUMBER", "DECIMAL", "NUMERIC"}:
|
|
214
|
+
try:
|
|
215
|
+
scale = int(numeric_scale) if numeric_scale is not None else 6
|
|
216
|
+
except (TypeError, ValueError):
|
|
217
|
+
scale = 6
|
|
218
|
+
return LogicalType.INTEGER if scale == 0 else LogicalType.DECIMAL
|
|
219
|
+
if t in {"INT", "INTEGER", "BIGINT", "SMALLINT", "TINYINT", "BYTEINT"}:
|
|
220
|
+
return LogicalType.INTEGER # aliases that may appear via DATA_TYPE_ALIAS
|
|
221
|
+
if t in {"FLOAT", "FLOAT4", "FLOAT8", "DOUBLE", "DOUBLE PRECISION", "REAL"}:
|
|
222
|
+
return LogicalType.FLOAT
|
|
223
|
+
if t == "BOOLEAN":
|
|
224
|
+
return LogicalType.BOOLEAN
|
|
225
|
+
if t == "DATE":
|
|
226
|
+
return LogicalType.DATE
|
|
227
|
+
if t.startswith("TIMESTAMP") or t == "DATETIME":
|
|
228
|
+
return LogicalType.TIMESTAMP
|
|
229
|
+
if t in {"TEXT", "VARCHAR", "CHAR", "CHARACTER", "STRING"}:
|
|
230
|
+
return LogicalType.STRING
|
|
231
|
+
return LogicalType.UNKNOWN
|