interloper-snowflake 0.94.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interloper_snowflake-0.94.0/PKG-INFO +148 -0
- interloper_snowflake-0.94.0/README.md +136 -0
- interloper_snowflake-0.94.0/pyproject.toml +63 -0
- interloper_snowflake-0.94.0/pyproject.toml.orig +43 -0
- interloper_snowflake-0.94.0/src/interloper_snowflake/__init__.py +9 -0
- interloper_snowflake-0.94.0/src/interloper_snowflake/connection.py +126 -0
- interloper_snowflake-0.94.0/src/interloper_snowflake/destination.py +448 -0
- interloper_snowflake-0.94.0/src/interloper_snowflake/types.py +43 -0
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: interloper-snowflake
|
|
3
|
+
Version: 0.94.0
|
|
4
|
+
Summary: Interloper Snowflake integration: connection and destination
|
|
5
|
+
Author: Guillaume Onfroy
|
|
6
|
+
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
+
Requires-Dist: interloper-core
|
|
8
|
+
Requires-Dist: interloper-pandas
|
|
9
|
+
Requires-Dist: snowflake-connector-python[pandas]>=3.12
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# interloper-snowflake
|
|
14
|
+
|
|
15
|
+
Snowflake integration for interloper: a `SnowflakeDestination` that stores
|
|
16
|
+
assets as Snowflake tables, and the `SnowflakeConnection` that holds the
|
|
17
|
+
credentials.
|
|
18
|
+
|
|
19
|
+
The destination is a `DatabaseDestination`: it writes the Snowflake dialect
|
|
20
|
+
and nothing else. Partition replacement, windows and reads by partition come
|
|
21
|
+
from core, exactly as for BigQuery.
|
|
22
|
+
|
|
23
|
+
## Setup
|
|
24
|
+
|
|
25
|
+
The connection needs an account identifier, a user and a password, and
|
|
26
|
+
optionally a role (the user's default role otherwise).
|
|
27
|
+
|
|
28
|
+
The account identifier is the part of your Snowflake URL before
|
|
29
|
+
`.snowflakecomputing.com`, in either format:
|
|
30
|
+
|
|
31
|
+
- `myorg-myaccount`: organisation and account name (preferred)
|
|
32
|
+
- `xy12345.eu-central-1`: the legacy account locator with its region, and
|
|
33
|
+
its cloud where your URL carries one (e.g. `xy12345.us-east-2.aws`)
|
|
34
|
+
|
|
35
|
+
The role the connection runs as needs:
|
|
36
|
+
|
|
37
|
+
- `USAGE` on the warehouse the destination loads through
|
|
38
|
+
- `USAGE` on the database
|
|
39
|
+
- `CREATE SCHEMA` on the database, or `USAGE` on every schema the assets
|
|
40
|
+
write to if you create them yourself
|
|
41
|
+
- `CREATE TABLE` and `CREATE STAGE` on those schemas (the load goes through a
|
|
42
|
+
temporary stage)
|
|
43
|
+
- `INSERT`, `DELETE`, `SELECT` on the tables, and `TRUNCATE` for
|
|
44
|
+
unpartitioned assets (the table owner has all of them)
|
|
45
|
+
|
|
46
|
+
A sketch for a dedicated loading role:
|
|
47
|
+
|
|
48
|
+
```sql
|
|
49
|
+
CREATE ROLE interloper;
|
|
50
|
+
GRANT USAGE ON WAREHOUSE load_wh TO ROLE interloper;
|
|
51
|
+
GRANT USAGE, CREATE SCHEMA ON DATABASE analytics TO ROLE interloper;
|
|
52
|
+
GRANT ROLE interloper TO USER interloper_loader;
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Schemas and tables the role creates are owned by it, so the remaining grants
|
|
56
|
+
follow.
|
|
57
|
+
|
|
58
|
+
## Usage
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
import interloper as il
|
|
62
|
+
from interloper_snowflake import SnowflakeConnection, SnowflakeDestination
|
|
63
|
+
|
|
64
|
+
destination = SnowflakeDestination(
|
|
65
|
+
connection=SnowflakeConnection(account="myorg-myaccount", user="LOADER", password="..."),
|
|
66
|
+
database="ANALYTICS",
|
|
67
|
+
warehouse="LOAD_WH",
|
|
68
|
+
default_dataset="raw",
|
|
69
|
+
)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The credentials also load from the environment (`SNOWFLAKE_ACCOUNT`,
|
|
73
|
+
`SNOWFLAKE_USER`, `SNOWFLAKE_PASSWORD`, `SNOWFLAKE_ROLE`), so
|
|
74
|
+
`SnowflakeConnection()` works with no arguments.
|
|
75
|
+
|
|
76
|
+
In a deployed instance you configure this through the UI instead: add a
|
|
77
|
+
Snowflake connection, then a Snowflake destination, picking the database and
|
|
78
|
+
warehouse from the lists the connection can see.
|
|
79
|
+
|
|
80
|
+
## Datasets are schemas
|
|
81
|
+
|
|
82
|
+
An asset's `dataset` is the Snowflake schema its table lives in, inside the
|
|
83
|
+
destination's `database`. An asset without a dataset falls back to
|
|
84
|
+
`default_dataset`; with neither, the write fails with a `ConfigError`. A
|
|
85
|
+
missing schema is created on the first write, and a missing table is created
|
|
86
|
+
typed from the asset's schema (or one inferred from the data):
|
|
87
|
+
|
|
88
|
+
| Field type | Snowflake type |
|
|
89
|
+
|------------|----------------|
|
|
90
|
+
| `bool` | `BOOLEAN` |
|
|
91
|
+
| `int` | `NUMBER(38,0)` |
|
|
92
|
+
| `float` | `FLOAT` |
|
|
93
|
+
| `Decimal` | `NUMBER(38,9)` |
|
|
94
|
+
| `datetime` | `TIMESTAMP_NTZ` |
|
|
95
|
+
| `date` | `DATE` |
|
|
96
|
+
| `bytes` | `BINARY` |
|
|
97
|
+
| `str`, `Any` | `VARCHAR` |
|
|
98
|
+
| nested models, lists, dicts | `VARIANT` |
|
|
99
|
+
|
|
100
|
+
An existing table is never altered: a column the data carries but the table
|
|
101
|
+
does not is dropped with a warning.
|
|
102
|
+
|
|
103
|
+
## Quoted identifiers
|
|
104
|
+
|
|
105
|
+
Every database, schema, table and column name is double-quoted, so Snowflake
|
|
106
|
+
keeps it exactly as the asset spells it. Snowflake folds *unquoted*
|
|
107
|
+
identifiers to upper case, so a lower-case asset must be queried with quotes:
|
|
108
|
+
|
|
109
|
+
```sql
|
|
110
|
+
SELECT "cost" FROM "ANALYTICS"."marts"."ads_stats";
|
|
111
|
+
-- SELECT cost FROM analytics.marts.ads_stats looks for "COST" in "ADS_STATS" and fails
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## Partitions
|
|
115
|
+
|
|
116
|
+
A partitioned write replaces the partition's rows: it deletes them, then loads
|
|
117
|
+
the data, inside one `BEGIN ... COMMIT`, rolled back on failure. A time
|
|
118
|
+
partition deletes by half-open bounds (`"day" >= %s AND "day" < %s`), so a
|
|
119
|
+
monthly partition whose rows hold daily dates is replaced whole; any other
|
|
120
|
+
partition deletes by equality on its id. A window deletes each partition it
|
|
121
|
+
covers and loads the whole batch once. An unpartitioned asset truncates its
|
|
122
|
+
table and reloads it.
|
|
123
|
+
|
|
124
|
+
Each write first does everything that is DDL or file transfer: it creates the
|
|
125
|
+
schema and table if missing, creates a temporary stage in the schema (once per
|
|
126
|
+
session), and uploads the data as one Parquet file with `PUT`, under a prefix of
|
|
127
|
+
its own. Only then does it open the transaction, which holds nothing but the
|
|
128
|
+
`DELETE` (or `TRUNCATE`) and a `COPY INTO` projecting the file's columns by name
|
|
129
|
+
(`$1:"cost"`). Snowflake commits an open transaction whenever it runs DDL, so
|
|
130
|
+
keeping DDL out of the block is what makes a replace atomic: a failed load
|
|
131
|
+
rolls back the delete and the partition keeps its old rows. Reads go through
|
|
132
|
+
`fetch_pandas_all`, so both directions stay columnar.
|
|
133
|
+
|
|
134
|
+
## Notes
|
|
135
|
+
|
|
136
|
+
Each destination opens its own session through the connection, on its own
|
|
137
|
+
warehouse and database, so destinations sharing a connection never switch
|
|
138
|
+
each other's warehouse. That session is shared by every asset the destination
|
|
139
|
+
writes, and a Snowflake transaction belongs to the session rather than to a
|
|
140
|
+
cursor, so the destination serialises its writes.
|
|
141
|
+
|
|
142
|
+
## Docker images
|
|
143
|
+
|
|
144
|
+
The published interloper images do not ship this package. They are built on
|
|
145
|
+
Alpine, and `snowflake-connector-python` publishes no musllinux wheels, so
|
|
146
|
+
installing it there means compiling its C++ extension from source. Run it from
|
|
147
|
+
a glibc-based image (for example a `python:3.12-slim` base with
|
|
148
|
+
`pip install interloper-snowflake`), or anywhere outside the images.
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# interloper-snowflake
|
|
2
|
+
|
|
3
|
+
Snowflake integration for interloper: a `SnowflakeDestination` that stores
|
|
4
|
+
assets as Snowflake tables, and the `SnowflakeConnection` that holds the
|
|
5
|
+
credentials.
|
|
6
|
+
|
|
7
|
+
The destination is a `DatabaseDestination`: it writes the Snowflake dialect
|
|
8
|
+
and nothing else. Partition replacement, windows and reads by partition come
|
|
9
|
+
from core, exactly as for BigQuery.
|
|
10
|
+
|
|
11
|
+
## Setup
|
|
12
|
+
|
|
13
|
+
The connection needs an account identifier, a user and a password, and
|
|
14
|
+
optionally a role (the user's default role otherwise).
|
|
15
|
+
|
|
16
|
+
The account identifier is the part of your Snowflake URL before
|
|
17
|
+
`.snowflakecomputing.com`, in either format:
|
|
18
|
+
|
|
19
|
+
- `myorg-myaccount`: organisation and account name (preferred)
|
|
20
|
+
- `xy12345.eu-central-1`: the legacy account locator with its region, and
|
|
21
|
+
its cloud where your URL carries one (e.g. `xy12345.us-east-2.aws`)
|
|
22
|
+
|
|
23
|
+
The role the connection runs as needs:
|
|
24
|
+
|
|
25
|
+
- `USAGE` on the warehouse the destination loads through
|
|
26
|
+
- `USAGE` on the database
|
|
27
|
+
- `CREATE SCHEMA` on the database, or `USAGE` on every schema the assets
|
|
28
|
+
write to if you create them yourself
|
|
29
|
+
- `CREATE TABLE` and `CREATE STAGE` on those schemas (the load goes through a
|
|
30
|
+
temporary stage)
|
|
31
|
+
- `INSERT`, `DELETE`, `SELECT` on the tables, and `TRUNCATE` for
|
|
32
|
+
unpartitioned assets (the table owner has all of them)
|
|
33
|
+
|
|
34
|
+
A sketch for a dedicated loading role:
|
|
35
|
+
|
|
36
|
+
```sql
|
|
37
|
+
CREATE ROLE interloper;
|
|
38
|
+
GRANT USAGE ON WAREHOUSE load_wh TO ROLE interloper;
|
|
39
|
+
GRANT USAGE, CREATE SCHEMA ON DATABASE analytics TO ROLE interloper;
|
|
40
|
+
GRANT ROLE interloper TO USER interloper_loader;
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Schemas and tables the role creates are owned by it, so the remaining grants
|
|
44
|
+
follow.
|
|
45
|
+
|
|
46
|
+
## Usage
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
import interloper as il
|
|
50
|
+
from interloper_snowflake import SnowflakeConnection, SnowflakeDestination
|
|
51
|
+
|
|
52
|
+
destination = SnowflakeDestination(
|
|
53
|
+
connection=SnowflakeConnection(account="myorg-myaccount", user="LOADER", password="..."),
|
|
54
|
+
database="ANALYTICS",
|
|
55
|
+
warehouse="LOAD_WH",
|
|
56
|
+
default_dataset="raw",
|
|
57
|
+
)
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
The credentials also load from the environment (`SNOWFLAKE_ACCOUNT`,
|
|
61
|
+
`SNOWFLAKE_USER`, `SNOWFLAKE_PASSWORD`, `SNOWFLAKE_ROLE`), so
|
|
62
|
+
`SnowflakeConnection()` works with no arguments.
|
|
63
|
+
|
|
64
|
+
In a deployed instance you configure this through the UI instead: add a
|
|
65
|
+
Snowflake connection, then a Snowflake destination, picking the database and
|
|
66
|
+
warehouse from the lists the connection can see.
|
|
67
|
+
|
|
68
|
+
## Datasets are schemas
|
|
69
|
+
|
|
70
|
+
An asset's `dataset` is the Snowflake schema its table lives in, inside the
|
|
71
|
+
destination's `database`. An asset without a dataset falls back to
|
|
72
|
+
`default_dataset`; with neither, the write fails with a `ConfigError`. A
|
|
73
|
+
missing schema is created on the first write, and a missing table is created
|
|
74
|
+
typed from the asset's schema (or one inferred from the data):
|
|
75
|
+
|
|
76
|
+
| Field type | Snowflake type |
|
|
77
|
+
|------------|----------------|
|
|
78
|
+
| `bool` | `BOOLEAN` |
|
|
79
|
+
| `int` | `NUMBER(38,0)` |
|
|
80
|
+
| `float` | `FLOAT` |
|
|
81
|
+
| `Decimal` | `NUMBER(38,9)` |
|
|
82
|
+
| `datetime` | `TIMESTAMP_NTZ` |
|
|
83
|
+
| `date` | `DATE` |
|
|
84
|
+
| `bytes` | `BINARY` |
|
|
85
|
+
| `str`, `Any` | `VARCHAR` |
|
|
86
|
+
| nested models, lists, dicts | `VARIANT` |
|
|
87
|
+
|
|
88
|
+
An existing table is never altered: a column the data carries but the table
|
|
89
|
+
does not is dropped with a warning.
|
|
90
|
+
|
|
91
|
+
## Quoted identifiers
|
|
92
|
+
|
|
93
|
+
Every database, schema, table and column name is double-quoted, so Snowflake
|
|
94
|
+
keeps it exactly as the asset spells it. Snowflake folds *unquoted*
|
|
95
|
+
identifiers to upper case, so a lower-case asset must be queried with quotes:
|
|
96
|
+
|
|
97
|
+
```sql
|
|
98
|
+
SELECT "cost" FROM "ANALYTICS"."marts"."ads_stats";
|
|
99
|
+
-- SELECT cost FROM analytics.marts.ads_stats looks for "COST" in "ADS_STATS" and fails
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## Partitions
|
|
103
|
+
|
|
104
|
+
A partitioned write replaces the partition's rows: it deletes them, then loads
|
|
105
|
+
the data, inside one `BEGIN ... COMMIT`, rolled back on failure. A time
|
|
106
|
+
partition deletes by half-open bounds (`"day" >= %s AND "day" < %s`), so a
|
|
107
|
+
monthly partition whose rows hold daily dates is replaced whole; any other
|
|
108
|
+
partition deletes by equality on its id. A window deletes each partition it
|
|
109
|
+
covers and loads the whole batch once. An unpartitioned asset truncates its
|
|
110
|
+
table and reloads it.
|
|
111
|
+
|
|
112
|
+
Each write first does everything that is DDL or file transfer: it creates the
|
|
113
|
+
schema and table if missing, creates a temporary stage in the schema (once per
|
|
114
|
+
session), and uploads the data as one Parquet file with `PUT`, under a prefix of
|
|
115
|
+
its own. Only then does it open the transaction, which holds nothing but the
|
|
116
|
+
`DELETE` (or `TRUNCATE`) and a `COPY INTO` projecting the file's columns by name
|
|
117
|
+
(`$1:"cost"`). Snowflake commits an open transaction whenever it runs DDL, so
|
|
118
|
+
keeping DDL out of the block is what makes a replace atomic: a failed load
|
|
119
|
+
rolls back the delete and the partition keeps its old rows. Reads go through
|
|
120
|
+
`fetch_pandas_all`, so both directions stay columnar.
|
|
121
|
+
|
|
122
|
+
## Notes
|
|
123
|
+
|
|
124
|
+
Each destination opens its own session through the connection, on its own
|
|
125
|
+
warehouse and database, so destinations sharing a connection never switch
|
|
126
|
+
each other's warehouse. That session is shared by every asset the destination
|
|
127
|
+
writes, and a Snowflake transaction belongs to the session rather than to a
|
|
128
|
+
cursor, so the destination serialises its writes.
|
|
129
|
+
|
|
130
|
+
## Docker images
|
|
131
|
+
|
|
132
|
+
The published interloper images do not ship this package. They are built on
|
|
133
|
+
Alpine, and `snowflake-connector-python` publishes no musllinux wheels, so
|
|
134
|
+
installing it there means compiling its C++ extension from source. Run it from
|
|
135
|
+
a glibc-based image (for example a `python:3.12-slim` base with
|
|
136
|
+
`pip install interloper-snowflake`), or anywhere outside the images.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "interloper-snowflake"
|
|
3
|
+
version = "0.94.0"
|
|
4
|
+
description = "Interloper Snowflake integration: connection and destination"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"interloper-core",
|
|
9
|
+
"interloper-pandas",
|
|
10
|
+
"snowflake-connector-python[pandas]>=3.12",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[[project.authors]]
|
|
14
|
+
name = "Guillaume Onfroy"
|
|
15
|
+
email = "guillaume@digitlcloud.com"
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
snowflake = "interloper_snowflake"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources.interloper-core]
|
|
25
|
+
workspace = true
|
|
26
|
+
|
|
27
|
+
[tool.uv.sources.interloper-pandas]
|
|
28
|
+
workspace = true
|
|
29
|
+
|
|
30
|
+
[tool.ruff]
|
|
31
|
+
line-length = 120
|
|
32
|
+
|
|
33
|
+
[tool.ruff.lint]
|
|
34
|
+
preview = true
|
|
35
|
+
extend-select = [
|
|
36
|
+
"E",
|
|
37
|
+
"I",
|
|
38
|
+
"UP",
|
|
39
|
+
"ANN001",
|
|
40
|
+
"ANN201",
|
|
41
|
+
"ANN202",
|
|
42
|
+
"DOC",
|
|
43
|
+
"D",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint.pydocstyle]
|
|
47
|
+
convention = "google"
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint.per-file-ignores]
|
|
50
|
+
"__init__.py" = [
|
|
51
|
+
"F401",
|
|
52
|
+
"F403",
|
|
53
|
+
]
|
|
54
|
+
"tests/**" = [
|
|
55
|
+
"ANN",
|
|
56
|
+
"F811",
|
|
57
|
+
"D101",
|
|
58
|
+
"D102",
|
|
59
|
+
"D103",
|
|
60
|
+
"D104",
|
|
61
|
+
"RUF069",
|
|
62
|
+
"PLW0108",
|
|
63
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# ###############
|
|
2
|
+
# PROJECT / UV
|
|
3
|
+
# ###############
|
|
4
|
+
[project]
|
|
5
|
+
name = "interloper-snowflake"
|
|
6
|
+
version = "0.94.0"
|
|
7
|
+
description = "Interloper Snowflake integration: connection and destination"
|
|
8
|
+
readme = "README.md"
|
|
9
|
+
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"interloper-core",
|
|
13
|
+
"interloper-pandas",
|
|
14
|
+
"snowflake-connector-python[pandas]>=3.12",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
snowflake = "interloper_snowflake"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources]
|
|
25
|
+
interloper-core = { workspace = true }
|
|
26
|
+
interloper-pandas = { workspace = true }
|
|
27
|
+
|
|
28
|
+
# ###############
|
|
29
|
+
# RUFF
|
|
30
|
+
# ###############
|
|
31
|
+
[tool.ruff]
|
|
32
|
+
line-length = 120
|
|
33
|
+
|
|
34
|
+
[tool.ruff.lint]
|
|
35
|
+
preview = true
|
|
36
|
+
extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
|
|
37
|
+
|
|
38
|
+
[tool.ruff.lint.pydocstyle]
|
|
39
|
+
convention = "google"
|
|
40
|
+
|
|
41
|
+
[tool.ruff.lint.per-file-ignores]
|
|
42
|
+
"__init__.py" = ["F401", "F403"]
|
|
43
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Interloper Snowflake integration: connection and destination."""
|
|
2
|
+
|
|
3
|
+
from interloper_snowflake.connection import SnowflakeConnection
|
|
4
|
+
from interloper_snowflake.destination import SnowflakeDestination
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"SnowflakeConnection",
|
|
8
|
+
"SnowflakeDestination",
|
|
9
|
+
]
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""Snowflake connection resource holding user and password credentials."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import cached_property
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import snowflake.connector
|
|
9
|
+
from interloper.connection import Connection, connection
|
|
10
|
+
from interloper.resource.fields import InputField, SecretField, fetch_field_provider
|
|
11
|
+
from pydantic_settings import SettingsConfigDict
|
|
12
|
+
from snowflake.connector import SnowflakeConnection as Session
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@connection(
|
|
16
|
+
key="snowflake_connection",
|
|
17
|
+
name="Snowflake",
|
|
18
|
+
icon="icon:snowflake",
|
|
19
|
+
tags=["Cloud"],
|
|
20
|
+
)
|
|
21
|
+
class SnowflakeConnection(Connection):
|
|
22
|
+
"""Connection resource holding Snowflake credentials.
|
|
23
|
+
|
|
24
|
+
The connection holds the credentials; each destination bound to it opens
|
|
25
|
+
its own session on its own database and warehouse.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
model_config = SettingsConfigDict(env_prefix="snowflake_")
|
|
29
|
+
|
|
30
|
+
account: str = InputField(label="Account identifier", description="e.g. xy12345.eu-central-1 or myorg-myaccount")
|
|
31
|
+
user: str = InputField(description="Snowflake user name")
|
|
32
|
+
password: str = SecretField(description="Snowflake user password")
|
|
33
|
+
role: str | None = InputField(default=None, description="Role to assume; the user's default role when empty")
|
|
34
|
+
|
|
35
|
+
def connect(self, **session: Any) -> Session:
|
|
36
|
+
"""Open a new Snowflake session with this connection's credentials.
|
|
37
|
+
|
|
38
|
+
Autocommit stays on so a lone statement commits by itself; a caller
|
|
39
|
+
that needs atomicity opens an explicit ``BEGIN``.
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
**session: Session settings passed to the connector, such as
|
|
43
|
+
``warehouse`` and ``database``.
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
The new connector session.
|
|
47
|
+
"""
|
|
48
|
+
return snowflake.connector.connect(
|
|
49
|
+
account=self.account,
|
|
50
|
+
user=self.user,
|
|
51
|
+
password=self.password,
|
|
52
|
+
role=self.role,
|
|
53
|
+
autocommit=True,
|
|
54
|
+
**session,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
@cached_property
|
|
58
|
+
def client(self) -> Session:
|
|
59
|
+
"""The session the connection's own check and pickers run on.
|
|
60
|
+
|
|
61
|
+
A destination opens its own session through :meth:`connect` instead,
|
|
62
|
+
so its warehouse and transactions never touch a session another
|
|
63
|
+
component shares.
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
The connector session, cached per connection instance.
|
|
67
|
+
"""
|
|
68
|
+
return self.connect()
|
|
69
|
+
|
|
70
|
+
def _names(self, sql: str) -> list[dict[str, str]]:
|
|
71
|
+
"""Run a ``SHOW`` statement and return its ``name`` column as options.
|
|
72
|
+
|
|
73
|
+
``SHOW`` output carries many columns whose order is not part of its
|
|
74
|
+
contract, so the column is found by name through the cursor's
|
|
75
|
+
description.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
sql: The ``SHOW`` statement to run.
|
|
79
|
+
|
|
80
|
+
Returns:
|
|
81
|
+
Options with ``name``, sorted case-insensitively.
|
|
82
|
+
"""
|
|
83
|
+
cursor = self.client.cursor()
|
|
84
|
+
try:
|
|
85
|
+
cursor.execute(sql)
|
|
86
|
+
rows: list[Any] = cursor.fetchall()
|
|
87
|
+
index = [column[0] for column in cursor.description].index("name")
|
|
88
|
+
finally:
|
|
89
|
+
cursor.close()
|
|
90
|
+
return sorted(({"name": row[index]} for row in rows), key=lambda option: option["name"].lower())
|
|
91
|
+
|
|
92
|
+
@fetch_field_provider
|
|
93
|
+
def databases(self) -> list[dict[str, str]]:
|
|
94
|
+
"""List the databases this connection's role can see.
|
|
95
|
+
|
|
96
|
+
Backs the destination's ``database`` ``FetchField``.
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
Database options with ``name``.
|
|
100
|
+
"""
|
|
101
|
+
return self._names("SHOW DATABASES")
|
|
102
|
+
|
|
103
|
+
@fetch_field_provider
|
|
104
|
+
def warehouses(self) -> list[dict[str, str]]:
|
|
105
|
+
"""List the virtual warehouses this connection's role can see.
|
|
106
|
+
|
|
107
|
+
Backs the destination's ``warehouse`` ``FetchField``.
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
Warehouse options with ``name``.
|
|
111
|
+
"""
|
|
112
|
+
return self._names("SHOW WAREHOUSES")
|
|
113
|
+
|
|
114
|
+
def check(self) -> bool:
|
|
115
|
+
"""Prove the credentials work by opening a session and running ``SELECT 1``.
|
|
116
|
+
|
|
117
|
+
Returns:
|
|
118
|
+
True; a login or network failure raises out of the connector.
|
|
119
|
+
"""
|
|
120
|
+
cursor = self.client.cursor()
|
|
121
|
+
try:
|
|
122
|
+
cursor.execute("SELECT 1")
|
|
123
|
+
cursor.fetchall()
|
|
124
|
+
finally:
|
|
125
|
+
cursor.close()
|
|
126
|
+
return True
|
|
@@ -0,0 +1,448 @@
|
|
|
1
|
+
"""Snowflake destination implementation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import tempfile
|
|
6
|
+
import threading
|
|
7
|
+
import uuid
|
|
8
|
+
import warnings
|
|
9
|
+
from collections.abc import Iterator, Sequence
|
|
10
|
+
from contextlib import contextmanager
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from functools import cached_property
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
import pandas as pd
|
|
17
|
+
from interloper.destination import IOContext, destination
|
|
18
|
+
from interloper.destination.database import DatabaseDestination, PartitionFilter
|
|
19
|
+
from interloper.errors import ConfigError, DataNotFoundError
|
|
20
|
+
from interloper.representation import Representation
|
|
21
|
+
from interloper.resource.fields import FetchField, InputField
|
|
22
|
+
from interloper.schema import FieldSpec
|
|
23
|
+
from interloper.utils.data import is_empty
|
|
24
|
+
from pydantic import PrivateAttr
|
|
25
|
+
from snowflake.connector import SnowflakeConnection as Session
|
|
26
|
+
from snowflake.connector.cursor import SnowflakeCursor
|
|
27
|
+
|
|
28
|
+
from interloper_snowflake.connection import SnowflakeConnection
|
|
29
|
+
from interloper_snowflake.types import column_type
|
|
30
|
+
|
|
31
|
+
_STAGE = "interloper_load"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _quote(identifier: str) -> str:
|
|
35
|
+
"""Quote an identifier so Snowflake keeps it exactly as written.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
identifier: A database, schema, table or column name.
|
|
39
|
+
|
|
40
|
+
Returns:
|
|
41
|
+
The identifier in double quotes, embedded quotes doubled.
|
|
42
|
+
"""
|
|
43
|
+
return '"' + identifier.replace('"', '""') + '"'
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class _Staged:
|
|
48
|
+
"""Data uploaded to a stage, ready to be copied into its table.
|
|
49
|
+
|
|
50
|
+
Attributes:
|
|
51
|
+
table: The target table name.
|
|
52
|
+
schema: The resolved schema name.
|
|
53
|
+
location: The stage path holding the file, ``@stage/prefix``.
|
|
54
|
+
columns: The file's columns, all of them the table's.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
table: str
|
|
58
|
+
schema: str
|
|
59
|
+
location: str
|
|
60
|
+
columns: tuple[str, ...]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@destination(
|
|
64
|
+
key="snowflake_destination",
|
|
65
|
+
name="Snowflake",
|
|
66
|
+
icon="icon:snowflake",
|
|
67
|
+
tags=["Cloud"],
|
|
68
|
+
)
|
|
69
|
+
class SnowflakeDestination(DatabaseDestination):
|
|
70
|
+
"""Snowflake destination.
|
|
71
|
+
|
|
72
|
+
A dataset is a Snowflake schema inside the destination's database. Every
|
|
73
|
+
identifier is quoted, so tables and columns keep the case the asset gives
|
|
74
|
+
them.
|
|
75
|
+
|
|
76
|
+
The destination opens its own session on its warehouse and database. That
|
|
77
|
+
session is still shared by every asset written through the destination,
|
|
78
|
+
and a Snowflake transaction belongs to the session, so writes are
|
|
79
|
+
serialised: two concurrent writes would otherwise commit or roll back each
|
|
80
|
+
other's statements.
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
connection: SnowflakeConnection
|
|
84
|
+
|
|
85
|
+
database: str = FetchField(
|
|
86
|
+
provider="connection.databases",
|
|
87
|
+
label_key="name",
|
|
88
|
+
value_key="name",
|
|
89
|
+
description="Snowflake database",
|
|
90
|
+
discriminator=True,
|
|
91
|
+
)
|
|
92
|
+
warehouse: str = FetchField(
|
|
93
|
+
provider="connection.warehouses",
|
|
94
|
+
label_key="name",
|
|
95
|
+
value_key="name",
|
|
96
|
+
description="Virtual warehouse running the loads and queries",
|
|
97
|
+
)
|
|
98
|
+
default_dataset: str | None = InputField(default=None, description="Default schema for assets without a dataset")
|
|
99
|
+
|
|
100
|
+
_stages: set[str] = PrivateAttr(default_factory=set)
|
|
101
|
+
_lock: Any = PrivateAttr(default_factory=threading.RLock)
|
|
102
|
+
_local: threading.local = PrivateAttr(default_factory=threading.local)
|
|
103
|
+
|
|
104
|
+
@cached_property
|
|
105
|
+
def client(self) -> Session:
|
|
106
|
+
"""The destination's own session, on its warehouse and database.
|
|
107
|
+
|
|
108
|
+
Returns:
|
|
109
|
+
The connector session, cached per destination instance.
|
|
110
|
+
"""
|
|
111
|
+
return self.connection.connect(warehouse=self.warehouse, database=self.database)
|
|
112
|
+
|
|
113
|
+
# -- Session -----------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
def _execute(self, sql: str, params: Sequence[Any] | None = None) -> SnowflakeCursor:
|
|
116
|
+
"""Run one statement, on the open transaction's cursor or on a fresh one.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
sql: The statement, with ``%s`` placeholders for *params*.
|
|
120
|
+
params: The statement's parameters; defaults to none.
|
|
121
|
+
|
|
122
|
+
Returns:
|
|
123
|
+
The cursor the statement ran on, ready to fetch from.
|
|
124
|
+
"""
|
|
125
|
+
cursor = getattr(self._local, "cursor", None)
|
|
126
|
+
if cursor is None:
|
|
127
|
+
cursor = self.client.cursor()
|
|
128
|
+
cursor.execute(sql, params)
|
|
129
|
+
return cursor
|
|
130
|
+
|
|
131
|
+
@contextmanager
|
|
132
|
+
def transaction(self) -> Iterator[None]:
|
|
133
|
+
"""Run one write, a delete followed by an insert, as ``BEGIN ... COMMIT``.
|
|
134
|
+
|
|
135
|
+
Yields:
|
|
136
|
+
``None``; the write runs inside the block, rolled back and
|
|
137
|
+
re-raised if it raises.
|
|
138
|
+
"""
|
|
139
|
+
with self._lock:
|
|
140
|
+
cursor = self.client.cursor()
|
|
141
|
+
cursor.execute("BEGIN")
|
|
142
|
+
self._local.cursor = cursor
|
|
143
|
+
try:
|
|
144
|
+
yield
|
|
145
|
+
except BaseException:
|
|
146
|
+
cursor.execute("ROLLBACK")
|
|
147
|
+
raise
|
|
148
|
+
else:
|
|
149
|
+
cursor.execute("COMMIT")
|
|
150
|
+
finally:
|
|
151
|
+
self._local.cursor = None
|
|
152
|
+
cursor.close()
|
|
153
|
+
|
|
154
|
+
# -- Destination interface -----------------------------------------------------
|
|
155
|
+
|
|
156
|
+
def write(self, context: IOContext, data: Any) -> None:
|
|
157
|
+
"""Stage the data, then let the base replace its partitions.
|
|
158
|
+
|
|
159
|
+
Snowflake commits an open transaction whenever it runs DDL, and the
|
|
160
|
+
base calls :meth:`insert` inside :meth:`transaction`. So everything
|
|
161
|
+
that is DDL or file transfer (creating the schema, the table and the
|
|
162
|
+
stage, and the ``PUT``) runs here, before the base opens ``BEGIN``;
|
|
163
|
+
inside it, only the ``DELETE`` and the ``COPY INTO`` run. Both of the
|
|
164
|
+
base's paths, a single partition and a window, pass through here.
|
|
165
|
+
|
|
166
|
+
The whole write holds the destination's lock: the session is shared,
|
|
167
|
+
so another write's DDL would otherwise commit this one's open
|
|
168
|
+
transaction.
|
|
169
|
+
|
|
170
|
+
Args:
|
|
171
|
+
context: IO context carrying the target asset, the partition or window,
|
|
172
|
+
and the effective schema.
|
|
173
|
+
data: The data to write, in its native representation.
|
|
174
|
+
"""
|
|
175
|
+
if is_empty(data):
|
|
176
|
+
return
|
|
177
|
+
table, dataset = self._target(context)
|
|
178
|
+
with self._lock:
|
|
179
|
+
self._local.staged = self._stage(table, dataset, data, context)
|
|
180
|
+
try:
|
|
181
|
+
super().write(context, data)
|
|
182
|
+
finally:
|
|
183
|
+
self._local.staged = None
|
|
184
|
+
|
|
185
|
+
# -- Naming --------------------------------------------------------------------
|
|
186
|
+
|
|
187
|
+
def _resolve_dataset(self, dataset: str | None) -> str:
|
|
188
|
+
"""Return the Snowflake schema to use.
|
|
189
|
+
|
|
190
|
+
Args:
|
|
191
|
+
dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
|
|
192
|
+
|
|
193
|
+
Returns:
|
|
194
|
+
The resolved schema name.
|
|
195
|
+
|
|
196
|
+
Raises:
|
|
197
|
+
ConfigError: If neither the asset nor the destination names a dataset.
|
|
198
|
+
"""
|
|
199
|
+
schema = dataset or self.default_dataset
|
|
200
|
+
if schema is None:
|
|
201
|
+
raise ConfigError(
|
|
202
|
+
"SnowflakeDestination requires a dataset. Either set 'dataset' on the asset "
|
|
203
|
+
"or provide 'default_dataset' on the destination."
|
|
204
|
+
)
|
|
205
|
+
return schema
|
|
206
|
+
|
|
207
|
+
def _table_ref(self, table: str, schema: str) -> str:
|
|
208
|
+
"""Build a fully-qualified, quoted table reference.
|
|
209
|
+
|
|
210
|
+
Args:
|
|
211
|
+
table: Table name.
|
|
212
|
+
schema: The resolved schema name.
|
|
213
|
+
|
|
214
|
+
Returns:
|
|
215
|
+
``"database"."schema"."table"``.
|
|
216
|
+
"""
|
|
217
|
+
return f"{_quote(self.database)}.{_quote(schema)}.{_quote(table)}"
|
|
218
|
+
|
|
219
|
+
def _table_exists(self, table: str, schema: str) -> bool:
|
|
220
|
+
"""Check whether a table exists, through the database's information schema.
|
|
221
|
+
|
|
222
|
+
Args:
|
|
223
|
+
table: Table name.
|
|
224
|
+
schema: The resolved schema name.
|
|
225
|
+
|
|
226
|
+
Returns:
|
|
227
|
+
``True`` if the table exists, ``False`` otherwise.
|
|
228
|
+
"""
|
|
229
|
+
cursor = self._execute(
|
|
230
|
+
f"SELECT 1 FROM {_quote(self.database)}.information_schema.tables "
|
|
231
|
+
"WHERE table_schema = %s AND table_name = %s",
|
|
232
|
+
(schema, table),
|
|
233
|
+
)
|
|
234
|
+
return bool(cursor.fetchall())
|
|
235
|
+
|
|
236
|
+
# -- Staging -------------------------------------------------------------------
|
|
237
|
+
|
|
238
|
+
def _ensure_table(self, table: str, schema: str, specs: Sequence[FieldSpec]) -> None:
|
|
239
|
+
"""Create the schema and a typed table when the table does not exist.
|
|
240
|
+
|
|
241
|
+
Args:
|
|
242
|
+
table: Table name.
|
|
243
|
+
schema: The resolved schema name.
|
|
244
|
+
specs: The table's field specs.
|
|
245
|
+
"""
|
|
246
|
+
if self._table_exists(table, schema):
|
|
247
|
+
return
|
|
248
|
+
self._execute(f"CREATE SCHEMA IF NOT EXISTS {_quote(self.database)}.{_quote(schema)}")
|
|
249
|
+
self._execute(f"CREATE TABLE IF NOT EXISTS {self._table_ref(table, schema)} ({_columns_ddl(specs)})")
|
|
250
|
+
|
|
251
|
+
def _ensure_stage(self, schema: str) -> str:
|
|
252
|
+
"""Create the session's temporary stage in a schema, once per session.
|
|
253
|
+
|
|
254
|
+
Args:
|
|
255
|
+
schema: The resolved schema name.
|
|
256
|
+
|
|
257
|
+
Returns:
|
|
258
|
+
The stage's fully-qualified, quoted name.
|
|
259
|
+
"""
|
|
260
|
+
stage = f"{_quote(self.database)}.{_quote(schema)}.{_quote(_STAGE)}"
|
|
261
|
+
if schema not in self._stages:
|
|
262
|
+
self._execute(f"CREATE TEMPORARY STAGE IF NOT EXISTS {stage}")
|
|
263
|
+
self._stages.add(schema)
|
|
264
|
+
return stage
|
|
265
|
+
|
|
266
|
+
def _stage(self, table: str, dataset: str | None, data: Any, context: IOContext) -> _Staged:
|
|
267
|
+
"""Create what the load needs and upload the data as one Parquet file.
|
|
268
|
+
|
|
269
|
+
A new table is typed from the effective schema (declared on the asset,
|
|
270
|
+
or inferred during conform), or from a schema inferred from the data
|
|
271
|
+
when the context carries none. The frame is aligned to the table's
|
|
272
|
+
columns: an extra column is dropped with a warning, since an existing
|
|
273
|
+
table is never altered. The file goes under a prefix of its own, so
|
|
274
|
+
concurrent writes never load each other's files.
|
|
275
|
+
|
|
276
|
+
Args:
|
|
277
|
+
table: Target table name.
|
|
278
|
+
dataset: The Snowflake schema, or ``None`` for the destination's default.
|
|
279
|
+
data: The data in its native representation.
|
|
280
|
+
context: IO context carrying the asset and effective schema.
|
|
281
|
+
|
|
282
|
+
Returns:
|
|
283
|
+
Where the file was staged and the columns it carries.
|
|
284
|
+
"""
|
|
285
|
+
schema = self._resolve_dataset(dataset)
|
|
286
|
+
specs = (context.schema or Representation.of(data).infer()).field_specs()
|
|
287
|
+
self._ensure_table(table, schema, specs)
|
|
288
|
+
stage = self._ensure_stage(schema)
|
|
289
|
+
|
|
290
|
+
frame = Representation.of(data).to("dataframe")
|
|
291
|
+
names = [spec.name for spec in specs]
|
|
292
|
+
extras = [str(c) for c in frame.columns if str(c) not in names]
|
|
293
|
+
if extras:
|
|
294
|
+
warnings.warn(
|
|
295
|
+
f"Columns {extras} are not in the schema for '{self._table_ref(table, schema)}' "
|
|
296
|
+
"and will not be written.",
|
|
297
|
+
UserWarning,
|
|
298
|
+
stacklevel=3,
|
|
299
|
+
)
|
|
300
|
+
columns = tuple(c for c in names if c in frame.columns)
|
|
301
|
+
|
|
302
|
+
location = f"@{stage}/{uuid.uuid4().hex}"
|
|
303
|
+
with tempfile.TemporaryDirectory() as directory:
|
|
304
|
+
path = Path(directory) / "data.parquet"
|
|
305
|
+
frame[list(columns)].to_parquet(path, index=False)
|
|
306
|
+
uri = path.as_posix().replace("'", "\\'")
|
|
307
|
+
self._execute(f"PUT 'file://{uri}' {location} OVERWRITE=TRUE AUTO_COMPRESS=FALSE")
|
|
308
|
+
return _Staged(table=table, schema=schema, location=location, columns=columns)
|
|
309
|
+
|
|
310
|
+
# -- DatabaseDestination hooks ---------------------------------------------------
|
|
311
|
+
|
|
312
|
+
def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
|
|
313
|
+
"""Copy the staged data into the table.
|
|
314
|
+
|
|
315
|
+
:meth:`write` has already staged the data outside the transaction;
|
|
316
|
+
called any other way, the data is staged here first.
|
|
317
|
+
|
|
318
|
+
Args:
|
|
319
|
+
table: Target table name.
|
|
320
|
+
dataset: The Snowflake schema, or ``None`` for the destination's default.
|
|
321
|
+
data: The data in its native representation.
|
|
322
|
+
context: IO context carrying the asset and effective schema.
|
|
323
|
+
"""
|
|
324
|
+
staged = getattr(self._local, "staged", None)
|
|
325
|
+
if staged is None or staged.table != table or staged.schema != self._resolve_dataset(dataset):
|
|
326
|
+
staged = self._stage(table, dataset, data, context)
|
|
327
|
+
self._copy(staged)
|
|
328
|
+
|
|
329
|
+
def _copy(self, staged: _Staged) -> None:
|
|
330
|
+
"""Load a staged Parquet file with ``COPY INTO``, projecting its columns by name.
|
|
331
|
+
|
|
332
|
+
Parquet data is one ``$1`` object per row, so each column is read as
|
|
333
|
+
``$1:"name"`` and the file's column order does not matter. Binary
|
|
334
|
+
columns stay binary, and logical types (dates, timestamps, decimals)
|
|
335
|
+
are honoured. The file is purged once loaded.
|
|
336
|
+
|
|
337
|
+
Args:
|
|
338
|
+
staged: The staged file and its columns.
|
|
339
|
+
"""
|
|
340
|
+
targets = ", ".join(_quote(c) for c in staged.columns)
|
|
341
|
+
projection = ", ".join(f"$1:{_quote(c)}" for c in staged.columns)
|
|
342
|
+
self._execute(
|
|
343
|
+
f"COPY INTO {self._table_ref(staged.table, staged.schema)} ({targets}) "
|
|
344
|
+
f"FROM (SELECT {projection} FROM {staged.location}) "
|
|
345
|
+
"FILE_FORMAT=(TYPE=PARQUET USE_LOGICAL_TYPE=TRUE BINARY_AS_TEXT=FALSE) PURGE=TRUE"
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
|
|
349
|
+
"""Delete the rows a filter selects, or truncate the table.
|
|
350
|
+
|
|
351
|
+
A table that does not exist has nothing to delete.
|
|
352
|
+
|
|
353
|
+
Args:
|
|
354
|
+
table: Target table name.
|
|
355
|
+
dataset: The Snowflake schema, or ``None`` for the destination's default.
|
|
356
|
+
where: The rows to delete; ``None`` for the whole table.
|
|
357
|
+
"""
|
|
358
|
+
schema = self._resolve_dataset(dataset)
|
|
359
|
+
if not self._table_exists(table, schema):
|
|
360
|
+
return
|
|
361
|
+
ref = self._table_ref(table, schema)
|
|
362
|
+
if where is None:
|
|
363
|
+
self._execute(f"TRUNCATE TABLE {ref}")
|
|
364
|
+
return
|
|
365
|
+
predicate, params = _predicate(where)
|
|
366
|
+
self._execute(f"DELETE FROM {ref} WHERE {predicate}", params)
|
|
367
|
+
|
|
368
|
+
def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
|
|
369
|
+
"""Select the rows a filter selects, or every row, as a DataFrame.
|
|
370
|
+
|
|
371
|
+
The connector builds the frame from the result's Arrow batches, so
|
|
372
|
+
column types survive the read without a pass through Python records.
|
|
373
|
+
|
|
374
|
+
Args:
|
|
375
|
+
table: Target table name.
|
|
376
|
+
dataset: The Snowflake schema, or ``None`` for the destination's default.
|
|
377
|
+
where: The rows to select; ``None`` for the whole table.
|
|
378
|
+
|
|
379
|
+
Returns:
|
|
380
|
+
The selected rows.
|
|
381
|
+
|
|
382
|
+
Raises:
|
|
383
|
+
DataNotFoundError: If the table does not exist yet.
|
|
384
|
+
"""
|
|
385
|
+
schema = self._resolve_dataset(dataset)
|
|
386
|
+
ref = self._table_ref(table, schema)
|
|
387
|
+
if not self._table_exists(table, schema):
|
|
388
|
+
raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
|
|
389
|
+
if where is None:
|
|
390
|
+
return self._execute(f"SELECT * FROM {ref}").fetch_pandas_all()
|
|
391
|
+
predicate, params = _predicate(where)
|
|
392
|
+
return self._execute(f"SELECT * FROM {ref} WHERE {predicate}", params).fetch_pandas_all()
|
|
393
|
+
|
|
394
|
+
def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
|
|
395
|
+
"""Return row counts grouped by a column.
|
|
396
|
+
|
|
397
|
+
Args:
|
|
398
|
+
table: Target table name.
|
|
399
|
+
dataset: The Snowflake schema, or ``None`` for the destination's default.
|
|
400
|
+
column: Column to group by.
|
|
401
|
+
|
|
402
|
+
Returns:
|
|
403
|
+
Mapping from the column's value (as string) to row count.
|
|
404
|
+
|
|
405
|
+
Raises:
|
|
406
|
+
DataNotFoundError: If the table does not exist yet.
|
|
407
|
+
"""
|
|
408
|
+
schema = self._resolve_dataset(dataset)
|
|
409
|
+
ref = self._table_ref(table, schema)
|
|
410
|
+
if not self._table_exists(table, schema):
|
|
411
|
+
raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
|
|
412
|
+
cursor = self._execute(
|
|
413
|
+
f"SELECT TO_VARCHAR({_quote(column)}) AS partition_value, COUNT(*) AS cnt FROM {ref} GROUP BY 1"
|
|
414
|
+
)
|
|
415
|
+
return {value: count for value, count in cursor.fetchall()}
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
# -- Utility functions ---------------------------------------------------------------
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _columns_ddl(specs: Sequence[FieldSpec]) -> str:
|
|
422
|
+
"""Render field specs as the column list of a ``CREATE TABLE``.
|
|
423
|
+
|
|
424
|
+
Args:
|
|
425
|
+
specs: The table's field specs.
|
|
426
|
+
|
|
427
|
+
Returns:
|
|
428
|
+
Comma-separated quoted column definitions. Every column is nullable:
|
|
429
|
+
conform already enforces the schema's nullability, and a constraint
|
|
430
|
+
here would only turn a schema change into a failed load.
|
|
431
|
+
"""
|
|
432
|
+
return ", ".join(f"{_quote(spec.name)} {column_type(spec)}" for spec in specs)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def _predicate(where: PartitionFilter) -> tuple[str, tuple[Any, ...]]:
|
|
436
|
+
"""Render a partition filter as a parameterised SQL predicate.
|
|
437
|
+
|
|
438
|
+
Args:
|
|
439
|
+
where: The filter to render.
|
|
440
|
+
|
|
441
|
+
Returns:
|
|
442
|
+
The predicate text and the parameters its ``%s`` placeholders name.
|
|
443
|
+
"""
|
|
444
|
+
column = _quote(where.column)
|
|
445
|
+
if where.bounds is None:
|
|
446
|
+
return f"{column} = %s", (where.value,)
|
|
447
|
+
start, end = where.bounds
|
|
448
|
+
return f"{column} >= %s AND {column} < %s", (start, end)
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Snowflake's view of interloper's field types."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
from decimal import Decimal
|
|
7
|
+
|
|
8
|
+
from interloper.schema import FieldSpec
|
|
9
|
+
|
|
10
|
+
# Ordered: the first base class that matches wins, so bool (a subclass of int)
|
|
11
|
+
# and datetime (a subclass of date) must come before their parents.
|
|
12
|
+
_PYTHON_TO_SNOWFLAKE: dict[type, str] = {
|
|
13
|
+
bool: "BOOLEAN",
|
|
14
|
+
int: "NUMBER(38,0)",
|
|
15
|
+
float: "FLOAT",
|
|
16
|
+
Decimal: "NUMBER(38,9)",
|
|
17
|
+
datetime.datetime: "TIMESTAMP_NTZ",
|
|
18
|
+
datetime.date: "DATE",
|
|
19
|
+
bytes: "BINARY",
|
|
20
|
+
str: "VARCHAR",
|
|
21
|
+
dict: "VARIANT",
|
|
22
|
+
list: "VARIANT",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def column_type(spec: FieldSpec) -> str:
|
|
27
|
+
"""Return the Snowflake column type for a field spec.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
``VARIANT`` for a nested or repeated field, the type the spec's Python
|
|
34
|
+
type maps to otherwise, and ``VARCHAR`` for anything unmapped
|
|
35
|
+
(``typing.Any``).
|
|
36
|
+
"""
|
|
37
|
+
if spec.fields is not None or spec.repeated:
|
|
38
|
+
return "VARIANT"
|
|
39
|
+
if isinstance(spec.type, type):
|
|
40
|
+
for base, name in _PYTHON_TO_SNOWFLAKE.items():
|
|
41
|
+
if issubclass(spec.type, base):
|
|
42
|
+
return name
|
|
43
|
+
return "VARCHAR"
|