interloper-snowflake 0.94.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,148 @@
1
+ Metadata-Version: 2.3
2
+ Name: interloper-snowflake
3
+ Version: 0.94.0
4
+ Summary: Interloper Snowflake integration: connection and destination
5
+ Author: Guillaume Onfroy
6
+ Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
7
+ Requires-Dist: interloper-core
8
+ Requires-Dist: interloper-pandas
9
+ Requires-Dist: snowflake-connector-python[pandas]>=3.12
10
+ Requires-Python: >=3.10
11
+ Description-Content-Type: text/markdown
12
+
13
+ # interloper-snowflake
14
+
15
+ Snowflake integration for interloper: a `SnowflakeDestination` that stores
16
+ assets as Snowflake tables, and the `SnowflakeConnection` that holds the
17
+ credentials.
18
+
19
+ The destination is a `DatabaseDestination`: it writes the Snowflake dialect
20
+ and nothing else. Partition replacement, windows and reads by partition come
21
+ from core, exactly as for BigQuery.
22
+
23
+ ## Setup
24
+
25
+ The connection needs an account identifier, a user and a password, and
26
+ optionally a role (the user's default role otherwise).
27
+
28
+ The account identifier is the part of your Snowflake URL before
29
+ `.snowflakecomputing.com`, in either format:
30
+
31
+ - `myorg-myaccount`: organisation and account name (preferred)
32
+ - `xy12345.eu-central-1`: the legacy account locator with its region, and
33
+ its cloud where your URL carries one (e.g. `xy12345.us-east-2.aws`)
34
+
35
+ The role the connection runs as needs:
36
+
37
+ - `USAGE` on the warehouse the destination loads through
38
+ - `USAGE` on the database
39
+ - `CREATE SCHEMA` on the database, or `USAGE` on every schema the assets
40
+ write to if you create them yourself
41
+ - `CREATE TABLE` and `CREATE STAGE` on those schemas (the load goes through a
42
+ temporary stage)
43
+ - `INSERT`, `DELETE`, `SELECT` on the tables, and `TRUNCATE` for
44
+ unpartitioned assets (the table owner has all of them)
45
+
46
+ A sketch for a dedicated loading role:
47
+
48
+ ```sql
49
+ CREATE ROLE interloper;
50
+ GRANT USAGE ON WAREHOUSE load_wh TO ROLE interloper;
51
+ GRANT USAGE, CREATE SCHEMA ON DATABASE analytics TO ROLE interloper;
52
+ GRANT ROLE interloper TO USER interloper_loader;
53
+ ```
54
+
55
+ Schemas and tables the role creates are owned by it, so the remaining grants
56
+ follow.
57
+
58
+ ## Usage
59
+
60
+ ```python
61
+ import interloper as il
62
+ from interloper_snowflake import SnowflakeConnection, SnowflakeDestination
63
+
64
+ destination = SnowflakeDestination(
65
+ connection=SnowflakeConnection(account="myorg-myaccount", user="LOADER", password="..."),
66
+ database="ANALYTICS",
67
+ warehouse="LOAD_WH",
68
+ default_dataset="raw",
69
+ )
70
+ ```
71
+
72
+ The credentials also load from the environment (`SNOWFLAKE_ACCOUNT`,
73
+ `SNOWFLAKE_USER`, `SNOWFLAKE_PASSWORD`, `SNOWFLAKE_ROLE`), so
74
+ `SnowflakeConnection()` works with no arguments.
75
+
76
+ In a deployed instance you configure this through the UI instead: add a
77
+ Snowflake connection, then a Snowflake destination, picking the database and
78
+ warehouse from the lists the connection can see.
79
+
80
+ ## Datasets are schemas
81
+
82
+ An asset's `dataset` is the Snowflake schema its table lives in, inside the
83
+ destination's `database`. An asset without a dataset falls back to
84
+ `default_dataset`; with neither, the write fails with a `ConfigError`. A
85
+ missing schema is created on the first write, and a missing table is created
86
+ typed from the asset's schema (or one inferred from the data):
87
+
88
+ | Field type | Snowflake type |
89
+ |------------|----------------|
90
+ | `bool` | `BOOLEAN` |
91
+ | `int` | `NUMBER(38,0)` |
92
+ | `float` | `FLOAT` |
93
+ | `Decimal` | `NUMBER(38,9)` |
94
+ | `datetime` | `TIMESTAMP_NTZ` |
95
+ | `date` | `DATE` |
96
+ | `bytes` | `BINARY` |
97
+ | `str`, `Any` | `VARCHAR` |
98
+ | nested models, lists, dicts | `VARIANT` |
99
+
100
+ An existing table is never altered: a column the data carries but the table
101
+ does not is dropped with a warning.
102
+
103
+ ## Quoted identifiers
104
+
105
+ Every database, schema, table and column name is double-quoted, so Snowflake
106
+ keeps it exactly as the asset spells it. Snowflake folds *unquoted*
107
+ identifiers to upper case, so a lower-case asset must be queried with quotes:
108
+
109
+ ```sql
110
+ SELECT "cost" FROM "ANALYTICS"."marts"."ads_stats";
111
+ -- SELECT cost FROM analytics.marts.ads_stats looks for "COST" in "ADS_STATS" and fails
112
+ ```
113
+
114
+ ## Partitions
115
+
116
+ A partitioned write replaces the partition's rows: it deletes them, then loads
117
+ the data, inside one `BEGIN ... COMMIT`, rolled back on failure. A time
118
+ partition deletes by half-open bounds (`"day" >= %s AND "day" < %s`), so a
119
+ monthly partition whose rows hold daily dates is replaced whole; any other
120
+ partition deletes by equality on its id. A window deletes each partition it
121
+ covers and loads the whole batch once. An unpartitioned asset truncates its
122
+ table and reloads it.
123
+
124
+ Each write first does everything that is DDL or file transfer: it creates the
125
+ schema and table if missing, creates a temporary stage in the schema (once per
126
+ session), and uploads the data as one Parquet file with `PUT`, under a prefix of
127
+ its own. Only then does it open the transaction, which holds nothing but the
128
+ `DELETE` (or `TRUNCATE`) and a `COPY INTO` projecting the file's columns by name
129
+ (`$1:"cost"`). Snowflake commits an open transaction whenever it runs DDL, so
130
+ keeping DDL out of the block is what makes a replace atomic: a failed load
131
+ rolls back the delete and the partition keeps its old rows. Reads go through
132
+ `fetch_pandas_all`, so both directions stay columnar.
133
+
134
+ ## Notes
135
+
136
+ Each destination opens its own session through the connection, on its own
137
+ warehouse and database, so destinations sharing a connection never switch
138
+ each other's warehouse. That session is shared by every asset the destination
139
+ writes, and a Snowflake transaction belongs to the session rather than to a
140
+ cursor, so the destination serialises its writes.
141
+
142
+ ## Docker images
143
+
144
+ The published interloper images do not ship this package. They are built on
145
+ Alpine, and `snowflake-connector-python` publishes no musllinux wheels, so
146
+ installing it there means compiling its C++ extension from source. Run it from
147
+ a glibc-based image (for example a `python:3.12-slim` base with
148
+ `pip install interloper-snowflake`), or anywhere outside the images.
@@ -0,0 +1,136 @@
1
+ # interloper-snowflake
2
+
3
+ Snowflake integration for interloper: a `SnowflakeDestination` that stores
4
+ assets as Snowflake tables, and the `SnowflakeConnection` that holds the
5
+ credentials.
6
+
7
+ The destination is a `DatabaseDestination`: it writes the Snowflake dialect
8
+ and nothing else. Partition replacement, windows and reads by partition come
9
+ from core, exactly as for BigQuery.
10
+
11
+ ## Setup
12
+
13
+ The connection needs an account identifier, a user and a password, and
14
+ optionally a role (the user's default role otherwise).
15
+
16
+ The account identifier is the part of your Snowflake URL before
17
+ `.snowflakecomputing.com`, in either format:
18
+
19
+ - `myorg-myaccount`: organisation and account name (preferred)
20
+ - `xy12345.eu-central-1`: the legacy account locator with its region, and
21
+ its cloud where your URL carries one (e.g. `xy12345.us-east-2.aws`)
22
+
23
+ The role the connection runs as needs:
24
+
25
+ - `USAGE` on the warehouse the destination loads through
26
+ - `USAGE` on the database
27
+ - `CREATE SCHEMA` on the database, or `USAGE` on every schema the assets
28
+ write to if you create them yourself
29
+ - `CREATE TABLE` and `CREATE STAGE` on those schemas (the load goes through a
30
+ temporary stage)
31
+ - `INSERT`, `DELETE`, `SELECT` on the tables, and `TRUNCATE` for
32
+ unpartitioned assets (the table owner has all of them)
33
+
34
+ A sketch for a dedicated loading role:
35
+
36
+ ```sql
37
+ CREATE ROLE interloper;
38
+ GRANT USAGE ON WAREHOUSE load_wh TO ROLE interloper;
39
+ GRANT USAGE, CREATE SCHEMA ON DATABASE analytics TO ROLE interloper;
40
+ GRANT ROLE interloper TO USER interloper_loader;
41
+ ```
42
+
43
+ Schemas and tables the role creates are owned by it, so the remaining grants
44
+ follow.
45
+
46
+ ## Usage
47
+
48
+ ```python
49
+ import interloper as il
50
+ from interloper_snowflake import SnowflakeConnection, SnowflakeDestination
51
+
52
+ destination = SnowflakeDestination(
53
+ connection=SnowflakeConnection(account="myorg-myaccount", user="LOADER", password="..."),
54
+ database="ANALYTICS",
55
+ warehouse="LOAD_WH",
56
+ default_dataset="raw",
57
+ )
58
+ ```
59
+
60
+ The credentials also load from the environment (`SNOWFLAKE_ACCOUNT`,
61
+ `SNOWFLAKE_USER`, `SNOWFLAKE_PASSWORD`, `SNOWFLAKE_ROLE`), so
62
+ `SnowflakeConnection()` works with no arguments.
63
+
64
+ In a deployed instance you configure this through the UI instead: add a
65
+ Snowflake connection, then a Snowflake destination, picking the database and
66
+ warehouse from the lists the connection can see.
67
+
68
+ ## Datasets are schemas
69
+
70
+ An asset's `dataset` is the Snowflake schema its table lives in, inside the
71
+ destination's `database`. An asset without a dataset falls back to
72
+ `default_dataset`; with neither, the write fails with a `ConfigError`. A
73
+ missing schema is created on the first write, and a missing table is created
74
+ typed from the asset's schema (or one inferred from the data):
75
+
76
+ | Field type | Snowflake type |
77
+ |------------|----------------|
78
+ | `bool` | `BOOLEAN` |
79
+ | `int` | `NUMBER(38,0)` |
80
+ | `float` | `FLOAT` |
81
+ | `Decimal` | `NUMBER(38,9)` |
82
+ | `datetime` | `TIMESTAMP_NTZ` |
83
+ | `date` | `DATE` |
84
+ | `bytes` | `BINARY` |
85
+ | `str`, `Any` | `VARCHAR` |
86
+ | nested models, lists, dicts | `VARIANT` |
87
+
88
+ An existing table is never altered: a column the data carries but the table
89
+ does not is dropped with a warning.
90
+
91
+ ## Quoted identifiers
92
+
93
+ Every database, schema, table and column name is double-quoted, so Snowflake
94
+ keeps it exactly as the asset spells it. Snowflake folds *unquoted*
95
+ identifiers to upper case, so a lower-case asset must be queried with quotes:
96
+
97
+ ```sql
98
+ SELECT "cost" FROM "ANALYTICS"."marts"."ads_stats";
99
+ -- SELECT cost FROM analytics.marts.ads_stats looks for "COST" in "ADS_STATS" and fails
100
+ ```
101
+
102
+ ## Partitions
103
+
104
+ A partitioned write replaces the partition's rows: it deletes them, then loads
105
+ the data, inside one `BEGIN ... COMMIT`, rolled back on failure. A time
106
+ partition deletes by half-open bounds (`"day" >= %s AND "day" < %s`), so a
107
+ monthly partition whose rows hold daily dates is replaced whole; any other
108
+ partition deletes by equality on its id. A window deletes each partition it
109
+ covers and loads the whole batch once. An unpartitioned asset truncates its
110
+ table and reloads it.
111
+
112
+ Each write first does everything that is DDL or file transfer: it creates the
113
+ schema and table if missing, creates a temporary stage in the schema (once per
114
+ session), and uploads the data as one Parquet file with `PUT`, under a prefix of
115
+ its own. Only then does it open the transaction, which holds nothing but the
116
+ `DELETE` (or `TRUNCATE`) and a `COPY INTO` projecting the file's columns by name
117
+ (`$1:"cost"`). Snowflake commits an open transaction whenever it runs DDL, so
118
+ keeping DDL out of the block is what makes a replace atomic: a failed load
119
+ rolls back the delete and the partition keeps its old rows. Reads go through
120
+ `fetch_pandas_all`, so both directions stay columnar.
121
+
122
+ ## Notes
123
+
124
+ Each destination opens its own session through the connection, on its own
125
+ warehouse and database, so destinations sharing a connection never switch
126
+ each other's warehouse. That session is shared by every asset the destination
127
+ writes, and a Snowflake transaction belongs to the session rather than to a
128
+ cursor, so the destination serialises its writes.
129
+
130
+ ## Docker images
131
+
132
+ The published interloper images do not ship this package. They are built on
133
+ Alpine, and `snowflake-connector-python` publishes no musllinux wheels, so
134
+ installing it there means compiling its C++ extension from source. Run it from
135
+ a glibc-based image (for example a `python:3.12-slim` base with
136
+ `pip install interloper-snowflake`), or anywhere outside the images.
@@ -0,0 +1,63 @@
1
+ [project]
2
+ name = "interloper-snowflake"
3
+ version = "0.94.0"
4
+ description = "Interloper Snowflake integration: connection and destination"
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ dependencies = [
8
+ "interloper-core",
9
+ "interloper-pandas",
10
+ "snowflake-connector-python[pandas]>=3.12",
11
+ ]
12
+
13
+ [[project.authors]]
14
+ name = "Guillaume Onfroy"
15
+ email = "guillaume@digitlcloud.com"
16
+
17
+ [project.entry-points."interloper.components"]
18
+ snowflake = "interloper_snowflake"
19
+
20
+ [build-system]
21
+ requires = ["uv_build>=0.12.9,<0.13"]
22
+ build-backend = "uv_build"
23
+
24
+ [tool.uv.sources.interloper-core]
25
+ workspace = true
26
+
27
+ [tool.uv.sources.interloper-pandas]
28
+ workspace = true
29
+
30
+ [tool.ruff]
31
+ line-length = 120
32
+
33
+ [tool.ruff.lint]
34
+ preview = true
35
+ extend-select = [
36
+ "E",
37
+ "I",
38
+ "UP",
39
+ "ANN001",
40
+ "ANN201",
41
+ "ANN202",
42
+ "DOC",
43
+ "D",
44
+ ]
45
+
46
+ [tool.ruff.lint.pydocstyle]
47
+ convention = "google"
48
+
49
+ [tool.ruff.lint.per-file-ignores]
50
+ "__init__.py" = [
51
+ "F401",
52
+ "F403",
53
+ ]
54
+ "tests/**" = [
55
+ "ANN",
56
+ "F811",
57
+ "D101",
58
+ "D102",
59
+ "D103",
60
+ "D104",
61
+ "RUF069",
62
+ "PLW0108",
63
+ ]
@@ -0,0 +1,43 @@
1
+ # ###############
2
+ # PROJECT / UV
3
+ # ###############
4
+ [project]
5
+ name = "interloper-snowflake"
6
+ version = "0.94.0"
7
+ description = "Interloper Snowflake integration: connection and destination"
8
+ readme = "README.md"
9
+ authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
10
+ requires-python = ">=3.10"
11
+ dependencies = [
12
+ "interloper-core",
13
+ "interloper-pandas",
14
+ "snowflake-connector-python[pandas]>=3.12",
15
+ ]
16
+
17
+ [project.entry-points."interloper.components"]
18
+ snowflake = "interloper_snowflake"
19
+
20
+ [build-system]
21
+ requires = ["uv_build>=0.12.9,<0.13"]
22
+ build-backend = "uv_build"
23
+
24
+ [tool.uv.sources]
25
+ interloper-core = { workspace = true }
26
+ interloper-pandas = { workspace = true }
27
+
28
+ # ###############
29
+ # RUFF
30
+ # ###############
31
+ [tool.ruff]
32
+ line-length = 120
33
+
34
+ [tool.ruff.lint]
35
+ preview = true
36
+ extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
37
+
38
+ [tool.ruff.lint.pydocstyle]
39
+ convention = "google"
40
+
41
+ [tool.ruff.lint.per-file-ignores]
42
+ "__init__.py" = ["F401", "F403"]
43
+ "tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
@@ -0,0 +1,9 @@
1
+ """Interloper Snowflake integration: connection and destination."""
2
+
3
+ from interloper_snowflake.connection import SnowflakeConnection
4
+ from interloper_snowflake.destination import SnowflakeDestination
5
+
6
+ __all__ = [
7
+ "SnowflakeConnection",
8
+ "SnowflakeDestination",
9
+ ]
@@ -0,0 +1,126 @@
1
+ """Snowflake connection resource holding user and password credentials."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from functools import cached_property
6
+ from typing import Any
7
+
8
+ import snowflake.connector
9
+ from interloper.connection import Connection, connection
10
+ from interloper.resource.fields import InputField, SecretField, fetch_field_provider
11
+ from pydantic_settings import SettingsConfigDict
12
+ from snowflake.connector import SnowflakeConnection as Session
13
+
14
+
15
+ @connection(
16
+ key="snowflake_connection",
17
+ name="Snowflake",
18
+ icon="icon:snowflake",
19
+ tags=["Cloud"],
20
+ )
21
+ class SnowflakeConnection(Connection):
22
+ """Connection resource holding Snowflake credentials.
23
+
24
+ The connection holds the credentials; each destination bound to it opens
25
+ its own session on its own database and warehouse.
26
+ """
27
+
28
+ model_config = SettingsConfigDict(env_prefix="snowflake_")
29
+
30
+ account: str = InputField(label="Account identifier", description="e.g. xy12345.eu-central-1 or myorg-myaccount")
31
+ user: str = InputField(description="Snowflake user name")
32
+ password: str = SecretField(description="Snowflake user password")
33
+ role: str | None = InputField(default=None, description="Role to assume; the user's default role when empty")
34
+
35
+ def connect(self, **session: Any) -> Session:
36
+ """Open a new Snowflake session with this connection's credentials.
37
+
38
+ Autocommit stays on so a lone statement commits by itself; a caller
39
+ that needs atomicity opens an explicit ``BEGIN``.
40
+
41
+ Args:
42
+ **session: Session settings passed to the connector, such as
43
+ ``warehouse`` and ``database``.
44
+
45
+ Returns:
46
+ The new connector session.
47
+ """
48
+ return snowflake.connector.connect(
49
+ account=self.account,
50
+ user=self.user,
51
+ password=self.password,
52
+ role=self.role,
53
+ autocommit=True,
54
+ **session,
55
+ )
56
+
57
+ @cached_property
58
+ def client(self) -> Session:
59
+ """The session the connection's own check and pickers run on.
60
+
61
+ A destination opens its own session through :meth:`connect` instead,
62
+ so its warehouse and transactions never touch a session another
63
+ component shares.
64
+
65
+ Returns:
66
+ The connector session, cached per connection instance.
67
+ """
68
+ return self.connect()
69
+
70
+ def _names(self, sql: str) -> list[dict[str, str]]:
71
+ """Run a ``SHOW`` statement and return its ``name`` column as options.
72
+
73
+ ``SHOW`` output carries many columns whose order is not part of its
74
+ contract, so the column is found by name through the cursor's
75
+ description.
76
+
77
+ Args:
78
+ sql: The ``SHOW`` statement to run.
79
+
80
+ Returns:
81
+ Options with ``name``, sorted case-insensitively.
82
+ """
83
+ cursor = self.client.cursor()
84
+ try:
85
+ cursor.execute(sql)
86
+ rows: list[Any] = cursor.fetchall()
87
+ index = [column[0] for column in cursor.description].index("name")
88
+ finally:
89
+ cursor.close()
90
+ return sorted(({"name": row[index]} for row in rows), key=lambda option: option["name"].lower())
91
+
92
+ @fetch_field_provider
93
+ def databases(self) -> list[dict[str, str]]:
94
+ """List the databases this connection's role can see.
95
+
96
+ Backs the destination's ``database`` ``FetchField``.
97
+
98
+ Returns:
99
+ Database options with ``name``.
100
+ """
101
+ return self._names("SHOW DATABASES")
102
+
103
+ @fetch_field_provider
104
+ def warehouses(self) -> list[dict[str, str]]:
105
+ """List the virtual warehouses this connection's role can see.
106
+
107
+ Backs the destination's ``warehouse`` ``FetchField``.
108
+
109
+ Returns:
110
+ Warehouse options with ``name``.
111
+ """
112
+ return self._names("SHOW WAREHOUSES")
113
+
114
+ def check(self) -> bool:
115
+ """Prove the credentials work by opening a session and running ``SELECT 1``.
116
+
117
+ Returns:
118
+ True; a login or network failure raises out of the connector.
119
+ """
120
+ cursor = self.client.cursor()
121
+ try:
122
+ cursor.execute("SELECT 1")
123
+ cursor.fetchall()
124
+ finally:
125
+ cursor.close()
126
+ return True
@@ -0,0 +1,448 @@
1
+ """Snowflake destination implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import tempfile
6
+ import threading
7
+ import uuid
8
+ import warnings
9
+ from collections.abc import Iterator, Sequence
10
+ from contextlib import contextmanager
11
+ from dataclasses import dataclass
12
+ from functools import cached_property
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ import pandas as pd
17
+ from interloper.destination import IOContext, destination
18
+ from interloper.destination.database import DatabaseDestination, PartitionFilter
19
+ from interloper.errors import ConfigError, DataNotFoundError
20
+ from interloper.representation import Representation
21
+ from interloper.resource.fields import FetchField, InputField
22
+ from interloper.schema import FieldSpec
23
+ from interloper.utils.data import is_empty
24
+ from pydantic import PrivateAttr
25
+ from snowflake.connector import SnowflakeConnection as Session
26
+ from snowflake.connector.cursor import SnowflakeCursor
27
+
28
+ from interloper_snowflake.connection import SnowflakeConnection
29
+ from interloper_snowflake.types import column_type
30
+
31
+ _STAGE = "interloper_load"
32
+
33
+
34
+ def _quote(identifier: str) -> str:
35
+ """Quote an identifier so Snowflake keeps it exactly as written.
36
+
37
+ Args:
38
+ identifier: A database, schema, table or column name.
39
+
40
+ Returns:
41
+ The identifier in double quotes, embedded quotes doubled.
42
+ """
43
+ return '"' + identifier.replace('"', '""') + '"'
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class _Staged:
48
+ """Data uploaded to a stage, ready to be copied into its table.
49
+
50
+ Attributes:
51
+ table: The target table name.
52
+ schema: The resolved schema name.
53
+ location: The stage path holding the file, ``@stage/prefix``.
54
+ columns: The file's columns, all of them the table's.
55
+ """
56
+
57
+ table: str
58
+ schema: str
59
+ location: str
60
+ columns: tuple[str, ...]
61
+
62
+
63
+ @destination(
64
+ key="snowflake_destination",
65
+ name="Snowflake",
66
+ icon="icon:snowflake",
67
+ tags=["Cloud"],
68
+ )
69
+ class SnowflakeDestination(DatabaseDestination):
70
+ """Snowflake destination.
71
+
72
+ A dataset is a Snowflake schema inside the destination's database. Every
73
+ identifier is quoted, so tables and columns keep the case the asset gives
74
+ them.
75
+
76
+ The destination opens its own session on its warehouse and database. That
77
+ session is still shared by every asset written through the destination,
78
+ and a Snowflake transaction belongs to the session, so writes are
79
+ serialised: two concurrent writes would otherwise commit or roll back each
80
+ other's statements.
81
+ """
82
+
83
+ connection: SnowflakeConnection
84
+
85
+ database: str = FetchField(
86
+ provider="connection.databases",
87
+ label_key="name",
88
+ value_key="name",
89
+ description="Snowflake database",
90
+ discriminator=True,
91
+ )
92
+ warehouse: str = FetchField(
93
+ provider="connection.warehouses",
94
+ label_key="name",
95
+ value_key="name",
96
+ description="Virtual warehouse running the loads and queries",
97
+ )
98
+ default_dataset: str | None = InputField(default=None, description="Default schema for assets without a dataset")
99
+
100
+ _stages: set[str] = PrivateAttr(default_factory=set)
101
+ _lock: Any = PrivateAttr(default_factory=threading.RLock)
102
+ _local: threading.local = PrivateAttr(default_factory=threading.local)
103
+
104
+ @cached_property
105
+ def client(self) -> Session:
106
+ """The destination's own session, on its warehouse and database.
107
+
108
+ Returns:
109
+ The connector session, cached per destination instance.
110
+ """
111
+ return self.connection.connect(warehouse=self.warehouse, database=self.database)
112
+
113
+ # -- Session -----------------------------------------------------------------
114
+
115
+ def _execute(self, sql: str, params: Sequence[Any] | None = None) -> SnowflakeCursor:
116
+ """Run one statement, on the open transaction's cursor or on a fresh one.
117
+
118
+ Args:
119
+ sql: The statement, with ``%s`` placeholders for *params*.
120
+ params: The statement's parameters; defaults to none.
121
+
122
+ Returns:
123
+ The cursor the statement ran on, ready to fetch from.
124
+ """
125
+ cursor = getattr(self._local, "cursor", None)
126
+ if cursor is None:
127
+ cursor = self.client.cursor()
128
+ cursor.execute(sql, params)
129
+ return cursor
130
+
131
+ @contextmanager
132
+ def transaction(self) -> Iterator[None]:
133
+ """Run one write, a delete followed by an insert, as ``BEGIN ... COMMIT``.
134
+
135
+ Yields:
136
+ ``None``; the write runs inside the block, rolled back and
137
+ re-raised if it raises.
138
+ """
139
+ with self._lock:
140
+ cursor = self.client.cursor()
141
+ cursor.execute("BEGIN")
142
+ self._local.cursor = cursor
143
+ try:
144
+ yield
145
+ except BaseException:
146
+ cursor.execute("ROLLBACK")
147
+ raise
148
+ else:
149
+ cursor.execute("COMMIT")
150
+ finally:
151
+ self._local.cursor = None
152
+ cursor.close()
153
+
154
+ # -- Destination interface -----------------------------------------------------
155
+
156
+ def write(self, context: IOContext, data: Any) -> None:
157
+ """Stage the data, then let the base replace its partitions.
158
+
159
+ Snowflake commits an open transaction whenever it runs DDL, and the
160
+ base calls :meth:`insert` inside :meth:`transaction`. So everything
161
+ that is DDL or file transfer (creating the schema, the table and the
162
+ stage, and the ``PUT``) runs here, before the base opens ``BEGIN``;
163
+ inside it, only the ``DELETE`` and the ``COPY INTO`` run. Both of the
164
+ base's paths, a single partition and a window, pass through here.
165
+
166
+ The whole write holds the destination's lock: the session is shared,
167
+ so another write's DDL would otherwise commit this one's open
168
+ transaction.
169
+
170
+ Args:
171
+ context: IO context carrying the target asset, the partition or window,
172
+ and the effective schema.
173
+ data: The data to write, in its native representation.
174
+ """
175
+ if is_empty(data):
176
+ return
177
+ table, dataset = self._target(context)
178
+ with self._lock:
179
+ self._local.staged = self._stage(table, dataset, data, context)
180
+ try:
181
+ super().write(context, data)
182
+ finally:
183
+ self._local.staged = None
184
+
185
+ # -- Naming --------------------------------------------------------------------
186
+
187
+ def _resolve_dataset(self, dataset: str | None) -> str:
188
+ """Return the Snowflake schema to use.
189
+
190
+ Args:
191
+ dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
192
+
193
+ Returns:
194
+ The resolved schema name.
195
+
196
+ Raises:
197
+ ConfigError: If neither the asset nor the destination names a dataset.
198
+ """
199
+ schema = dataset or self.default_dataset
200
+ if schema is None:
201
+ raise ConfigError(
202
+ "SnowflakeDestination requires a dataset. Either set 'dataset' on the asset "
203
+ "or provide 'default_dataset' on the destination."
204
+ )
205
+ return schema
206
+
207
+ def _table_ref(self, table: str, schema: str) -> str:
208
+ """Build a fully-qualified, quoted table reference.
209
+
210
+ Args:
211
+ table: Table name.
212
+ schema: The resolved schema name.
213
+
214
+ Returns:
215
+ ``"database"."schema"."table"``.
216
+ """
217
+ return f"{_quote(self.database)}.{_quote(schema)}.{_quote(table)}"
218
+
219
+ def _table_exists(self, table: str, schema: str) -> bool:
220
+ """Check whether a table exists, through the database's information schema.
221
+
222
+ Args:
223
+ table: Table name.
224
+ schema: The resolved schema name.
225
+
226
+ Returns:
227
+ ``True`` if the table exists, ``False`` otherwise.
228
+ """
229
+ cursor = self._execute(
230
+ f"SELECT 1 FROM {_quote(self.database)}.information_schema.tables "
231
+ "WHERE table_schema = %s AND table_name = %s",
232
+ (schema, table),
233
+ )
234
+ return bool(cursor.fetchall())
235
+
236
+ # -- Staging -------------------------------------------------------------------
237
+
238
+ def _ensure_table(self, table: str, schema: str, specs: Sequence[FieldSpec]) -> None:
239
+ """Create the schema and a typed table when the table does not exist.
240
+
241
+ Args:
242
+ table: Table name.
243
+ schema: The resolved schema name.
244
+ specs: The table's field specs.
245
+ """
246
+ if self._table_exists(table, schema):
247
+ return
248
+ self._execute(f"CREATE SCHEMA IF NOT EXISTS {_quote(self.database)}.{_quote(schema)}")
249
+ self._execute(f"CREATE TABLE IF NOT EXISTS {self._table_ref(table, schema)} ({_columns_ddl(specs)})")
250
+
251
+ def _ensure_stage(self, schema: str) -> str:
252
+ """Create the session's temporary stage in a schema, once per session.
253
+
254
+ Args:
255
+ schema: The resolved schema name.
256
+
257
+ Returns:
258
+ The stage's fully-qualified, quoted name.
259
+ """
260
+ stage = f"{_quote(self.database)}.{_quote(schema)}.{_quote(_STAGE)}"
261
+ if schema not in self._stages:
262
+ self._execute(f"CREATE TEMPORARY STAGE IF NOT EXISTS {stage}")
263
+ self._stages.add(schema)
264
+ return stage
265
+
266
+ def _stage(self, table: str, dataset: str | None, data: Any, context: IOContext) -> _Staged:
267
+ """Create what the load needs and upload the data as one Parquet file.
268
+
269
+ A new table is typed from the effective schema (declared on the asset,
270
+ or inferred during conform), or from a schema inferred from the data
271
+ when the context carries none. The frame is aligned to the table's
272
+ columns: an extra column is dropped with a warning, since an existing
273
+ table is never altered. The file goes under a prefix of its own, so
274
+ concurrent writes never load each other's files.
275
+
276
+ Args:
277
+ table: Target table name.
278
+ dataset: The Snowflake schema, or ``None`` for the destination's default.
279
+ data: The data in its native representation.
280
+ context: IO context carrying the asset and effective schema.
281
+
282
+ Returns:
283
+ Where the file was staged and the columns it carries.
284
+ """
285
+ schema = self._resolve_dataset(dataset)
286
+ specs = (context.schema or Representation.of(data).infer()).field_specs()
287
+ self._ensure_table(table, schema, specs)
288
+ stage = self._ensure_stage(schema)
289
+
290
+ frame = Representation.of(data).to("dataframe")
291
+ names = [spec.name for spec in specs]
292
+ extras = [str(c) for c in frame.columns if str(c) not in names]
293
+ if extras:
294
+ warnings.warn(
295
+ f"Columns {extras} are not in the schema for '{self._table_ref(table, schema)}' "
296
+ "and will not be written.",
297
+ UserWarning,
298
+ stacklevel=3,
299
+ )
300
+ columns = tuple(c for c in names if c in frame.columns)
301
+
302
+ location = f"@{stage}/{uuid.uuid4().hex}"
303
+ with tempfile.TemporaryDirectory() as directory:
304
+ path = Path(directory) / "data.parquet"
305
+ frame[list(columns)].to_parquet(path, index=False)
306
+ uri = path.as_posix().replace("'", "\\'")
307
+ self._execute(f"PUT 'file://{uri}' {location} OVERWRITE=TRUE AUTO_COMPRESS=FALSE")
308
+ return _Staged(table=table, schema=schema, location=location, columns=columns)
309
+
310
+ # -- DatabaseDestination hooks ---------------------------------------------------
311
+
312
+ def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
313
+ """Copy the staged data into the table.
314
+
315
+ :meth:`write` has already staged the data outside the transaction;
316
+ called any other way, the data is staged here first.
317
+
318
+ Args:
319
+ table: Target table name.
320
+ dataset: The Snowflake schema, or ``None`` for the destination's default.
321
+ data: The data in its native representation.
322
+ context: IO context carrying the asset and effective schema.
323
+ """
324
+ staged = getattr(self._local, "staged", None)
325
+ if staged is None or staged.table != table or staged.schema != self._resolve_dataset(dataset):
326
+ staged = self._stage(table, dataset, data, context)
327
+ self._copy(staged)
328
+
329
+ def _copy(self, staged: _Staged) -> None:
330
+ """Load a staged Parquet file with ``COPY INTO``, projecting its columns by name.
331
+
332
+ Parquet data is one ``$1`` object per row, so each column is read as
333
+ ``$1:"name"`` and the file's column order does not matter. Binary
334
+ columns stay binary, and logical types (dates, timestamps, decimals)
335
+ are honoured. The file is purged once loaded.
336
+
337
+ Args:
338
+ staged: The staged file and its columns.
339
+ """
340
+ targets = ", ".join(_quote(c) for c in staged.columns)
341
+ projection = ", ".join(f"$1:{_quote(c)}" for c in staged.columns)
342
+ self._execute(
343
+ f"COPY INTO {self._table_ref(staged.table, staged.schema)} ({targets}) "
344
+ f"FROM (SELECT {projection} FROM {staged.location}) "
345
+ "FILE_FORMAT=(TYPE=PARQUET USE_LOGICAL_TYPE=TRUE BINARY_AS_TEXT=FALSE) PURGE=TRUE"
346
+ )
347
+
348
+ def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
349
+ """Delete the rows a filter selects, or truncate the table.
350
+
351
+ A table that does not exist has nothing to delete.
352
+
353
+ Args:
354
+ table: Target table name.
355
+ dataset: The Snowflake schema, or ``None`` for the destination's default.
356
+ where: The rows to delete; ``None`` for the whole table.
357
+ """
358
+ schema = self._resolve_dataset(dataset)
359
+ if not self._table_exists(table, schema):
360
+ return
361
+ ref = self._table_ref(table, schema)
362
+ if where is None:
363
+ self._execute(f"TRUNCATE TABLE {ref}")
364
+ return
365
+ predicate, params = _predicate(where)
366
+ self._execute(f"DELETE FROM {ref} WHERE {predicate}", params)
367
+
368
+ def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
369
+ """Select the rows a filter selects, or every row, as a DataFrame.
370
+
371
+ The connector builds the frame from the result's Arrow batches, so
372
+ column types survive the read without a pass through Python records.
373
+
374
+ Args:
375
+ table: Target table name.
376
+ dataset: The Snowflake schema, or ``None`` for the destination's default.
377
+ where: The rows to select; ``None`` for the whole table.
378
+
379
+ Returns:
380
+ The selected rows.
381
+
382
+ Raises:
383
+ DataNotFoundError: If the table does not exist yet.
384
+ """
385
+ schema = self._resolve_dataset(dataset)
386
+ ref = self._table_ref(table, schema)
387
+ if not self._table_exists(table, schema):
388
+ raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
389
+ if where is None:
390
+ return self._execute(f"SELECT * FROM {ref}").fetch_pandas_all()
391
+ predicate, params = _predicate(where)
392
+ return self._execute(f"SELECT * FROM {ref} WHERE {predicate}", params).fetch_pandas_all()
393
+
394
+ def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
395
+ """Return row counts grouped by a column.
396
+
397
+ Args:
398
+ table: Target table name.
399
+ dataset: The Snowflake schema, or ``None`` for the destination's default.
400
+ column: Column to group by.
401
+
402
+ Returns:
403
+ Mapping from the column's value (as string) to row count.
404
+
405
+ Raises:
406
+ DataNotFoundError: If the table does not exist yet.
407
+ """
408
+ schema = self._resolve_dataset(dataset)
409
+ ref = self._table_ref(table, schema)
410
+ if not self._table_exists(table, schema):
411
+ raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
412
+ cursor = self._execute(
413
+ f"SELECT TO_VARCHAR({_quote(column)}) AS partition_value, COUNT(*) AS cnt FROM {ref} GROUP BY 1"
414
+ )
415
+ return {value: count for value, count in cursor.fetchall()}
416
+
417
+
418
+ # -- Utility functions ---------------------------------------------------------------
419
+
420
+
421
+ def _columns_ddl(specs: Sequence[FieldSpec]) -> str:
422
+ """Render field specs as the column list of a ``CREATE TABLE``.
423
+
424
+ Args:
425
+ specs: The table's field specs.
426
+
427
+ Returns:
428
+ Comma-separated quoted column definitions. Every column is nullable:
429
+ conform already enforces the schema's nullability, and a constraint
430
+ here would only turn a schema change into a failed load.
431
+ """
432
+ return ", ".join(f"{_quote(spec.name)} {column_type(spec)}" for spec in specs)
433
+
434
+
435
+ def _predicate(where: PartitionFilter) -> tuple[str, tuple[Any, ...]]:
436
+ """Render a partition filter as a parameterised SQL predicate.
437
+
438
+ Args:
439
+ where: The filter to render.
440
+
441
+ Returns:
442
+ The predicate text and the parameters its ``%s`` placeholders name.
443
+ """
444
+ column = _quote(where.column)
445
+ if where.bounds is None:
446
+ return f"{column} = %s", (where.value,)
447
+ start, end = where.bounds
448
+ return f"{column} >= %s AND {column} < %s", (start, end)
@@ -0,0 +1,43 @@
1
+ """Snowflake's view of interloper's field types."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ from decimal import Decimal
7
+
8
+ from interloper.schema import FieldSpec
9
+
10
+ # Ordered: the first base class that matches wins, so bool (a subclass of int)
11
+ # and datetime (a subclass of date) must come before their parents.
12
+ _PYTHON_TO_SNOWFLAKE: dict[type, str] = {
13
+ bool: "BOOLEAN",
14
+ int: "NUMBER(38,0)",
15
+ float: "FLOAT",
16
+ Decimal: "NUMBER(38,9)",
17
+ datetime.datetime: "TIMESTAMP_NTZ",
18
+ datetime.date: "DATE",
19
+ bytes: "BINARY",
20
+ str: "VARCHAR",
21
+ dict: "VARIANT",
22
+ list: "VARIANT",
23
+ }
24
+
25
+
26
+ def column_type(spec: FieldSpec) -> str:
27
+ """Return the Snowflake column type for a field spec.
28
+
29
+ Args:
30
+ spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
31
+
32
+ Returns:
33
+ ``VARIANT`` for a nested or repeated field, the type the spec's Python
34
+ type maps to otherwise, and ``VARCHAR`` for anything unmapped
35
+ (``typing.Any``).
36
+ """
37
+ if spec.fields is not None or spec.repeated:
38
+ return "VARIANT"
39
+ if isinstance(spec.type, type):
40
+ for base, name in _PYTHON_TO_SNOWFLAKE.items():
41
+ if issubclass(spec.type, base):
42
+ return name
43
+ return "VARCHAR"