interloper-databricks 0.96.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,9 @@
1
+ """Interloper Databricks integration: connection and destination."""
2
+
3
+ from interloper_databricks.connection import DatabricksConnection
4
+ from interloper_databricks.destination import DatabricksDestination
5
+
6
+ __all__ = [
7
+ "DatabricksConnection",
8
+ "DatabricksDestination",
9
+ ]
@@ -0,0 +1,298 @@
1
+ """Databricks connection resource holding service principal or personal access token credentials."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from collections.abc import Callable, Generator
7
+ from functools import cached_property
8
+ from typing import Any
9
+ from urllib.parse import urlsplit
10
+
11
+ import httpx2
12
+ from databricks import sql
13
+ from databricks.sql.client import Connection as Session
14
+ from interloper.connection import Connection, connection
15
+ from interloper.resource.fields import InputField, SecretField, fetch_field_provider
16
+ from interloper.rest import JSONCursorPaginator, RESTClient
17
+ from pydantic import PrivateAttr, field_validator, model_validator
18
+ from pydantic_settings import SettingsConfigDict
19
+
20
+ _TOKEN_ENDPOINT = "/oidc/v1/token"
21
+
22
+ #: A token this close to expiry is replaced before it is handed out, so a
23
+ #: statement never starts on a token that lapses mid-flight.
24
+ _REFRESH_MARGIN = 300.0
25
+
26
+ #: Every REST call here serves an operator waiting on a form or a check.
27
+ _TIMEOUT = 30.0
28
+
29
+
30
+ class _TokenAuth(httpx2.Auth):
31
+ """Bearer authentication reading the token from a callable on every request."""
32
+
33
+ def __init__(self, token: Callable[[], str]) -> None:
34
+ """Bind the auth to its token source.
35
+
36
+ Args:
37
+ token: Returns a valid access token, refreshing it when due.
38
+ """
39
+ self._token = token
40
+
41
+ def auth_flow(self, request: httpx2.Request) -> Generator[httpx2.Request, httpx2.Response, None]:
42
+ """Attach the current token as a bearer header.
43
+
44
+ Args:
45
+ request: The request to authenticate.
46
+
47
+ Yields:
48
+ The authenticated request.
49
+ """
50
+ request.headers["Authorization"] = f"Bearer {self._token()}"
51
+ yield request
52
+
53
+
54
+ @connection(
55
+ key="databricks_connection",
56
+ name="Databricks",
57
+ icon="icon:databricks",
58
+ tags=["Cloud"],
59
+ )
60
+ class DatabricksConnection(Connection):
61
+ """Connection resource holding Databricks workspace credentials.
62
+
63
+ Authenticates either as a service principal through OAuth
64
+ machine-to-machine (``client_id`` and ``client_secret``, the recommended
65
+ path) or with a personal access token; exactly one of the two must be
66
+ set. The connection's own check and pickers call the workspace REST API
67
+ over plain HTTP; each destination opens its own SQL session through
68
+ :meth:`connect`.
69
+ """
70
+
71
+ model_config = SettingsConfigDict(env_prefix="databricks_")
72
+
73
+ host: str = InputField(
74
+ label="Workspace URL",
75
+ description="e.g. https://dbc-a1b2345c-d6e7.cloud.databricks.com",
76
+ )
77
+ client_id: str | None = InputField(
78
+ default=None,
79
+ label="Client ID",
80
+ description="Service principal application ID",
81
+ section="Service principal (recommended)",
82
+ )
83
+ client_secret: str | None = SecretField(
84
+ default=None,
85
+ label="Client secret",
86
+ description="Service principal OAuth secret",
87
+ section="Service principal (recommended)",
88
+ )
89
+ access_token: str | None = SecretField(
90
+ default=None,
91
+ label="Access token",
92
+ description="Personal access token, instead of a service principal",
93
+ section="Personal access token",
94
+ )
95
+
96
+ _token: tuple[str, float] | None = PrivateAttr(default=None)
97
+
98
+ @field_validator("host")
99
+ @classmethod
100
+ def normalize_host(cls, value: str) -> str:
101
+ """Give the workspace URL a scheme and drop any trailing slash.
102
+
103
+ Args:
104
+ value: The workspace URL or bare hostname.
105
+
106
+ Returns:
107
+ The URL as ``https://<hostname>``.
108
+ """
109
+ value = value.strip().rstrip("/")
110
+ return value if "://" in value else f"https://{value}"
111
+
112
+ @field_validator("client_id", "client_secret", "access_token", mode="before")
113
+ @classmethod
114
+ def blank_is_unset(cls, value: Any) -> Any:
115
+ """Treat an empty credential, as a form submits it, as unset.
116
+
117
+ Args:
118
+ value: The raw field value.
119
+
120
+ Returns:
121
+ ``None`` for an empty string, the value otherwise.
122
+ """
123
+ return None if value == "" else value
124
+
125
+ @model_validator(mode="after")
126
+ def one_credential(self) -> DatabricksConnection:
127
+ """Require exactly one credential: a complete service principal or an access token.
128
+
129
+ Returns:
130
+ The validated connection.
131
+
132
+ Raises:
133
+ ValueError: If the service principal is half set, or if both or
134
+ neither credentials are set.
135
+ """
136
+ principal = self.client_id is not None or self.client_secret is not None
137
+ if principal and (self.client_id is None or self.client_secret is None):
138
+ raise ValueError("A service principal needs both 'client_id' and 'client_secret'")
139
+ if principal and self.access_token is not None:
140
+ raise ValueError("Set either a service principal or an access token, not both")
141
+ if not principal and self.access_token is None:
142
+ raise ValueError("Set a service principal ('client_id' and 'client_secret') or an 'access_token'")
143
+ return self
144
+
145
+ @property
146
+ def hostname(self) -> str:
147
+ """The workspace hostname, as the SQL connector takes it.
148
+
149
+ Returns:
150
+ The host without its scheme.
151
+ """
152
+ return urlsplit(self.host).netloc
153
+
154
+ # -- Credentials ---------------------------------------------------------------
155
+
156
+ def token(self) -> str:
157
+ """Return a bearer token for the workspace.
158
+
159
+ A personal access token is returned as is. A service principal's
160
+ token comes from the workspace's OAuth token endpoint (client
161
+ credentials, ``all-apis`` scope) and is cached until close to its
162
+ expiry; two threads refreshing at once only cost a second exchange.
163
+
164
+ Returns:
165
+ The access token.
166
+ """
167
+ if self.access_token is not None:
168
+ return self.access_token
169
+ cached = self._token
170
+ if cached is not None and cached[1] - time.monotonic() > _REFRESH_MARGIN:
171
+ return cached[0]
172
+ assert self.client_id is not None and self.client_secret is not None
173
+ with RESTClient(self.host, timeout=_TIMEOUT) as oauth:
174
+ response = oauth.post(
175
+ _TOKEN_ENDPOINT,
176
+ data={"grant_type": "client_credentials", "scope": "all-apis"},
177
+ auth=httpx2.BasicAuth(self.client_id, self.client_secret),
178
+ )
179
+ response.raise_for_status()
180
+ body = response.json()
181
+ self._token = (body["access_token"], time.monotonic() + float(body.get("expires_in", 3600)))
182
+ return body["access_token"]
183
+
184
+ def _authorization(self) -> dict[str, str]:
185
+ """Build the header the SQL connector sends with each request.
186
+
187
+ Returns:
188
+ The ``Authorization`` header carrying a current token.
189
+ """
190
+ return {"Authorization": f"Bearer {self.token()}"}
191
+
192
+ def _credentials_provider(self) -> Callable[[], dict[str, str]]:
193
+ """Hand the SQL connector its header factory.
194
+
195
+ The connector calls the provider once per session and the factory it
196
+ returns whenever it needs headers, so the factory refreshes the
197
+ service principal's token as it ages.
198
+
199
+ Returns:
200
+ The header factory.
201
+ """
202
+ return self._authorization
203
+
204
+ def connect(self, **session: Any) -> Session:
205
+ """Open a new SQL session on the workspace with this connection's credentials.
206
+
207
+ A personal access token goes to the connector as is; a service
208
+ principal goes through a ``credentials_provider`` built on this
209
+ connection's own token exchange, which spares the Databricks SDK as
210
+ a dependency.
211
+
212
+ Args:
213
+ **session: Session settings passed to the connector, such as
214
+ ``http_path`` and ``catalog``.
215
+
216
+ Returns:
217
+ The new connector session.
218
+ """
219
+ if self.access_token is not None:
220
+ credentials: dict[str, Any] = {"access_token": self.access_token}
221
+ else:
222
+ credentials = {"credentials_provider": self._credentials_provider}
223
+ return sql.connect(server_hostname=self.hostname, **credentials, **session)
224
+
225
+ # -- REST API ------------------------------------------------------------------
226
+
227
+ @cached_property
228
+ def client(self) -> RESTClient:
229
+ """The workspace REST API client the check and the pickers share.
230
+
231
+ Plain HTTP rather than the SQL connector: both run in the API
232
+ process, where opening a warehouse session would be slow and could
233
+ start a stopped warehouse.
234
+
235
+ Returns:
236
+ The bearer-authenticated client, cached per connection instance.
237
+ """
238
+ return RESTClient(self.host, auth=_TokenAuth(self.token), timeout=_TIMEOUT)
239
+
240
+ def _list(self, path: str, key: str, params: dict[str, str] | None = None) -> list[dict[str, Any]]:
241
+ """Walk a paginated list endpoint and collect its items.
242
+
243
+ Args:
244
+ path: The endpoint path.
245
+ key: The response key holding each page's items.
246
+ params: Static query parameters for every page.
247
+
248
+ Returns:
249
+ Every item across the pages.
250
+ """
251
+ pages = self.client.paginate(
252
+ path,
253
+ JSONCursorPaginator(cursor_path="next_page_token", cursor_param="page_token"),
254
+ params=params,
255
+ data_selector=lambda response: response.json().get(key, []),
256
+ )
257
+ return [item for page in pages for item in page]
258
+
259
+ @fetch_field_provider
260
+ def warehouses(self) -> list[dict[str, str]]:
261
+ """List the SQL warehouses this principal can use.
262
+
263
+ Backs the destination's ``warehouse`` ``FetchField``. The stored value
264
+ is the warehouse's HTTP path, which is what the SQL connector takes.
265
+
266
+ Returns:
267
+ Warehouse options with ``path`` and ``name``, sorted case-insensitively by name.
268
+ """
269
+ warehouses = self._list("/api/2.0/sql/warehouses", "warehouses")
270
+ options = [{"path": w["odbc_params"]["path"], "name": w["name"]} for w in warehouses]
271
+ return sorted(options, key=lambda option: option["name"].lower())
272
+
273
+ @fetch_field_provider
274
+ def catalogs(self) -> list[dict[str, str]]:
275
+ """List the Unity Catalog catalogs this principal can see.
276
+
277
+ Backs the destination's ``catalog`` ``FetchField``. ``max_results=0``
278
+ asks for the paginated form, which Databricks recommends over the
279
+ unpaginated one.
280
+
281
+ Returns:
282
+ Catalog options with ``name``, sorted case-insensitively.
283
+ """
284
+ catalogs = self._list("/api/2.1/unity-catalog/catalogs", "catalogs", params={"max_results": "0"})
285
+ return sorted(({"name": c["name"]} for c in catalogs), key=lambda option: option["name"].lower())
286
+
287
+ def check(self) -> bool:
288
+ """Prove the credentials work by asking the workspace who the caller is.
289
+
290
+ ``/api/2.0/preview/scim/v2/Me`` needs nothing beyond a valid token,
291
+ so a failure isolates a bad credential from a missing grant; for a
292
+ service principal it also exercises the token exchange.
293
+
294
+ Returns:
295
+ True; a rejected credential raises out of the HTTP call.
296
+ """
297
+ self.client.get("/api/2.0/preview/scim/v2/Me").raise_for_status()
298
+ return True
@@ -0,0 +1,665 @@
1
+ """Databricks destination implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ import inspect
7
+ import io
8
+ import json
9
+ import math
10
+ import threading
11
+ import uuid
12
+ import warnings
13
+ from collections.abc import Callable, Sequence
14
+ from dataclasses import dataclass
15
+ from decimal import Decimal
16
+ from functools import cached_property
17
+ from typing import Any
18
+
19
+ import pandas as pd
20
+ import pyarrow as pa
21
+ import pyarrow.parquet as pq
22
+ from databricks.sql.client import Connection as Session
23
+ from databricks.sql.client import Cursor
24
+ from interloper.destination import IOContext, destination
25
+ from interloper.destination.database import DatabaseDestination, PartitionFilter
26
+ from interloper.errors import ConfigError, DataNotFoundError
27
+ from interloper.partitioning.base import Partition
28
+ from interloper.representation import Representation
29
+ from interloper.resource.fields import FetchField, InputField
30
+ from interloper.schema import FieldSpec
31
+ from interloper.utils.data import is_empty
32
+ from interloper.utils.json import json_default, replace_non_finite
33
+ from pydantic import PrivateAttr, field_validator
34
+ from typing_extensions import Self
35
+
36
+ from interloper_databricks.connection import DatabricksConnection
37
+ from interloper_databricks.types import TIMESTAMP, VARIANT, clusterable, column_type
38
+
39
+ #: Delta collects statistics on a table's first 32 columns, and a clustering
40
+ #: key must be one of them.
41
+ _STATS_COLUMNS = 32
42
+
43
+
44
+ class _Lock:
45
+ """A re-entrant lock that a deep copy replaces with a fresh one.
46
+
47
+ A destination is deep-copied with the source it is bound to, and a bare
48
+ ``threading.RLock`` cannot be copied.
49
+ """
50
+
51
+ def __init__(self) -> None:
52
+ """Create the underlying lock."""
53
+ self._lock = threading.RLock()
54
+
55
+ def __enter__(self) -> Self:
56
+ """Acquire the lock.
57
+
58
+ Returns:
59
+ The lock.
60
+ """
61
+ self._lock.acquire()
62
+ return self
63
+
64
+ def __exit__(self, *exc: object) -> None:
65
+ """Release the lock.
66
+
67
+ Args:
68
+ *exc: The exception details, if the block raised; unused.
69
+ """
70
+ self._lock.release()
71
+
72
+ def __deepcopy__(self, memo: dict[int, Any]) -> _Lock:
73
+ """Copy as a new, unheld lock.
74
+
75
+ Args:
76
+ memo: The deep-copy memo; unused.
77
+
78
+ Returns:
79
+ A fresh lock.
80
+ """
81
+ return _Lock()
82
+
83
+
84
+ @dataclass(frozen=True)
85
+ class _Staged:
86
+ """Data uploaded to the staging volume, ready to be inserted.
87
+
88
+ Attributes:
89
+ path: The file's ``/Volumes/...`` path.
90
+ query: The ``SELECT`` reading the file, its columns cast to the table's types.
91
+ """
92
+
93
+ path: str
94
+ query: str
95
+
96
+
97
+ @destination(
98
+ key="databricks_destination",
99
+ name="Databricks",
100
+ icon="icon:databricks",
101
+ tags=["Cloud"],
102
+ )
103
+ class DatabricksDestination(DatabaseDestination):
104
+ """Databricks destination writing Delta tables in Unity Catalog through a SQL warehouse.
105
+
106
+ A dataset is a schema inside the destination's catalog. Every write
107
+ uploads the data as one Parquet file to the staging volume and loads it
108
+ with a single statement: ``INSERT INTO ... REPLACE WHERE`` for a
109
+ partition or a window, ``INSERT OVERWRITE`` for an unpartitioned asset.
110
+
111
+ That is why :meth:`write` and :meth:`write_partition` are overridden
112
+ rather than left to the base's delete-then-insert: Databricks has no
113
+ generally available multi-statement transaction, so a ``DELETE``
114
+ followed by an insert could leave a partition empty when the insert
115
+ fails, while ``REPLACE WHERE`` deletes and inserts in one Delta commit.
116
+ The rows to replace still come from the base's partition filters, and
117
+ :meth:`delete`, :meth:`select` and :meth:`count` remain the base's hooks.
118
+
119
+ The destination opens one SQL session, shared by every asset written
120
+ through it. The connector's sessions are not thread-safe, so every
121
+ statement runs under the destination's lock, and a write holds it from
122
+ the upload to the cleanup.
123
+ """
124
+
125
+ connection: DatabricksConnection
126
+
127
+ warehouse: str = FetchField(
128
+ provider="connection.warehouses",
129
+ label_key="name",
130
+ value_key="path",
131
+ label="SQL warehouse",
132
+ description="SQL warehouse running the loads and queries",
133
+ )
134
+ catalog: str = FetchField(
135
+ provider="connection.catalogs",
136
+ label_key="name",
137
+ value_key="name",
138
+ description="Unity Catalog catalog",
139
+ discriminator=True,
140
+ )
141
+ default_dataset: str | None = InputField(default=None, description="Default schema for assets without a dataset")
142
+ staging_volume: str = InputField(
143
+ label="Staging volume",
144
+ description="Unity Catalog volume the loads stage files in, as catalog.schema.volume",
145
+ )
146
+
147
+ _lock: _Lock = PrivateAttr(default_factory=_Lock)
148
+
149
+ @field_validator("staging_volume")
150
+ @classmethod
151
+ def three_part_volume(cls, value: str) -> str:
152
+ """Require a fully qualified volume name.
153
+
154
+ Args:
155
+ value: The volume name.
156
+
157
+ Returns:
158
+ The volume name, unchanged.
159
+
160
+ Raises:
161
+ ValueError: If the name is not ``catalog.schema.volume``.
162
+ """
163
+ parts = value.split(".")
164
+ if len(parts) != 3 or not all(parts):
165
+ raise ValueError(f"staging_volume must be 'catalog.schema.volume', got '{value}'")
166
+ return value
167
+
168
+ @property
169
+ def volume_path(self) -> str:
170
+ """The staging volume's path.
171
+
172
+ Returns:
173
+ ``/Volumes/<catalog>/<schema>/<volume>``.
174
+ """
175
+ return "/Volumes/" + "/".join(self.staging_volume.split("."))
176
+
177
+ @cached_property
178
+ def client(self) -> Session:
179
+ """The destination's own session, on its warehouse and catalog.
180
+
181
+ Returns:
182
+ The connector session, cached per destination instance.
183
+ """
184
+ return self.connection.connect(http_path=self.warehouse, catalog=self.catalog)
185
+
186
+ # -- Session -----------------------------------------------------------------
187
+
188
+ def _execute(
189
+ self,
190
+ sql: str,
191
+ *,
192
+ fetch: Callable[[Cursor], Any] | None = None,
193
+ input_stream: io.BytesIO | None = None,
194
+ ) -> Any:
195
+ """Run one statement on a fresh cursor, under the destination's lock.
196
+
197
+ Args:
198
+ sql: The statement.
199
+ fetch: Reads the result off the cursor; ``None`` when the
200
+ statement returns nothing of interest.
201
+ input_stream: The bytes a ``PUT '__input_stream__'`` uploads.
202
+
203
+ Returns:
204
+ What *fetch* returns, or ``None``.
205
+ """
206
+ with self._lock:
207
+ cursor = self.client.cursor()
208
+ try:
209
+ cursor.execute(sql, input_stream=input_stream)
210
+ return fetch(cursor) if fetch is not None else None
211
+ finally:
212
+ cursor.close()
213
+
214
+ # -- Destination interface -----------------------------------------------------
215
+
216
+ def write(self, context: IOContext, data: Any) -> None:
217
+ """Replace the partitions the context covers with the data, in one statement.
218
+
219
+ Args:
220
+ context: IO context carrying the target asset, the partition or window,
221
+ and the effective schema.
222
+ data: The data to write, in its native representation.
223
+ """
224
+ if is_empty(data):
225
+ return
226
+ self._warn_missing_partition_column(data, context)
227
+ self._replace(context, context.partitions, data)
228
+
229
+ def write_partition(self, context: IOContext, partition: Partition | None, data: Any) -> None:
230
+ """Replace one partition's rows with the data, in one statement.
231
+
232
+ Args:
233
+ context: IO context carrying the target asset and the effective schema.
234
+ partition: The partition being stored, or ``None`` for the whole table.
235
+ data: The data to store, in its native representation.
236
+ """
237
+ self._replace(context, [partition], data)
238
+
239
+ def _replace(self, context: IOContext, partitions: Sequence[Partition | None], data: Any) -> None:
240
+ """Load the data in place of the rows the partitions cover.
241
+
242
+ Args:
243
+ context: IO context carrying the target asset and the effective schema.
244
+ partitions: The partitions replaced; ``[None]`` for the whole table.
245
+ data: The data, in its native representation.
246
+ """
247
+ table, dataset = self._target(context)
248
+ filters = [self._filter(context, partition) for partition in partitions]
249
+ if None in filters:
250
+ self._load(table, dataset, data, context, "OVERWRITE")
251
+ return
252
+ where = _predicate([f for f in filters if f is not None])
253
+ self._load(table, dataset, data, context, "INTO", f" REPLACE WHERE {where}")
254
+
255
+ # -- Naming --------------------------------------------------------------------
256
+
257
+ def _resolve_dataset(self, dataset: str | None) -> str:
258
+ """Return the schema to use.
259
+
260
+ Args:
261
+ dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
262
+
263
+ Returns:
264
+ The resolved schema name.
265
+
266
+ Raises:
267
+ ConfigError: If neither the asset nor the destination names a dataset.
268
+ """
269
+ schema = dataset or self.default_dataset
270
+ if schema is None:
271
+ raise ConfigError(
272
+ "DatabricksDestination requires a dataset. Either set 'dataset' on the asset "
273
+ "or provide 'default_dataset' on the destination."
274
+ )
275
+ return schema
276
+
277
+ def _table_ref(self, table: str, schema: str) -> str:
278
+ """Build a fully-qualified, quoted table reference.
279
+
280
+ Args:
281
+ table: Table name.
282
+ schema: The resolved schema name.
283
+
284
+ Returns:
285
+ ```catalog`.`schema`.`table```.
286
+ """
287
+ return f"{_quote(self.catalog)}.{_quote(schema)}.{_quote(table)}"
288
+
289
+ def _table_exists(self, table: str, schema: str) -> bool:
290
+ """Check whether a table exists, through the catalog's information schema.
291
+
292
+ Unity Catalog stores object names in lower case, so the lookup does too.
293
+
294
+ Args:
295
+ table: Table name.
296
+ schema: The resolved schema name.
297
+
298
+ Returns:
299
+ ``True`` if the table exists, ``False`` otherwise.
300
+ """
301
+ rows = self._execute(
302
+ f"SELECT 1 FROM {_quote(self.catalog)}.information_schema.tables "
303
+ f"WHERE table_schema = {_literal(schema.lower())} AND table_name = {_literal(table.lower())}",
304
+ fetch=lambda cursor: cursor.fetchall(),
305
+ )
306
+ return bool(rows)
307
+
308
+ # -- Loading -------------------------------------------------------------------
309
+
310
+ def _ensure_table(self, table: str, schema: str, specs: Sequence[FieldSpec], context: IOContext) -> None:
311
+ """Create the schema and a typed Delta table when the table does not exist.
312
+
313
+ A partitioned asset's table is clustered on its partition column,
314
+ which every replace and partition read filters on, when liquid
315
+ clustering accepts that column as a key; Databricks recommends
316
+ clustering over partitioning for tables of this size. Field
317
+ descriptions become column comments, the asset's description the
318
+ table comment.
319
+
320
+ Args:
321
+ table: Table name.
322
+ schema: The resolved schema name.
323
+ specs: The table's field specs.
324
+ context: IO context carrying the asset.
325
+ """
326
+ if self._table_exists(table, schema):
327
+ return
328
+ self._execute(f"CREATE SCHEMA IF NOT EXISTS {_quote(self.catalog)}.{_quote(schema)}")
329
+ columns = ", ".join(_column_ddl(spec) for spec in specs)
330
+ sql = f"CREATE TABLE IF NOT EXISTS {self._table_ref(table, schema)} ({columns}) USING DELTA"
331
+ partitioning = context.asset.partitioning
332
+ keys = [spec for spec in specs[:_STATS_COLUMNS] if partitioning and spec.name == partitioning.column]
333
+ if keys and clusterable(keys[0]):
334
+ sql += f" CLUSTER BY ({_quote(keys[0].name)})"
335
+ description = _asset_description(context.asset)
336
+ if description:
337
+ sql += f" COMMENT {_literal(description)}"
338
+ self._execute(sql)
339
+
340
+ def _stage(self, table: str, schema: str, data: Any, context: IOContext) -> _Staged:
341
+ """Create what the load needs and upload the data as one Parquet file.
342
+
343
+ A new table is typed from the effective schema (declared on the asset,
344
+ or inferred during conform), or from a schema inferred from the data
345
+ when the context carries none. The frame is aligned to the schema's
346
+ columns: an extra column is dropped with a warning, since an existing
347
+ table is never altered. Each file gets a name of its own, so
348
+ concurrent writes never read each other's files.
349
+
350
+ Args:
351
+ table: Target table name.
352
+ schema: The resolved schema name.
353
+ data: The data in its native representation.
354
+ context: IO context carrying the asset and effective schema.
355
+
356
+ Returns:
357
+ Where the file was staged and the query reading it.
358
+ """
359
+ specs = (context.schema or Representation.of(data).infer()).field_specs()
360
+ self._ensure_table(table, schema, specs, context)
361
+
362
+ frame = Representation.of(data).to("dataframe")
363
+ names = [spec.name for spec in specs]
364
+ extras = [str(c) for c in frame.columns if str(c) not in names]
365
+ if extras:
366
+ warnings.warn(
367
+ f"Columns {extras} are not in the schema for '{self._table_ref(table, schema)}' "
368
+ "and will not be written.",
369
+ UserWarning,
370
+ stacklevel=4,
371
+ )
372
+ kept = [spec for spec in specs if spec.name in frame.columns]
373
+
374
+ path = f"{self.volume_path}/interloper/{uuid.uuid4().hex}.parquet"
375
+ self._execute(
376
+ f"PUT '__input_stream__' INTO {_literal(path)} OVERWRITE",
377
+ input_stream=io.BytesIO(_parquet(frame, kept)),
378
+ )
379
+ projection = ", ".join(_projection(spec) for spec in kept)
380
+ return _Staged(path=path, query=f"SELECT {projection} FROM read_files({_literal(path)}, format => 'parquet')")
381
+
382
+ def _load(
383
+ self,
384
+ table: str,
385
+ dataset: str | None,
386
+ data: Any,
387
+ context: IOContext,
388
+ mode: str,
389
+ clause: str = "",
390
+ ) -> None:
391
+ """Stage the data and insert it with one statement, removing the file afterwards.
392
+
393
+ The file is removed whether or not the insert succeeds; a failed
394
+ removal only warns, so it never hides the insert's own outcome.
395
+
396
+ Args:
397
+ table: Target table name.
398
+ dataset: The schema, or ``None`` for the destination's default.
399
+ data: The data in its native representation.
400
+ context: IO context carrying the asset and effective schema.
401
+ mode: ``INTO`` to add rows, ``OVERWRITE`` to replace the whole table.
402
+ clause: What follows the column matching, such as a ``REPLACE WHERE``.
403
+ """
404
+ schema = self._resolve_dataset(dataset)
405
+ with self._lock:
406
+ staged = self._stage(table, schema, data, context)
407
+ try:
408
+ self._execute(f"INSERT {mode} {self._table_ref(table, schema)} BY NAME{clause} {staged.query}")
409
+ finally:
410
+ try:
411
+ self._execute(f"REMOVE {_literal(staged.path)}")
412
+ except Exception as error: # noqa: BLE001
413
+ warnings.warn(
414
+ f"Could not remove the staged file '{staged.path}': {error}", UserWarning, stacklevel=2
415
+ )
416
+
417
+ # -- DatabaseDestination hooks ---------------------------------------------------
418
+
419
+ def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
420
+ """Append the data to the table through a staged Parquet file.
421
+
422
+ Args:
423
+ table: Target table name.
424
+ dataset: The schema, or ``None`` for the destination's default.
425
+ data: The data in its native representation.
426
+ context: IO context carrying the asset and effective schema.
427
+ """
428
+ self._load(table, dataset, data, context, "INTO")
429
+
430
+ def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
431
+ """Delete the rows a filter selects, or every row.
432
+
433
+ A table that does not exist has nothing to delete.
434
+
435
+ Args:
436
+ table: Target table name.
437
+ dataset: The schema, or ``None`` for the destination's default.
438
+ where: The rows to delete; ``None`` for the whole table.
439
+ """
440
+ schema = self._resolve_dataset(dataset)
441
+ if not self._table_exists(table, schema):
442
+ return
443
+ sql = f"DELETE FROM {self._table_ref(table, schema)}"
444
+ self._execute(sql if where is None else f"{sql} WHERE {_predicate([where])}")
445
+
446
+ def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
447
+ """Select the rows a filter selects, or every row, as a DataFrame.
448
+
449
+ The result arrives as Arrow, so column types survive the read without
450
+ a pass through Python records.
451
+
452
+ Args:
453
+ table: Target table name.
454
+ dataset: The schema, or ``None`` for the destination's default.
455
+ where: The rows to select; ``None`` for the whole table.
456
+
457
+ Returns:
458
+ The selected rows.
459
+
460
+ Raises:
461
+ DataNotFoundError: If the table does not exist yet.
462
+ """
463
+ schema = self._resolve_dataset(dataset)
464
+ ref = self._table_ref(table, schema)
465
+ if not self._table_exists(table, schema):
466
+ raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
467
+ sql = f"SELECT * FROM {ref}" if where is None else f"SELECT * FROM {ref} WHERE {_predicate([where])}"
468
+ return self._execute(sql, fetch=lambda cursor: cursor.fetchall_arrow().to_pandas())
469
+
470
+ def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
471
+ """Return row counts grouped by a column.
472
+
473
+ Args:
474
+ table: Target table name.
475
+ dataset: The schema, or ``None`` for the destination's default.
476
+ column: Column to group by.
477
+
478
+ Returns:
479
+ Mapping from the column's value (as string) to row count.
480
+
481
+ Raises:
482
+ DataNotFoundError: If the table does not exist yet.
483
+ """
484
+ schema = self._resolve_dataset(dataset)
485
+ ref = self._table_ref(table, schema)
486
+ if not self._table_exists(table, schema):
487
+ raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
488
+ rows = self._execute(
489
+ f"SELECT CAST({_quote(column)} AS STRING) AS partition_value, COUNT(*) AS cnt FROM {ref} GROUP BY 1",
490
+ fetch=lambda cursor: cursor.fetchall(),
491
+ )
492
+ return {row[0]: row[1] for row in rows}
493
+
494
+
495
+ # -- Utility functions ---------------------------------------------------------------
496
+
497
+
498
+ def _quote(identifier: str) -> str:
499
+ """Quote an identifier with backticks.
500
+
501
+ Args:
502
+ identifier: A catalog, schema, table or column name.
503
+
504
+ Returns:
505
+ The identifier in backticks, embedded backticks doubled.
506
+ """
507
+ return "`" + identifier.replace("`", "``") + "`"
508
+
509
+
510
+ def _literal(value: Any) -> str:
511
+ """Render a value as a Databricks SQL literal.
512
+
513
+ Every value a statement here carries is a literal: ``REPLACE WHERE``
514
+ admits literal values in its predicate, and ``PUT``, ``REMOVE`` and
515
+ ``read_files`` take their paths as string literals, so one renderer
516
+ serves them all. A naive datetime is read as UTC, as the load writes it.
517
+
518
+ Args:
519
+ value: A string, date, datetime, number, boolean or ``None``.
520
+
521
+ Returns:
522
+ The literal: a backslash-escaped string, ``DATE'...'``,
523
+ ``TIMESTAMP'...'``, a bare number, ``TRUE``/``FALSE`` or ``NULL``.
524
+ """
525
+ if value is None:
526
+ return "NULL"
527
+ if isinstance(value, bool):
528
+ return "TRUE" if value else "FALSE"
529
+ if isinstance(value, (int, float, Decimal)):
530
+ return str(value)
531
+ if isinstance(value, datetime.datetime):
532
+ stamp = value.isoformat() if value.tzinfo is not None else f"{value.isoformat()}Z"
533
+ return f"TIMESTAMP'{stamp}'"
534
+ if isinstance(value, datetime.date):
535
+ return f"DATE'{value.isoformat()}'"
536
+ return "'" + str(value).replace("\\", "\\\\").replace("'", "\\'") + "'"
537
+
538
+
539
+ def _predicate(filters: Sequence[PartitionFilter]) -> str:
540
+ """Render the rows several partition filters cover as one predicate.
541
+
542
+ Time partitions contribute their half-open bounds, merged where they
543
+ touch, so a contiguous window becomes the single range from its first
544
+ start to its last end. Other partitions match their ids by equality.
545
+
546
+ Args:
547
+ filters: The filters, all on the same column.
548
+
549
+ Returns:
550
+ The predicate text.
551
+ """
552
+ column = _quote(filters[0].column)
553
+ ranges: list[tuple[Any, Any]] = []
554
+ for start, end in sorted(f.bounds for f in filters if f.bounds is not None):
555
+ if ranges and start <= ranges[-1][1]:
556
+ ranges[-1] = (ranges[-1][0], max(ranges[-1][1], end))
557
+ else:
558
+ ranges.append((start, end))
559
+ parts = [f"{column} >= {_literal(start)} AND {column} < {_literal(end)}" for start, end in ranges]
560
+ values = [f.value for f in filters if f.bounds is None]
561
+ if len(values) == 1:
562
+ parts.append(f"{column} = {_literal(values[0])}")
563
+ elif values:
564
+ parts.append(f"{column} IN ({', '.join(_literal(v) for v in values)})")
565
+ return parts[0] if len(parts) == 1 else " OR ".join(f"({part})" for part in parts)
566
+
567
+
568
+ def _column_ddl(spec: FieldSpec) -> str:
569
+ """Render a field spec as a column definition.
570
+
571
+ Args:
572
+ spec: The field spec.
573
+
574
+ Returns:
575
+ The quoted column and its type, with its description as a comment.
576
+ Every column is nullable: conform already enforces the schema's
577
+ nullability, and a constraint here would only turn a schema change
578
+ into a failed load.
579
+ """
580
+ ddl = f"{_quote(spec.name)} {column_type(spec)}"
581
+ return f"{ddl} COMMENT {_literal(spec.description)}" if spec.description else ddl
582
+
583
+
584
+ def _projection(spec: FieldSpec) -> str:
585
+ """Read one staged column as the table's type.
586
+
587
+ Args:
588
+ spec: The column's field spec.
589
+
590
+ Returns:
591
+ ``PARSE_JSON`` of the JSON text for a ``VARIANT``, a ``CAST`` to the
592
+ column type otherwise, aliased to the column so ``BY NAME`` matches it.
593
+ """
594
+ name = _quote(spec.name)
595
+ kind = column_type(spec)
596
+ expression = f"PARSE_JSON({name})" if kind == VARIANT else f"CAST({name} AS {kind})"
597
+ return f"{expression} AS {name}"
598
+
599
+
600
+ def _parquet(frame: pd.DataFrame, specs: Sequence[FieldSpec]) -> bytes:
601
+ """Encode the columns a load keeps as one Parquet file.
602
+
603
+ This only chooses the file's encoding; the values are already conformed.
604
+ ``VARIANT`` columns travel as JSON text for ``PARSE_JSON``, and
605
+ ``TIMESTAMP`` columns as UTC instants in microseconds (a naive datetime
606
+ read as UTC, so the session's time zone never shifts them; Databricks
607
+ timestamps hold microseconds, and Spark does not read nanosecond Parquet
608
+ timestamps as timestamps). A column with no
609
+ values at all is written as strings, since Parquet's null type is not
610
+ something every reader accepts; the load casts it to its real type.
611
+
612
+ Args:
613
+ frame: The data as a DataFrame.
614
+ specs: The kept columns' field specs, in table order.
615
+
616
+ Returns:
617
+ The Parquet file's bytes.
618
+ """
619
+ columns: dict[str, pd.Series] = {}
620
+ for spec in specs:
621
+ series = frame[spec.name]
622
+ kind = column_type(spec)
623
+ if kind == VARIANT:
624
+ series = series.map(_json)
625
+ elif kind == TIMESTAMP:
626
+ series = pd.to_datetime(series, utc=True)
627
+ columns[spec.name] = series
628
+ table = pa.Table.from_pandas(pd.DataFrame(columns, index=frame.index), preserve_index=False)
629
+ for index, field in enumerate(table.schema):
630
+ if pa.types.is_null(field.type):
631
+ table = table.set_column(index, field.name, table.column(index).cast(pa.string()))
632
+ buffer = io.BytesIO()
633
+ pq.write_table(table, buffer, coerce_timestamps="us", allow_truncated_timestamps=True)
634
+ return buffer.getvalue()
635
+
636
+
637
+ def _json(value: Any) -> str | None:
638
+ """Encode one nested value as JSON text.
639
+
640
+ Args:
641
+ value: A dict, a list, an array or a missing value.
642
+
643
+ Returns:
644
+ The JSON text, or ``None`` for a missing value.
645
+ """
646
+ if value is None or value is pd.NA or (isinstance(value, float) and math.isnan(value)):
647
+ return None
648
+ if hasattr(value, "tolist"):
649
+ value = value.tolist()
650
+ return json.dumps(replace_non_finite(value), default=json_default)
651
+
652
+
653
+ def _asset_description(asset: Any) -> str | None:
654
+ """Return the asset's description (its class docstring), cleaned.
655
+
656
+ This mirrors how ``Component.definition()`` derives descriptions.
657
+
658
+ Args:
659
+ asset: The asset whose docstring is read.
660
+
661
+ Returns:
662
+ The cleaned docstring, or ``None`` when the asset has none.
663
+ """
664
+ doc = type(asset).__doc__
665
+ return inspect.cleandoc(doc) if doc else None
@@ -0,0 +1,67 @@
1
+ """Databricks' view of interloper's field types."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ from decimal import Decimal
7
+
8
+ from interloper.schema import FieldSpec
9
+
10
+ VARIANT = "VARIANT"
11
+ TIMESTAMP = "TIMESTAMP"
12
+
13
+ # Ordered: the first base class that matches wins, so bool (a subclass of int)
14
+ # and datetime (a subclass of date) must come before their parents.
15
+ _PYTHON_TO_DATABRICKS: dict[type, str] = {
16
+ bool: "BOOLEAN",
17
+ int: "BIGINT",
18
+ float: "DOUBLE",
19
+ Decimal: "DECIMAL(38,9)",
20
+ datetime.datetime: TIMESTAMP,
21
+ datetime.date: "DATE",
22
+ bytes: "BINARY",
23
+ str: "STRING",
24
+ dict: VARIANT,
25
+ list: VARIANT,
26
+ }
27
+
28
+ _CLUSTERABLE = frozenset({"BIGINT", "DOUBLE", "DECIMAL(38,9)", TIMESTAMP, "DATE", "STRING"})
29
+
30
+
31
+ def column_type(spec: FieldSpec) -> str:
32
+ """Return the Databricks column type for a field spec.
33
+
34
+ A ``datetime`` is a ``TIMESTAMP``, an absolute instant like BigQuery's:
35
+ ``TIMESTAMP_NTZ`` is still in Public Preview and upgrades the Delta table
36
+ protocol. Nested and repeated fields are ``VARIANT``, Databricks' type for
37
+ semi-structured data, queryable with the ``:`` path operator.
38
+
39
+ Args:
40
+ spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
41
+
42
+ Returns:
43
+ ``VARIANT`` for a nested or repeated field, the type the spec's Python
44
+ type maps to otherwise, and ``STRING`` for anything unmapped
45
+ (``typing.Any``).
46
+ """
47
+ if spec.fields is not None or spec.repeated:
48
+ return VARIANT
49
+ if isinstance(spec.type, type):
50
+ for base, name in _PYTHON_TO_DATABRICKS.items():
51
+ if issubclass(spec.type, base):
52
+ return name
53
+ return "STRING"
54
+
55
+
56
+ def clusterable(spec: FieldSpec) -> bool:
57
+ """Whether a field can be a liquid clustering key.
58
+
59
+ Args:
60
+ spec: The field spec.
61
+
62
+ Returns:
63
+ True when the field's column type is one liquid clustering accepts as a
64
+ key (dates, timestamps, strings and numbers; not booleans, binary or
65
+ ``VARIANT``).
66
+ """
67
+ return column_type(spec) in _CLUSTERABLE
@@ -0,0 +1,190 @@
1
+ Metadata-Version: 2.3
2
+ Name: interloper-databricks
3
+ Version: 0.96.0
4
+ Summary: Interloper Databricks integration: connection and destination
5
+ Author: Guillaume Onfroy
6
+ Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
7
+ Requires-Dist: interloper-core
8
+ Requires-Dist: interloper-pandas
9
+ Requires-Dist: databricks-sql-connector[pyarrow]>=4.1.2
10
+ Requires-Python: >=3.10
11
+ Description-Content-Type: text/markdown
12
+
13
+ # interloper-databricks
14
+
15
+ Databricks integration for interloper: a `DatabricksDestination` that stores
16
+ assets as Delta tables in Unity Catalog through a SQL warehouse, and the
17
+ `DatabricksConnection` that holds the workspace credentials.
18
+
19
+ The destination is a `DatabaseDestination`: the rows a partition covers come
20
+ from core, exactly as for BigQuery. What differs is how they are replaced (see
21
+ [Partitions](#partitions)).
22
+
23
+ ## Setup
24
+
25
+ ### Credentials
26
+
27
+ A service principal with an OAuth secret is the recommended credential; a
28
+ personal access token works too. The connection takes exactly one of them.
29
+
30
+ 1. In the account console (or the workspace admin settings), create a service
31
+ principal and add it to the workspace.
32
+ 2. Under its **Secrets** tab, click **Generate secret**. Note the client ID
33
+ and the secret; the secret is shown once. The connection asks for the
34
+ `all-apis` scope, so leave the secret unscoped (or select all APIs): a
35
+ secret restricted to some scopes cannot issue that token.
36
+ 3. Give the service principal the **Can use** permission on the SQL warehouse
37
+ the destination loads through.
38
+
39
+ The connection exchanges the client ID and secret for a one-hour access token
40
+ at the workspace's `/oidc/v1/token` endpoint (client credentials grant), and
41
+ replaces the token before it expires. The connection check calls
42
+ `/api/2.0/preview/scim/v2/Me`, and the pickers list SQL warehouses and Unity
43
+ Catalog catalogs over the REST API, all with that token.
44
+
45
+ ### Grants
46
+
47
+ The principal needs, in Unity Catalog:
48
+
49
+ - `USE CATALOG` on the destination's catalog
50
+ - `CREATE SCHEMA` on the catalog, or `USE SCHEMA` on every schema the assets
51
+ write to if you create them yourself
52
+ - `CREATE TABLE` on those schemas
53
+ - `MODIFY` and `SELECT` on the tables (a table the principal creates is its
54
+ own, so it has both)
55
+ - `USE SCHEMA`, `READ VOLUME` and `WRITE VOLUME` on the staging volume and its
56
+ schema
57
+
58
+ A sketch, with the service principal's application ID as the grantee:
59
+
60
+ ```sql
61
+ GRANT USE CATALOG, CREATE SCHEMA ON CATALOG analytics TO `<application-id>`;
62
+ GRANT USE SCHEMA ON SCHEMA analytics.staging TO `<application-id>`;
63
+ GRANT READ VOLUME, WRITE VOLUME ON VOLUME analytics.staging.loads TO `<application-id>`;
64
+ ```
65
+
66
+ ### Staging volume
67
+
68
+ Every load uploads one Parquet file to a Unity Catalog volume, reads it from
69
+ there and removes it. Create a managed volume for it once:
70
+
71
+ ```sql
72
+ CREATE SCHEMA IF NOT EXISTS analytics.staging;
73
+ CREATE VOLUME IF NOT EXISTS analytics.staging.loads COMMENT 'interloper load staging';
74
+ ```
75
+
76
+ and name it on the destination as `analytics.staging.loads`. Files go under
77
+ `/Volumes/analytics/staging/loads/interloper/`, one per write. A file whose
78
+ removal fails (the load's outcome stands, with a warning) is left there and
79
+ can be deleted by hand.
80
+
81
+ ## Usage
82
+
83
+ ```python
84
+ import interloper as il
85
+ from interloper_databricks import DatabricksConnection, DatabricksDestination
86
+
87
+ destination = DatabricksDestination(
88
+ connection=DatabricksConnection(
89
+ host="https://dbc-a1b2345c-d6e7.cloud.databricks.com",
90
+ client_id="...",
91
+ client_secret="...",
92
+ ),
93
+ warehouse="/sql/1.0/warehouses/a1b234c567d8e9fa",
94
+ catalog="analytics",
95
+ staging_volume="analytics.staging.loads",
96
+ default_dataset="raw",
97
+ )
98
+ ```
99
+
100
+ `warehouse` is the warehouse's HTTP path (its **Connection details** tab). The
101
+ credentials also load from the environment (`DATABRICKS_HOST`,
102
+ `DATABRICKS_CLIENT_ID`, `DATABRICKS_CLIENT_SECRET`, or
103
+ `DATABRICKS_ACCESS_TOKEN` for a personal access token; note that Databricks'
104
+ own tools name the token `DATABRICKS_TOKEN`).
105
+
106
+ In a deployed instance you configure this through the UI instead: add a
107
+ Databricks connection, then a Databricks destination, picking the warehouse
108
+ and the catalog from the lists the connection can see.
109
+
110
+ ## Datasets are schemas
111
+
112
+ An asset's `dataset` is the schema its table lives in, inside the
113
+ destination's `catalog`. An asset without a dataset falls back to
114
+ `default_dataset`; with neither, the write fails with a `ConfigError`. A
115
+ missing schema is created on the first write, and a missing table is created
116
+ as a Delta table typed from the asset's schema (or one inferred from the
117
+ data), with field descriptions as column comments and the asset's description
118
+ as the table comment:
119
+
120
+ | Field type | Databricks type |
121
+ |------------|-----------------|
122
+ | `bool` | `BOOLEAN` |
123
+ | `int` | `BIGINT` |
124
+ | `float` | `DOUBLE` |
125
+ | `Decimal` | `DECIMAL(38,9)` |
126
+ | `datetime` | `TIMESTAMP` |
127
+ | `date` | `DATE` |
128
+ | `bytes` | `BINARY` |
129
+ | `str`, `Any` | `STRING` |
130
+ | nested models, lists, dicts | `VARIANT` |
131
+
132
+ A `datetime` is a `TIMESTAMP`, an absolute instant; a naive datetime is taken
133
+ as UTC. `VARIANT` columns are queried with the path operator
134
+ (`SELECT payload:campaign.id FROM ...`); creating a table with one enables
135
+ Delta's `variantType` feature, which readers need Databricks Runtime 15.4 or a
136
+ recent Delta client for.
137
+
138
+ Identifiers are quoted with backticks. Unity Catalog stores schema and table
139
+ names in lower case and keeps column names as written; queries match either
140
+ case-insensitively.
141
+
142
+ An existing table is never altered: a column the data carries but the schema
143
+ does not is dropped with a warning.
144
+
145
+ A partitioned asset's table is clustered on its partition column with liquid
146
+ clustering (`CLUSTER BY`), which is what Databricks recommends over
147
+ partitioning for tables under 1 TB, when that column is one liquid clustering
148
+ accepts as a key (a date, timestamp, string or number among the first 32
149
+ columns). Tables are never `PARTITIONED BY`.
150
+
151
+ ## Partitions
152
+
153
+ Each write is one statement, so a partition is replaced atomically:
154
+
155
+ - a time partition:
156
+ ``INSERT INTO ... BY NAME REPLACE WHERE `day` >= DATE'2024-01-01' AND `day` < DATE'2024-01-02' SELECT ...``
157
+ (half-open bounds, so a monthly partition whose rows hold daily dates is
158
+ replaced whole)
159
+ - any other partition: ``REPLACE WHERE `region` = 'eu'``
160
+ - a window: one predicate covering every partition in it, a single range from
161
+ its first start to its last end when the partitions are contiguous
162
+ - an unpartitioned asset: `INSERT OVERWRITE ... BY NAME SELECT ...`
163
+
164
+ The `SELECT` reads the staged file with `read_files(..., format => 'parquet')`,
165
+ casting each column to the table's type. `REPLACE WHERE` deletes the matching
166
+ rows and inserts the new ones in a single Delta commit, so a failed load
167
+ leaves the partition's old rows in place.
168
+
169
+ `REPLACE WHERE` is strict: every row written must match the predicate, or the
170
+ statement fails with `DELTA_REPLACE_WHERE_MISMATCH` and writes nothing. A row
171
+ whose partition column is null, or falls outside the partition, fails the
172
+ write instead of landing where the next replace would not remove it.
173
+
174
+ Reads come back as a DataFrame built from the result's Arrow batches.
175
+
176
+ ## Notes
177
+
178
+ Each destination opens its own SQL session through the connection, on its own
179
+ warehouse and catalog, so destinations sharing a connection never affect each
180
+ other. That session is shared by every asset the destination writes, and the
181
+ connector's sessions are not thread-safe, so the destination runs one
182
+ statement at a time and holds a write from its upload to its cleanup.
183
+
184
+ The SQL warehouse does the loading work: it reads the staged file, casts it
185
+ and rewrites the replaced rows, so its size bounds load throughput. Since a
186
+ destination runs one statement at a time, size the warehouse for the largest
187
+ single write (a partition, or a whole window) rather than for concurrency;
188
+ several destinations, or several runs, can share a warehouse that scales out.
189
+ A stopped warehouse adds its start-up time to the first write after it
190
+ auto-stops, which serverless warehouses keep short.
@@ -0,0 +1,8 @@
1
+ interloper_databricks/__init__.py,sha256=ZUOTzYoMMHFFmOjZMl0mZC-Qr5gZ1z3vy1_aD6axwP0,276
2
+ interloper_databricks/connection.py,sha256=paI0zI7UglgfmIvvVtHdQqMy9aI-CV45Ep0F7AaeKfE,11310
3
+ interloper_databricks/destination.py,sha256=LV8MOE6ioex4vctmka-psicqJa1TvzkHldc4cbh7I8Y,25338
4
+ interloper_databricks/types.py,sha256=LdYpqX0FTuz5AIKLWCo9YQ6qOKYld69OZUgZ5qELmK4,2073
5
+ interloper_databricks-0.96.0.dist-info/WHEEL,sha256=e4_1dyBeezi8ZjfxrZ3bnVOxFDa3ksqVqH0jTHkUZ3k,81
6
+ interloper_databricks-0.96.0.dist-info/entry_points.txt,sha256=-iwSBH6L6INfkEsrZH_S6kCznO_R8E_ObMG_TD-KaUk,60
7
+ interloper_databricks-0.96.0.dist-info/METADATA,sha256=nmjeljZw6vDqZIUS07HknFdYY0Bs67bMZiLe1isq08Q,8044
8
+ interloper_databricks-0.96.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: uv 0.12.19
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [interloper.components]
2
+ databricks = interloper_databricks
3
+