interloper-databricks 0.96.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interloper_databricks-0.96.0/PKG-INFO +190 -0
- interloper_databricks-0.96.0/README.md +178 -0
- interloper_databricks-0.96.0/pyproject.toml +63 -0
- interloper_databricks-0.96.0/pyproject.toml.orig +43 -0
- interloper_databricks-0.96.0/src/interloper_databricks/__init__.py +9 -0
- interloper_databricks-0.96.0/src/interloper_databricks/connection.py +298 -0
- interloper_databricks-0.96.0/src/interloper_databricks/destination.py +665 -0
- interloper_databricks-0.96.0/src/interloper_databricks/types.py +67 -0
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: interloper-databricks
|
|
3
|
+
Version: 0.96.0
|
|
4
|
+
Summary: Interloper Databricks integration: connection and destination
|
|
5
|
+
Author: Guillaume Onfroy
|
|
6
|
+
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
+
Requires-Dist: interloper-core
|
|
8
|
+
Requires-Dist: interloper-pandas
|
|
9
|
+
Requires-Dist: databricks-sql-connector[pyarrow]>=4.1.2
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# interloper-databricks
|
|
14
|
+
|
|
15
|
+
Databricks integration for interloper: a `DatabricksDestination` that stores
|
|
16
|
+
assets as Delta tables in Unity Catalog through a SQL warehouse, and the
|
|
17
|
+
`DatabricksConnection` that holds the workspace credentials.
|
|
18
|
+
|
|
19
|
+
The destination is a `DatabaseDestination`: the rows a partition covers come
|
|
20
|
+
from core, exactly as for BigQuery. What differs is how they are replaced (see
|
|
21
|
+
[Partitions](#partitions)).
|
|
22
|
+
|
|
23
|
+
## Setup
|
|
24
|
+
|
|
25
|
+
### Credentials
|
|
26
|
+
|
|
27
|
+
A service principal with an OAuth secret is the recommended credential; a
|
|
28
|
+
personal access token works too. The connection takes exactly one of them.
|
|
29
|
+
|
|
30
|
+
1. In the account console (or the workspace admin settings), create a service
|
|
31
|
+
principal and add it to the workspace.
|
|
32
|
+
2. Under its **Secrets** tab, click **Generate secret**. Note the client ID
|
|
33
|
+
and the secret; the secret is shown once. The connection asks for the
|
|
34
|
+
`all-apis` scope, so leave the secret unscoped (or select all APIs): a
|
|
35
|
+
secret restricted to some scopes cannot issue that token.
|
|
36
|
+
3. Give the service principal the **Can use** permission on the SQL warehouse
|
|
37
|
+
the destination loads through.
|
|
38
|
+
|
|
39
|
+
The connection exchanges the client ID and secret for a one-hour access token
|
|
40
|
+
at the workspace's `/oidc/v1/token` endpoint (client credentials grant), and
|
|
41
|
+
replaces the token before it expires. The connection check calls
|
|
42
|
+
`/api/2.0/preview/scim/v2/Me`, and the pickers list SQL warehouses and Unity
|
|
43
|
+
Catalog catalogs over the REST API, all with that token.
|
|
44
|
+
|
|
45
|
+
### Grants
|
|
46
|
+
|
|
47
|
+
The principal needs, in Unity Catalog:
|
|
48
|
+
|
|
49
|
+
- `USE CATALOG` on the destination's catalog
|
|
50
|
+
- `CREATE SCHEMA` on the catalog, or `USE SCHEMA` on every schema the assets
|
|
51
|
+
write to if you create them yourself
|
|
52
|
+
- `CREATE TABLE` on those schemas
|
|
53
|
+
- `MODIFY` and `SELECT` on the tables (a table the principal creates is its
|
|
54
|
+
own, so it has both)
|
|
55
|
+
- `USE SCHEMA`, `READ VOLUME` and `WRITE VOLUME` on the staging volume and its
|
|
56
|
+
schema
|
|
57
|
+
|
|
58
|
+
A sketch, with the service principal's application ID as the grantee:
|
|
59
|
+
|
|
60
|
+
```sql
|
|
61
|
+
GRANT USE CATALOG, CREATE SCHEMA ON CATALOG analytics TO `<application-id>`;
|
|
62
|
+
GRANT USE SCHEMA ON SCHEMA analytics.staging TO `<application-id>`;
|
|
63
|
+
GRANT READ VOLUME, WRITE VOLUME ON VOLUME analytics.staging.loads TO `<application-id>`;
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
### Staging volume
|
|
67
|
+
|
|
68
|
+
Every load uploads one Parquet file to a Unity Catalog volume, reads it from
|
|
69
|
+
there and removes it. Create a managed volume for it once:
|
|
70
|
+
|
|
71
|
+
```sql
|
|
72
|
+
CREATE SCHEMA IF NOT EXISTS analytics.staging;
|
|
73
|
+
CREATE VOLUME IF NOT EXISTS analytics.staging.loads COMMENT 'interloper load staging';
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
and name it on the destination as `analytics.staging.loads`. Files go under
|
|
77
|
+
`/Volumes/analytics/staging/loads/interloper/`, one per write. A file whose
|
|
78
|
+
removal fails (the load's outcome stands, with a warning) is left there and
|
|
79
|
+
can be deleted by hand.
|
|
80
|
+
|
|
81
|
+
## Usage
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
import interloper as il
|
|
85
|
+
from interloper_databricks import DatabricksConnection, DatabricksDestination
|
|
86
|
+
|
|
87
|
+
destination = DatabricksDestination(
|
|
88
|
+
connection=DatabricksConnection(
|
|
89
|
+
host="https://dbc-a1b2345c-d6e7.cloud.databricks.com",
|
|
90
|
+
client_id="...",
|
|
91
|
+
client_secret="...",
|
|
92
|
+
),
|
|
93
|
+
warehouse="/sql/1.0/warehouses/a1b234c567d8e9fa",
|
|
94
|
+
catalog="analytics",
|
|
95
|
+
staging_volume="analytics.staging.loads",
|
|
96
|
+
default_dataset="raw",
|
|
97
|
+
)
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
`warehouse` is the warehouse's HTTP path (its **Connection details** tab). The
|
|
101
|
+
credentials also load from the environment (`DATABRICKS_HOST`,
|
|
102
|
+
`DATABRICKS_CLIENT_ID`, `DATABRICKS_CLIENT_SECRET`, or
|
|
103
|
+
`DATABRICKS_ACCESS_TOKEN` for a personal access token; note that Databricks'
|
|
104
|
+
own tools name the token `DATABRICKS_TOKEN`).
|
|
105
|
+
|
|
106
|
+
In a deployed instance you configure this through the UI instead: add a
|
|
107
|
+
Databricks connection, then a Databricks destination, picking the warehouse
|
|
108
|
+
and the catalog from the lists the connection can see.
|
|
109
|
+
|
|
110
|
+
## Datasets are schemas
|
|
111
|
+
|
|
112
|
+
An asset's `dataset` is the schema its table lives in, inside the
|
|
113
|
+
destination's `catalog`. An asset without a dataset falls back to
|
|
114
|
+
`default_dataset`; with neither, the write fails with a `ConfigError`. A
|
|
115
|
+
missing schema is created on the first write, and a missing table is created
|
|
116
|
+
as a Delta table typed from the asset's schema (or one inferred from the
|
|
117
|
+
data), with field descriptions as column comments and the asset's description
|
|
118
|
+
as the table comment:
|
|
119
|
+
|
|
120
|
+
| Field type | Databricks type |
|
|
121
|
+
|------------|-----------------|
|
|
122
|
+
| `bool` | `BOOLEAN` |
|
|
123
|
+
| `int` | `BIGINT` |
|
|
124
|
+
| `float` | `DOUBLE` |
|
|
125
|
+
| `Decimal` | `DECIMAL(38,9)` |
|
|
126
|
+
| `datetime` | `TIMESTAMP` |
|
|
127
|
+
| `date` | `DATE` |
|
|
128
|
+
| `bytes` | `BINARY` |
|
|
129
|
+
| `str`, `Any` | `STRING` |
|
|
130
|
+
| nested models, lists, dicts | `VARIANT` |
|
|
131
|
+
|
|
132
|
+
A `datetime` is a `TIMESTAMP`, an absolute instant; a naive datetime is taken
|
|
133
|
+
as UTC. `VARIANT` columns are queried with the path operator
|
|
134
|
+
(`SELECT payload:campaign.id FROM ...`); creating a table with one enables
|
|
135
|
+
Delta's `variantType` feature, which readers need Databricks Runtime 15.4 or a
|
|
136
|
+
recent Delta client for.
|
|
137
|
+
|
|
138
|
+
Identifiers are quoted with backticks. Unity Catalog stores schema and table
|
|
139
|
+
names in lower case and keeps column names as written; queries match either
|
|
140
|
+
case-insensitively.
|
|
141
|
+
|
|
142
|
+
An existing table is never altered: a column the data carries but the schema
|
|
143
|
+
does not is dropped with a warning.
|
|
144
|
+
|
|
145
|
+
A partitioned asset's table is clustered on its partition column with liquid
|
|
146
|
+
clustering (`CLUSTER BY`), which is what Databricks recommends over
|
|
147
|
+
partitioning for tables under 1 TB, when that column is one liquid clustering
|
|
148
|
+
accepts as a key (a date, timestamp, string or number among the first 32
|
|
149
|
+
columns). Tables are never `PARTITIONED BY`.
|
|
150
|
+
|
|
151
|
+
## Partitions
|
|
152
|
+
|
|
153
|
+
Each write is one statement, so a partition is replaced atomically:
|
|
154
|
+
|
|
155
|
+
- a time partition:
|
|
156
|
+
``INSERT INTO ... BY NAME REPLACE WHERE `day` >= DATE'2024-01-01' AND `day` < DATE'2024-01-02' SELECT ...``
|
|
157
|
+
(half-open bounds, so a monthly partition whose rows hold daily dates is
|
|
158
|
+
replaced whole)
|
|
159
|
+
- any other partition: ``REPLACE WHERE `region` = 'eu'``
|
|
160
|
+
- a window: one predicate covering every partition in it, a single range from
|
|
161
|
+
its first start to its last end when the partitions are contiguous
|
|
162
|
+
- an unpartitioned asset: `INSERT OVERWRITE ... BY NAME SELECT ...`
|
|
163
|
+
|
|
164
|
+
The `SELECT` reads the staged file with `read_files(..., format => 'parquet')`,
|
|
165
|
+
casting each column to the table's type. `REPLACE WHERE` deletes the matching
|
|
166
|
+
rows and inserts the new ones in a single Delta commit, so a failed load
|
|
167
|
+
leaves the partition's old rows in place.
|
|
168
|
+
|
|
169
|
+
`REPLACE WHERE` is strict: every row written must match the predicate, or the
|
|
170
|
+
statement fails with `DELTA_REPLACE_WHERE_MISMATCH` and writes nothing. A row
|
|
171
|
+
whose partition column is null, or falls outside the partition, fails the
|
|
172
|
+
write instead of landing where the next replace would not remove it.
|
|
173
|
+
|
|
174
|
+
Reads come back as a DataFrame built from the result's Arrow batches.
|
|
175
|
+
|
|
176
|
+
## Notes
|
|
177
|
+
|
|
178
|
+
Each destination opens its own SQL session through the connection, on its own
|
|
179
|
+
warehouse and catalog, so destinations sharing a connection never affect each
|
|
180
|
+
other. That session is shared by every asset the destination writes, and the
|
|
181
|
+
connector's sessions are not thread-safe, so the destination runs one
|
|
182
|
+
statement at a time and holds a write from its upload to its cleanup.
|
|
183
|
+
|
|
184
|
+
The SQL warehouse does the loading work: it reads the staged file, casts it
|
|
185
|
+
and rewrites the replaced rows, so its size bounds load throughput. Since a
|
|
186
|
+
destination runs one statement at a time, size the warehouse for the largest
|
|
187
|
+
single write (a partition, or a whole window) rather than for concurrency;
|
|
188
|
+
several destinations, or several runs, can share a warehouse that scales out.
|
|
189
|
+
A stopped warehouse adds its start-up time to the first write after it
|
|
190
|
+
auto-stops, which serverless warehouses keep short.
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
# interloper-databricks
|
|
2
|
+
|
|
3
|
+
Databricks integration for interloper: a `DatabricksDestination` that stores
|
|
4
|
+
assets as Delta tables in Unity Catalog through a SQL warehouse, and the
|
|
5
|
+
`DatabricksConnection` that holds the workspace credentials.
|
|
6
|
+
|
|
7
|
+
The destination is a `DatabaseDestination`: the rows a partition covers come
|
|
8
|
+
from core, exactly as for BigQuery. What differs is how they are replaced (see
|
|
9
|
+
[Partitions](#partitions)).
|
|
10
|
+
|
|
11
|
+
## Setup
|
|
12
|
+
|
|
13
|
+
### Credentials
|
|
14
|
+
|
|
15
|
+
A service principal with an OAuth secret is the recommended credential; a
|
|
16
|
+
personal access token works too. The connection takes exactly one of them.
|
|
17
|
+
|
|
18
|
+
1. In the account console (or the workspace admin settings), create a service
|
|
19
|
+
principal and add it to the workspace.
|
|
20
|
+
2. Under its **Secrets** tab, click **Generate secret**. Note the client ID
|
|
21
|
+
and the secret; the secret is shown once. The connection asks for the
|
|
22
|
+
`all-apis` scope, so leave the secret unscoped (or select all APIs): a
|
|
23
|
+
secret restricted to some scopes cannot issue that token.
|
|
24
|
+
3. Give the service principal the **Can use** permission on the SQL warehouse
|
|
25
|
+
the destination loads through.
|
|
26
|
+
|
|
27
|
+
The connection exchanges the client ID and secret for a one-hour access token
|
|
28
|
+
at the workspace's `/oidc/v1/token` endpoint (client credentials grant), and
|
|
29
|
+
replaces the token before it expires. The connection check calls
|
|
30
|
+
`/api/2.0/preview/scim/v2/Me`, and the pickers list SQL warehouses and Unity
|
|
31
|
+
Catalog catalogs over the REST API, all with that token.
|
|
32
|
+
|
|
33
|
+
### Grants
|
|
34
|
+
|
|
35
|
+
The principal needs, in Unity Catalog:
|
|
36
|
+
|
|
37
|
+
- `USE CATALOG` on the destination's catalog
|
|
38
|
+
- `CREATE SCHEMA` on the catalog, or `USE SCHEMA` on every schema the assets
|
|
39
|
+
write to if you create them yourself
|
|
40
|
+
- `CREATE TABLE` on those schemas
|
|
41
|
+
- `MODIFY` and `SELECT` on the tables (a table the principal creates is its
|
|
42
|
+
own, so it has both)
|
|
43
|
+
- `USE SCHEMA`, `READ VOLUME` and `WRITE VOLUME` on the staging volume and its
|
|
44
|
+
schema
|
|
45
|
+
|
|
46
|
+
A sketch, with the service principal's application ID as the grantee:
|
|
47
|
+
|
|
48
|
+
```sql
|
|
49
|
+
GRANT USE CATALOG, CREATE SCHEMA ON CATALOG analytics TO `<application-id>`;
|
|
50
|
+
GRANT USE SCHEMA ON SCHEMA analytics.staging TO `<application-id>`;
|
|
51
|
+
GRANT READ VOLUME, WRITE VOLUME ON VOLUME analytics.staging.loads TO `<application-id>`;
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Staging volume
|
|
55
|
+
|
|
56
|
+
Every load uploads one Parquet file to a Unity Catalog volume, reads it from
|
|
57
|
+
there and removes it. Create a managed volume for it once:
|
|
58
|
+
|
|
59
|
+
```sql
|
|
60
|
+
CREATE SCHEMA IF NOT EXISTS analytics.staging;
|
|
61
|
+
CREATE VOLUME IF NOT EXISTS analytics.staging.loads COMMENT 'interloper load staging';
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
and name it on the destination as `analytics.staging.loads`. Files go under
|
|
65
|
+
`/Volumes/analytics/staging/loads/interloper/`, one per write. A file whose
|
|
66
|
+
removal fails (the load's outcome stands, with a warning) is left there and
|
|
67
|
+
can be deleted by hand.
|
|
68
|
+
|
|
69
|
+
## Usage
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
import interloper as il
|
|
73
|
+
from interloper_databricks import DatabricksConnection, DatabricksDestination
|
|
74
|
+
|
|
75
|
+
destination = DatabricksDestination(
|
|
76
|
+
connection=DatabricksConnection(
|
|
77
|
+
host="https://dbc-a1b2345c-d6e7.cloud.databricks.com",
|
|
78
|
+
client_id="...",
|
|
79
|
+
client_secret="...",
|
|
80
|
+
),
|
|
81
|
+
warehouse="/sql/1.0/warehouses/a1b234c567d8e9fa",
|
|
82
|
+
catalog="analytics",
|
|
83
|
+
staging_volume="analytics.staging.loads",
|
|
84
|
+
default_dataset="raw",
|
|
85
|
+
)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`warehouse` is the warehouse's HTTP path (its **Connection details** tab). The
|
|
89
|
+
credentials also load from the environment (`DATABRICKS_HOST`,
|
|
90
|
+
`DATABRICKS_CLIENT_ID`, `DATABRICKS_CLIENT_SECRET`, or
|
|
91
|
+
`DATABRICKS_ACCESS_TOKEN` for a personal access token; note that Databricks'
|
|
92
|
+
own tools name the token `DATABRICKS_TOKEN`).
|
|
93
|
+
|
|
94
|
+
In a deployed instance you configure this through the UI instead: add a
|
|
95
|
+
Databricks connection, then a Databricks destination, picking the warehouse
|
|
96
|
+
and the catalog from the lists the connection can see.
|
|
97
|
+
|
|
98
|
+
## Datasets are schemas
|
|
99
|
+
|
|
100
|
+
An asset's `dataset` is the schema its table lives in, inside the
|
|
101
|
+
destination's `catalog`. An asset without a dataset falls back to
|
|
102
|
+
`default_dataset`; with neither, the write fails with a `ConfigError`. A
|
|
103
|
+
missing schema is created on the first write, and a missing table is created
|
|
104
|
+
as a Delta table typed from the asset's schema (or one inferred from the
|
|
105
|
+
data), with field descriptions as column comments and the asset's description
|
|
106
|
+
as the table comment:
|
|
107
|
+
|
|
108
|
+
| Field type | Databricks type |
|
|
109
|
+
|------------|-----------------|
|
|
110
|
+
| `bool` | `BOOLEAN` |
|
|
111
|
+
| `int` | `BIGINT` |
|
|
112
|
+
| `float` | `DOUBLE` |
|
|
113
|
+
| `Decimal` | `DECIMAL(38,9)` |
|
|
114
|
+
| `datetime` | `TIMESTAMP` |
|
|
115
|
+
| `date` | `DATE` |
|
|
116
|
+
| `bytes` | `BINARY` |
|
|
117
|
+
| `str`, `Any` | `STRING` |
|
|
118
|
+
| nested models, lists, dicts | `VARIANT` |
|
|
119
|
+
|
|
120
|
+
A `datetime` is a `TIMESTAMP`, an absolute instant; a naive datetime is taken
|
|
121
|
+
as UTC. `VARIANT` columns are queried with the path operator
|
|
122
|
+
(`SELECT payload:campaign.id FROM ...`); creating a table with one enables
|
|
123
|
+
Delta's `variantType` feature, which readers need Databricks Runtime 15.4 or a
|
|
124
|
+
recent Delta client for.
|
|
125
|
+
|
|
126
|
+
Identifiers are quoted with backticks. Unity Catalog stores schema and table
|
|
127
|
+
names in lower case and keeps column names as written; queries match either
|
|
128
|
+
case-insensitively.
|
|
129
|
+
|
|
130
|
+
An existing table is never altered: a column the data carries but the schema
|
|
131
|
+
does not is dropped with a warning.
|
|
132
|
+
|
|
133
|
+
A partitioned asset's table is clustered on its partition column with liquid
|
|
134
|
+
clustering (`CLUSTER BY`), which is what Databricks recommends over
|
|
135
|
+
partitioning for tables under 1 TB, when that column is one liquid clustering
|
|
136
|
+
accepts as a key (a date, timestamp, string or number among the first 32
|
|
137
|
+
columns). Tables are never `PARTITIONED BY`.
|
|
138
|
+
|
|
139
|
+
## Partitions
|
|
140
|
+
|
|
141
|
+
Each write is one statement, so a partition is replaced atomically:
|
|
142
|
+
|
|
143
|
+
- a time partition:
|
|
144
|
+
``INSERT INTO ... BY NAME REPLACE WHERE `day` >= DATE'2024-01-01' AND `day` < DATE'2024-01-02' SELECT ...``
|
|
145
|
+
(half-open bounds, so a monthly partition whose rows hold daily dates is
|
|
146
|
+
replaced whole)
|
|
147
|
+
- any other partition: ``REPLACE WHERE `region` = 'eu'``
|
|
148
|
+
- a window: one predicate covering every partition in it, a single range from
|
|
149
|
+
its first start to its last end when the partitions are contiguous
|
|
150
|
+
- an unpartitioned asset: `INSERT OVERWRITE ... BY NAME SELECT ...`
|
|
151
|
+
|
|
152
|
+
The `SELECT` reads the staged file with `read_files(..., format => 'parquet')`,
|
|
153
|
+
casting each column to the table's type. `REPLACE WHERE` deletes the matching
|
|
154
|
+
rows and inserts the new ones in a single Delta commit, so a failed load
|
|
155
|
+
leaves the partition's old rows in place.
|
|
156
|
+
|
|
157
|
+
`REPLACE WHERE` is strict: every row written must match the predicate, or the
|
|
158
|
+
statement fails with `DELTA_REPLACE_WHERE_MISMATCH` and writes nothing. A row
|
|
159
|
+
whose partition column is null, or falls outside the partition, fails the
|
|
160
|
+
write instead of landing where the next replace would not remove it.
|
|
161
|
+
|
|
162
|
+
Reads come back as a DataFrame built from the result's Arrow batches.
|
|
163
|
+
|
|
164
|
+
## Notes
|
|
165
|
+
|
|
166
|
+
Each destination opens its own SQL session through the connection, on its own
|
|
167
|
+
warehouse and catalog, so destinations sharing a connection never affect each
|
|
168
|
+
other. That session is shared by every asset the destination writes, and the
|
|
169
|
+
connector's sessions are not thread-safe, so the destination runs one
|
|
170
|
+
statement at a time and holds a write from its upload to its cleanup.
|
|
171
|
+
|
|
172
|
+
The SQL warehouse does the loading work: it reads the staged file, casts it
|
|
173
|
+
and rewrites the replaced rows, so its size bounds load throughput. Since a
|
|
174
|
+
destination runs one statement at a time, size the warehouse for the largest
|
|
175
|
+
single write (a partition, or a whole window) rather than for concurrency;
|
|
176
|
+
several destinations, or several runs, can share a warehouse that scales out.
|
|
177
|
+
A stopped warehouse adds its start-up time to the first write after it
|
|
178
|
+
auto-stops, which serverless warehouses keep short.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "interloper-databricks"
|
|
3
|
+
version = "0.96.0"
|
|
4
|
+
description = "Interloper Databricks integration: connection and destination"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"interloper-core",
|
|
9
|
+
"interloper-pandas",
|
|
10
|
+
"databricks-sql-connector[pyarrow]>=4.1.2",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[[project.authors]]
|
|
14
|
+
name = "Guillaume Onfroy"
|
|
15
|
+
email = "guillaume@digitlcloud.com"
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
databricks = "interloper_databricks"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources.interloper-core]
|
|
25
|
+
workspace = true
|
|
26
|
+
|
|
27
|
+
[tool.uv.sources.interloper-pandas]
|
|
28
|
+
workspace = true
|
|
29
|
+
|
|
30
|
+
[tool.ruff]
|
|
31
|
+
line-length = 120
|
|
32
|
+
|
|
33
|
+
[tool.ruff.lint]
|
|
34
|
+
preview = true
|
|
35
|
+
extend-select = [
|
|
36
|
+
"E",
|
|
37
|
+
"I",
|
|
38
|
+
"UP",
|
|
39
|
+
"ANN001",
|
|
40
|
+
"ANN201",
|
|
41
|
+
"ANN202",
|
|
42
|
+
"DOC",
|
|
43
|
+
"D",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint.pydocstyle]
|
|
47
|
+
convention = "google"
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint.per-file-ignores]
|
|
50
|
+
"__init__.py" = [
|
|
51
|
+
"F401",
|
|
52
|
+
"F403",
|
|
53
|
+
]
|
|
54
|
+
"tests/**" = [
|
|
55
|
+
"ANN",
|
|
56
|
+
"F811",
|
|
57
|
+
"D101",
|
|
58
|
+
"D102",
|
|
59
|
+
"D103",
|
|
60
|
+
"D104",
|
|
61
|
+
"RUF069",
|
|
62
|
+
"PLW0108",
|
|
63
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# ###############
|
|
2
|
+
# PROJECT / UV
|
|
3
|
+
# ###############
|
|
4
|
+
[project]
|
|
5
|
+
name = "interloper-databricks"
|
|
6
|
+
version = "0.96.0"
|
|
7
|
+
description = "Interloper Databricks integration: connection and destination"
|
|
8
|
+
readme = "README.md"
|
|
9
|
+
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"interloper-core",
|
|
13
|
+
"interloper-pandas",
|
|
14
|
+
"databricks-sql-connector[pyarrow]>=4.1.2",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
databricks = "interloper_databricks"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources]
|
|
25
|
+
interloper-core = { workspace = true }
|
|
26
|
+
interloper-pandas = { workspace = true }
|
|
27
|
+
|
|
28
|
+
# ###############
|
|
29
|
+
# RUFF
|
|
30
|
+
# ###############
|
|
31
|
+
[tool.ruff]
|
|
32
|
+
line-length = 120
|
|
33
|
+
|
|
34
|
+
[tool.ruff.lint]
|
|
35
|
+
preview = true
|
|
36
|
+
extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
|
|
37
|
+
|
|
38
|
+
[tool.ruff.lint.pydocstyle]
|
|
39
|
+
convention = "google"
|
|
40
|
+
|
|
41
|
+
[tool.ruff.lint.per-file-ignores]
|
|
42
|
+
"__init__.py" = ["F401", "F403"]
|
|
43
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Interloper Databricks integration: connection and destination."""
|
|
2
|
+
|
|
3
|
+
from interloper_databricks.connection import DatabricksConnection
|
|
4
|
+
from interloper_databricks.destination import DatabricksDestination
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"DatabricksConnection",
|
|
8
|
+
"DatabricksDestination",
|
|
9
|
+
]
|