interloper-azure 0.96.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interloper_azure-0.96.0/PKG-INFO +194 -0
- interloper_azure-0.96.0/README.md +182 -0
- interloper_azure-0.96.0/pyproject.toml +60 -0
- interloper_azure-0.96.0/pyproject.toml.orig +42 -0
- interloper_azure-0.96.0/src/interloper_azure/__init__.py +9 -0
- interloper_azure-0.96.0/src/interloper_azure/connection.py +161 -0
- interloper_azure-0.96.0/src/interloper_azure/fabric/__init__.py +7 -0
- interloper_azure-0.96.0/src/interloper_azure/fabric/destination.py +554 -0
- interloper_azure-0.96.0/src/interloper_azure/fabric/types.py +46 -0
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: interloper-azure
|
|
3
|
+
Version: 0.96.0
|
|
4
|
+
Summary: Interloper Microsoft Azure integration: Entra connection and Microsoft Fabric Warehouse destination
|
|
5
|
+
Author: Guillaume Onfroy
|
|
6
|
+
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
+
Requires-Dist: interloper-core
|
|
8
|
+
Requires-Dist: azure-identity>=1.19
|
|
9
|
+
Requires-Dist: mssql-python>=1.15
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# interloper-azure
|
|
14
|
+
|
|
15
|
+
Microsoft Azure integration for interloper: the `AzureConnection` that holds a
|
|
16
|
+
Microsoft Entra service principal, and a `FabricWarehouseDestination` that
|
|
17
|
+
stores assets as tables in a Microsoft Fabric Warehouse.
|
|
18
|
+
|
|
19
|
+
The destination is a `DatabaseDestination`: it writes Fabric's T-SQL dialect
|
|
20
|
+
and nothing else. Partition replacement, windows and reads by partition come
|
|
21
|
+
from core, exactly as for BigQuery or Snowflake.
|
|
22
|
+
|
|
23
|
+
It targets a Fabric **Warehouse**. A Lakehouse's SQL analytics endpoint speaks
|
|
24
|
+
the same protocol but is read-only: creating tables and inserting or deleting
|
|
25
|
+
rows is only supported in a Warehouse.
|
|
26
|
+
|
|
27
|
+
## Setup
|
|
28
|
+
|
|
29
|
+
### 1. Create a service principal
|
|
30
|
+
|
|
31
|
+
In the [Microsoft Entra admin center](https://entra.microsoft.com), under
|
|
32
|
+
*App registrations*, register an application (no redirect URI needed), then on
|
|
33
|
+
its *Certificates & secrets* page create a client secret. Note three values:
|
|
34
|
+
|
|
35
|
+
- the *Directory (tenant) ID*
|
|
36
|
+
- the *Application (client) ID*
|
|
37
|
+
- the secret's *Value* (shown once, not its *Secret ID*)
|
|
38
|
+
|
|
39
|
+
### 2. Let service principals use Fabric
|
|
40
|
+
|
|
41
|
+
A Fabric administrator must enable **Service principals can use Fabric APIs**
|
|
42
|
+
in the Fabric admin portal (*Tenant settings*, *Developer settings*), either for
|
|
43
|
+
the whole organisation or for a security group the principal belongs to. The
|
|
44
|
+
same setting governs both the REST API (used by the connection check and the
|
|
45
|
+
workspace picker) and SQL connections to a warehouse.
|
|
46
|
+
|
|
47
|
+
### 3. Give the principal access to the workspace
|
|
48
|
+
|
|
49
|
+
In the workspace, open *Manage access* and add the app with the
|
|
50
|
+
**Contributor** role. Contributor, like Admin and Member, grants `CONTROL` on
|
|
51
|
+
every warehouse of the workspace, which covers creating schemas and tables and
|
|
52
|
+
writing rows; `CREATE SCHEMA` in particular requires one of those three roles.
|
|
53
|
+
Viewer only reads, and sharing a single warehouse with no extra permissions
|
|
54
|
+
only lets the principal connect.
|
|
55
|
+
|
|
56
|
+
### 4. Find the SQL connection string
|
|
57
|
+
|
|
58
|
+
Open the warehouse's *Settings* and its *SQL connection string* page; the
|
|
59
|
+
string looks like `xxxxxxxx-xxxx.datawarehouse.fabric.microsoft.com`. The string
|
|
60
|
+
belongs to the workspace: every warehouse in it shares the same one, and the
|
|
61
|
+
warehouse's name selects the database. In the UI you do not need to copy it:
|
|
62
|
+
the destination's *Workspace* picker lists the workspaces the principal can see
|
|
63
|
+
that hold a warehouse, and stores that workspace's connection string.
|
|
64
|
+
|
|
65
|
+
## Usage
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
from interloper_azure import AzureConnection, FabricWarehouseDestination
|
|
69
|
+
|
|
70
|
+
destination = FabricWarehouseDestination(
|
|
71
|
+
connection=AzureConnection(tenant_id="...", client_id="...", client_secret="..."),
|
|
72
|
+
server="xxxxxxxx-xxxx.datawarehouse.fabric.microsoft.com",
|
|
73
|
+
warehouse="Analytics",
|
|
74
|
+
default_dataset="raw",
|
|
75
|
+
)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
The credentials also load from the environment under the standard
|
|
79
|
+
azure-identity names (`AZURE_TENANT_ID`, `AZURE_CLIENT_ID`,
|
|
80
|
+
`AZURE_CLIENT_SECRET`), so `AzureConnection()` works with no arguments.
|
|
81
|
+
|
|
82
|
+
In a deployed instance you configure this through the UI: add a Microsoft
|
|
83
|
+
Azure connection, then a Microsoft Fabric Warehouse destination, pick the
|
|
84
|
+
workspace and type the warehouse name.
|
|
85
|
+
|
|
86
|
+
## The driver and its system libraries
|
|
87
|
+
|
|
88
|
+
Statements go through [`mssql-python`](https://github.com/microsoft/mssql-python),
|
|
89
|
+
Microsoft's DB-API driver, which signs in with an Entra access token obtained
|
|
90
|
+
from the connection's credential (scope `https://database.windows.net/.default`).
|
|
91
|
+
It is pip-installable and bundles Microsoft's ODBC driver, but that driver
|
|
92
|
+
loads a few system libraries on Linux. On Debian or Ubuntu:
|
|
93
|
+
|
|
94
|
+
```sh
|
|
95
|
+
apt-get install -y libltdl7 libkrb5-3 libgssapi-krb5-2
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
The published scheduler and core images, the ones that run assets, ship these
|
|
99
|
+
libraries. The api and mcp images only describe the destination and never open
|
|
100
|
+
a session, so they leave them out.
|
|
101
|
+
|
|
102
|
+
Without them the package imports fine (the catalog, the connection check and
|
|
103
|
+
the workspace picker all work), but opening a warehouse session fails with
|
|
104
|
+
`Failed to load the driver`. `mssql-python` also offers an alternate, Rust-based
|
|
105
|
+
native provider that needs none of these libraries: install `mssql-python-rs`
|
|
106
|
+
and set `MSSQL_PYTHON_NATIVE_PROVIDER=mssql-odbc`. Microsoft labels that
|
|
107
|
+
provider alpha.
|
|
108
|
+
|
|
109
|
+
## Datasets are schemas
|
|
110
|
+
|
|
111
|
+
An asset's `dataset` is the warehouse schema its table lives in. An asset
|
|
112
|
+
without a dataset falls back to `default_dataset`, then to `dbo`. A missing
|
|
113
|
+
schema is created on the first write, and a missing table is created typed
|
|
114
|
+
from the asset's schema (or one inferred from the data):
|
|
115
|
+
|
|
116
|
+
| Field type | Fabric Warehouse type |
|
|
117
|
+
|------------|-----------------------|
|
|
118
|
+
| `bool` | `bit` |
|
|
119
|
+
| `int` | `bigint` |
|
|
120
|
+
| `float` | `float` |
|
|
121
|
+
| `Decimal` | `decimal(38,9)` |
|
|
122
|
+
| `datetime` | `datetime2(6)` |
|
|
123
|
+
| `date` | `date` |
|
|
124
|
+
| `bytes` | `varbinary(max)` |
|
|
125
|
+
| `str`, `Any` | `varchar(max)` |
|
|
126
|
+
| nested models, lists, dicts | `varchar(max)` holding JSON |
|
|
127
|
+
|
|
128
|
+
An existing table is never altered: a column the data carries but the table
|
|
129
|
+
does not is dropped with a warning. Reads return records; nested values come
|
|
130
|
+
back as the JSON text they are stored as. Identifiers are bracketed
|
|
131
|
+
(`[marts].[ads_stats]`), and the warehouse's default collation is
|
|
132
|
+
case-sensitive, so a table is queried with the exact case the asset gives it.
|
|
133
|
+
|
|
134
|
+
## How Fabric's T-SQL shaped the design
|
|
135
|
+
|
|
136
|
+
Fabric Warehouse speaks T-SQL over the SQL Server protocol, but it is not SQL
|
|
137
|
+
Server. The differences that matter here:
|
|
138
|
+
|
|
139
|
+
- **Types.** Tables accept a subset of SQL Server's types: `datetime2` and
|
|
140
|
+
`time` keep at most six fractional digits, and there is no `datetime`,
|
|
141
|
+
`datetimeoffset`, `nvarchar`, `nchar`, `tinyint`, `money` or `json`. Hence
|
|
142
|
+
`varchar` (UTF-8) for text and JSON, and `datetime2(6)` for timestamps;
|
|
143
|
+
timezone-aware values are written as UTC. `varchar(max)` and
|
|
144
|
+
`varbinary(max)` exist but hold at most 16 MB per value. Microsoft
|
|
145
|
+
recommends the shortest fitting `varchar(n)` for query performance;
|
|
146
|
+
`varchar(max)` is used because the destination cannot know a column's
|
|
147
|
+
longest value up front.
|
|
148
|
+
- **Schemas.** There is no `CREATE SCHEMA IF NOT EXISTS`, and `CREATE SCHEMA`
|
|
149
|
+
must be alone in its batch, so the destination checks `sys.schemas` first
|
|
150
|
+
and creates the schema in a statement of its own. Table existence is read
|
|
151
|
+
from `sys.tables` and `sys.columns`; the warehouse refuses a query mixing
|
|
152
|
+
system and user tables, so the catalog check never touches the data.
|
|
153
|
+
- **Transactions.** Explicit `BEGIN TRANSACTION ... COMMIT TRANSACTION` is
|
|
154
|
+
supported, always under snapshot isolation, and DDL is allowed inside one.
|
|
155
|
+
DDL in a transaction holds locks on the catalog views until commit, though,
|
|
156
|
+
blocking every other write's existence checks, so a new schema and table
|
|
157
|
+
are created and committed before the write's transaction opens; inside it
|
|
158
|
+
run only the `DELETE` and the `INSERT`s.
|
|
159
|
+
- **Whole-table replaces delete rather than truncate.** `TRUNCATE TABLE`
|
|
160
|
+
takes a schema-modification lock that blocks readers until commit; a
|
|
161
|
+
`DELETE` does not.
|
|
162
|
+
- **Write-write conflicts are per table.** Two transactions deleting from the
|
|
163
|
+
same table conflict even when they touch different rows: the first to commit
|
|
164
|
+
wins and the other fails with error 24556 or 24706. Writes to different
|
|
165
|
+
tables never conflict. Concurrent writes of different partitions of one
|
|
166
|
+
asset (overlapping backfills, for instance) should be avoided, or covered by
|
|
167
|
+
a retry policy.
|
|
168
|
+
|
|
169
|
+
## Partitions
|
|
170
|
+
|
|
171
|
+
A partitioned write replaces the partition's rows: it deletes them, then
|
|
172
|
+
inserts the data, inside one transaction, rolled back on failure. A time
|
|
173
|
+
partition deletes by half-open bounds (`[day] >= ? AND [day] < ?`), so a
|
|
174
|
+
monthly partition whose rows hold daily dates is replaced whole; any other
|
|
175
|
+
partition deletes by equality on its id. A window deletes each partition it
|
|
176
|
+
covers and inserts the whole batch once. An unpartitioned asset deletes every
|
|
177
|
+
row and reloads the table. Readers keep seeing the previous rows until the
|
|
178
|
+
commit.
|
|
179
|
+
|
|
180
|
+
## The load path and its limits
|
|
181
|
+
|
|
182
|
+
Rows are written with parameterised multi-row `INSERT ... VALUES` statements.
|
|
183
|
+
Each statement carries as many rows as fit under SQL Server's 2100-parameter
|
|
184
|
+
cap (2099 parameters, one per value) and 1000 rows, whichever is smaller: a
|
|
185
|
+
5-column table loads 419 rows per statement, a 40-column table 52. Every
|
|
186
|
+
statement is a round trip and writes new Parquet files under the table, which
|
|
187
|
+
the warehouse compacts in the background.
|
|
188
|
+
|
|
189
|
+
This needs no infrastructure beyond the warehouse and suits incremental loads:
|
|
190
|
+
thousands to low hundreds of thousands of rows per write. Microsoft recommends
|
|
191
|
+
against frequent small inserts, and for bulk volumes (millions of rows per
|
|
192
|
+
write, or a full reload of a large table) the documented high-throughput path
|
|
193
|
+
is `COPY INTO` from files staged in OneLake or Azure Data Lake Storage, which
|
|
194
|
+
this destination does not implement.
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
# interloper-azure
|
|
2
|
+
|
|
3
|
+
Microsoft Azure integration for interloper: the `AzureConnection` that holds a
|
|
4
|
+
Microsoft Entra service principal, and a `FabricWarehouseDestination` that
|
|
5
|
+
stores assets as tables in a Microsoft Fabric Warehouse.
|
|
6
|
+
|
|
7
|
+
The destination is a `DatabaseDestination`: it writes Fabric's T-SQL dialect
|
|
8
|
+
and nothing else. Partition replacement, windows and reads by partition come
|
|
9
|
+
from core, exactly as for BigQuery or Snowflake.
|
|
10
|
+
|
|
11
|
+
It targets a Fabric **Warehouse**. A Lakehouse's SQL analytics endpoint speaks
|
|
12
|
+
the same protocol but is read-only: creating tables and inserting or deleting
|
|
13
|
+
rows is only supported in a Warehouse.
|
|
14
|
+
|
|
15
|
+
## Setup
|
|
16
|
+
|
|
17
|
+
### 1. Create a service principal
|
|
18
|
+
|
|
19
|
+
In the [Microsoft Entra admin center](https://entra.microsoft.com), under
|
|
20
|
+
*App registrations*, register an application (no redirect URI needed), then on
|
|
21
|
+
its *Certificates & secrets* page create a client secret. Note three values:
|
|
22
|
+
|
|
23
|
+
- the *Directory (tenant) ID*
|
|
24
|
+
- the *Application (client) ID*
|
|
25
|
+
- the secret's *Value* (shown once, not its *Secret ID*)
|
|
26
|
+
|
|
27
|
+
### 2. Let service principals use Fabric
|
|
28
|
+
|
|
29
|
+
A Fabric administrator must enable **Service principals can use Fabric APIs**
|
|
30
|
+
in the Fabric admin portal (*Tenant settings*, *Developer settings*), either for
|
|
31
|
+
the whole organisation or for a security group the principal belongs to. The
|
|
32
|
+
same setting governs both the REST API (used by the connection check and the
|
|
33
|
+
workspace picker) and SQL connections to a warehouse.
|
|
34
|
+
|
|
35
|
+
### 3. Give the principal access to the workspace
|
|
36
|
+
|
|
37
|
+
In the workspace, open *Manage access* and add the app with the
|
|
38
|
+
**Contributor** role. Contributor, like Admin and Member, grants `CONTROL` on
|
|
39
|
+
every warehouse of the workspace, which covers creating schemas and tables and
|
|
40
|
+
writing rows; `CREATE SCHEMA` in particular requires one of those three roles.
|
|
41
|
+
Viewer only reads, and sharing a single warehouse with no extra permissions
|
|
42
|
+
only lets the principal connect.
|
|
43
|
+
|
|
44
|
+
### 4. Find the SQL connection string
|
|
45
|
+
|
|
46
|
+
Open the warehouse's *Settings* and its *SQL connection string* page; the
|
|
47
|
+
string looks like `xxxxxxxx-xxxx.datawarehouse.fabric.microsoft.com`. The string
|
|
48
|
+
belongs to the workspace: every warehouse in it shares the same one, and the
|
|
49
|
+
warehouse's name selects the database. In the UI you do not need to copy it:
|
|
50
|
+
the destination's *Workspace* picker lists the workspaces the principal can see
|
|
51
|
+
that hold a warehouse, and stores that workspace's connection string.
|
|
52
|
+
|
|
53
|
+
## Usage
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from interloper_azure import AzureConnection, FabricWarehouseDestination
|
|
57
|
+
|
|
58
|
+
destination = FabricWarehouseDestination(
|
|
59
|
+
connection=AzureConnection(tenant_id="...", client_id="...", client_secret="..."),
|
|
60
|
+
server="xxxxxxxx-xxxx.datawarehouse.fabric.microsoft.com",
|
|
61
|
+
warehouse="Analytics",
|
|
62
|
+
default_dataset="raw",
|
|
63
|
+
)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
The credentials also load from the environment under the standard
|
|
67
|
+
azure-identity names (`AZURE_TENANT_ID`, `AZURE_CLIENT_ID`,
|
|
68
|
+
`AZURE_CLIENT_SECRET`), so `AzureConnection()` works with no arguments.
|
|
69
|
+
|
|
70
|
+
In a deployed instance you configure this through the UI: add a Microsoft
|
|
71
|
+
Azure connection, then a Microsoft Fabric Warehouse destination, pick the
|
|
72
|
+
workspace and type the warehouse name.
|
|
73
|
+
|
|
74
|
+
## The driver and its system libraries
|
|
75
|
+
|
|
76
|
+
Statements go through [`mssql-python`](https://github.com/microsoft/mssql-python),
|
|
77
|
+
Microsoft's DB-API driver, which signs in with an Entra access token obtained
|
|
78
|
+
from the connection's credential (scope `https://database.windows.net/.default`).
|
|
79
|
+
It is pip-installable and bundles Microsoft's ODBC driver, but that driver
|
|
80
|
+
loads a few system libraries on Linux. On Debian or Ubuntu:
|
|
81
|
+
|
|
82
|
+
```sh
|
|
83
|
+
apt-get install -y libltdl7 libkrb5-3 libgssapi-krb5-2
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
The published scheduler and core images, the ones that run assets, ship these
|
|
87
|
+
libraries. The api and mcp images only describe the destination and never open
|
|
88
|
+
a session, so they leave them out.
|
|
89
|
+
|
|
90
|
+
Without them the package imports fine (the catalog, the connection check and
|
|
91
|
+
the workspace picker all work), but opening a warehouse session fails with
|
|
92
|
+
`Failed to load the driver`. `mssql-python` also offers an alternate, Rust-based
|
|
93
|
+
native provider that needs none of these libraries: install `mssql-python-rs`
|
|
94
|
+
and set `MSSQL_PYTHON_NATIVE_PROVIDER=mssql-odbc`. Microsoft labels that
|
|
95
|
+
provider alpha.
|
|
96
|
+
|
|
97
|
+
## Datasets are schemas
|
|
98
|
+
|
|
99
|
+
An asset's `dataset` is the warehouse schema its table lives in. An asset
|
|
100
|
+
without a dataset falls back to `default_dataset`, then to `dbo`. A missing
|
|
101
|
+
schema is created on the first write, and a missing table is created typed
|
|
102
|
+
from the asset's schema (or one inferred from the data):
|
|
103
|
+
|
|
104
|
+
| Field type | Fabric Warehouse type |
|
|
105
|
+
|------------|-----------------------|
|
|
106
|
+
| `bool` | `bit` |
|
|
107
|
+
| `int` | `bigint` |
|
|
108
|
+
| `float` | `float` |
|
|
109
|
+
| `Decimal` | `decimal(38,9)` |
|
|
110
|
+
| `datetime` | `datetime2(6)` |
|
|
111
|
+
| `date` | `date` |
|
|
112
|
+
| `bytes` | `varbinary(max)` |
|
|
113
|
+
| `str`, `Any` | `varchar(max)` |
|
|
114
|
+
| nested models, lists, dicts | `varchar(max)` holding JSON |
|
|
115
|
+
|
|
116
|
+
An existing table is never altered: a column the data carries but the table
|
|
117
|
+
does not is dropped with a warning. Reads return records; nested values come
|
|
118
|
+
back as the JSON text they are stored as. Identifiers are bracketed
|
|
119
|
+
(`[marts].[ads_stats]`), and the warehouse's default collation is
|
|
120
|
+
case-sensitive, so a table is queried with the exact case the asset gives it.
|
|
121
|
+
|
|
122
|
+
## How Fabric's T-SQL shaped the design
|
|
123
|
+
|
|
124
|
+
Fabric Warehouse speaks T-SQL over the SQL Server protocol, but it is not SQL
|
|
125
|
+
Server. The differences that matter here:
|
|
126
|
+
|
|
127
|
+
- **Types.** Tables accept a subset of SQL Server's types: `datetime2` and
|
|
128
|
+
`time` keep at most six fractional digits, and there is no `datetime`,
|
|
129
|
+
`datetimeoffset`, `nvarchar`, `nchar`, `tinyint`, `money` or `json`. Hence
|
|
130
|
+
`varchar` (UTF-8) for text and JSON, and `datetime2(6)` for timestamps;
|
|
131
|
+
timezone-aware values are written as UTC. `varchar(max)` and
|
|
132
|
+
`varbinary(max)` exist but hold at most 16 MB per value. Microsoft
|
|
133
|
+
recommends the shortest fitting `varchar(n)` for query performance;
|
|
134
|
+
`varchar(max)` is used because the destination cannot know a column's
|
|
135
|
+
longest value up front.
|
|
136
|
+
- **Schemas.** There is no `CREATE SCHEMA IF NOT EXISTS`, and `CREATE SCHEMA`
|
|
137
|
+
must be alone in its batch, so the destination checks `sys.schemas` first
|
|
138
|
+
and creates the schema in a statement of its own. Table existence is read
|
|
139
|
+
from `sys.tables` and `sys.columns`; the warehouse refuses a query mixing
|
|
140
|
+
system and user tables, so the catalog check never touches the data.
|
|
141
|
+
- **Transactions.** Explicit `BEGIN TRANSACTION ... COMMIT TRANSACTION` is
|
|
142
|
+
supported, always under snapshot isolation, and DDL is allowed inside one.
|
|
143
|
+
DDL in a transaction holds locks on the catalog views until commit, though,
|
|
144
|
+
blocking every other write's existence checks, so a new schema and table
|
|
145
|
+
are created and committed before the write's transaction opens; inside it
|
|
146
|
+
run only the `DELETE` and the `INSERT`s.
|
|
147
|
+
- **Whole-table replaces delete rather than truncate.** `TRUNCATE TABLE`
|
|
148
|
+
takes a schema-modification lock that blocks readers until commit; a
|
|
149
|
+
`DELETE` does not.
|
|
150
|
+
- **Write-write conflicts are per table.** Two transactions deleting from the
|
|
151
|
+
same table conflict even when they touch different rows: the first to commit
|
|
152
|
+
wins and the other fails with error 24556 or 24706. Writes to different
|
|
153
|
+
tables never conflict. Concurrent writes of different partitions of one
|
|
154
|
+
asset (overlapping backfills, for instance) should be avoided, or covered by
|
|
155
|
+
a retry policy.
|
|
156
|
+
|
|
157
|
+
## Partitions
|
|
158
|
+
|
|
159
|
+
A partitioned write replaces the partition's rows: it deletes them, then
|
|
160
|
+
inserts the data, inside one transaction, rolled back on failure. A time
|
|
161
|
+
partition deletes by half-open bounds (`[day] >= ? AND [day] < ?`), so a
|
|
162
|
+
monthly partition whose rows hold daily dates is replaced whole; any other
|
|
163
|
+
partition deletes by equality on its id. A window deletes each partition it
|
|
164
|
+
covers and inserts the whole batch once. An unpartitioned asset deletes every
|
|
165
|
+
row and reloads the table. Readers keep seeing the previous rows until the
|
|
166
|
+
commit.
|
|
167
|
+
|
|
168
|
+
## The load path and its limits
|
|
169
|
+
|
|
170
|
+
Rows are written with parameterised multi-row `INSERT ... VALUES` statements.
|
|
171
|
+
Each statement carries as many rows as fit under SQL Server's 2100-parameter
|
|
172
|
+
cap (2099 parameters, one per value) and 1000 rows, whichever is smaller: a
|
|
173
|
+
5-column table loads 419 rows per statement, a 40-column table 52. Every
|
|
174
|
+
statement is a round trip and writes new Parquet files under the table, which
|
|
175
|
+
the warehouse compacts in the background.
|
|
176
|
+
|
|
177
|
+
This needs no infrastructure beyond the warehouse and suits incremental loads:
|
|
178
|
+
thousands to low hundreds of thousands of rows per write. Microsoft recommends
|
|
179
|
+
against frequent small inserts, and for bulk volumes (millions of rows per
|
|
180
|
+
write, or a full reload of a large table) the documented high-throughput path
|
|
181
|
+
is `COPY INTO` from files staged in OneLake or Azure Data Lake Storage, which
|
|
182
|
+
this destination does not implement.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "interloper-azure"
|
|
3
|
+
version = "0.96.0"
|
|
4
|
+
description = "Interloper Microsoft Azure integration: Entra connection and Microsoft Fabric Warehouse destination"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"interloper-core",
|
|
9
|
+
"azure-identity>=1.19",
|
|
10
|
+
"mssql-python>=1.15",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[[project.authors]]
|
|
14
|
+
name = "Guillaume Onfroy"
|
|
15
|
+
email = "guillaume@digitlcloud.com"
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
azure = "interloper_azure"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources.interloper-core]
|
|
25
|
+
workspace = true
|
|
26
|
+
|
|
27
|
+
[tool.ruff]
|
|
28
|
+
line-length = 120
|
|
29
|
+
|
|
30
|
+
[tool.ruff.lint]
|
|
31
|
+
preview = true
|
|
32
|
+
extend-select = [
|
|
33
|
+
"E",
|
|
34
|
+
"I",
|
|
35
|
+
"UP",
|
|
36
|
+
"ANN001",
|
|
37
|
+
"ANN201",
|
|
38
|
+
"ANN202",
|
|
39
|
+
"DOC",
|
|
40
|
+
"D",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[tool.ruff.lint.pydocstyle]
|
|
44
|
+
convention = "google"
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint.per-file-ignores]
|
|
47
|
+
"__init__.py" = [
|
|
48
|
+
"F401",
|
|
49
|
+
"F403",
|
|
50
|
+
]
|
|
51
|
+
"tests/**" = [
|
|
52
|
+
"ANN",
|
|
53
|
+
"F811",
|
|
54
|
+
"D101",
|
|
55
|
+
"D102",
|
|
56
|
+
"D103",
|
|
57
|
+
"D104",
|
|
58
|
+
"RUF069",
|
|
59
|
+
"PLW0108",
|
|
60
|
+
]
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# ###############
|
|
2
|
+
# PROJECT / UV
|
|
3
|
+
# ###############
|
|
4
|
+
[project]
|
|
5
|
+
name = "interloper-azure"
|
|
6
|
+
version = "0.96.0"
|
|
7
|
+
description = "Interloper Microsoft Azure integration: Entra connection and Microsoft Fabric Warehouse destination"
|
|
8
|
+
readme = "README.md"
|
|
9
|
+
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"interloper-core",
|
|
13
|
+
"azure-identity>=1.19",
|
|
14
|
+
"mssql-python>=1.15",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
azure = "interloper_azure"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources]
|
|
25
|
+
interloper-core = { workspace = true }
|
|
26
|
+
|
|
27
|
+
# ###############
|
|
28
|
+
# RUFF
|
|
29
|
+
# ###############
|
|
30
|
+
[tool.ruff]
|
|
31
|
+
line-length = 120
|
|
32
|
+
|
|
33
|
+
[tool.ruff.lint]
|
|
34
|
+
preview = true
|
|
35
|
+
extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
|
|
36
|
+
|
|
37
|
+
[tool.ruff.lint.pydocstyle]
|
|
38
|
+
convention = "google"
|
|
39
|
+
|
|
40
|
+
[tool.ruff.lint.per-file-ignores]
|
|
41
|
+
"__init__.py" = ["F401", "F403"]
|
|
42
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Interloper Microsoft Azure integration: Entra connection and Microsoft Fabric Warehouse destination."""
|
|
2
|
+
|
|
3
|
+
from interloper_azure.connection import AzureConnection
|
|
4
|
+
from interloper_azure.fabric import FabricWarehouseDestination
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"AzureConnection",
|
|
8
|
+
"FabricWarehouseDestination",
|
|
9
|
+
]
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Microsoft Azure connection resource holding a Microsoft Entra service principal."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Generator
|
|
6
|
+
from functools import cached_property
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import httpx2
|
|
10
|
+
from azure.identity import ClientSecretCredential
|
|
11
|
+
from interloper.connection import Connection, connection
|
|
12
|
+
from interloper.resource.fields import InputField, SecretField, fetch_field_provider
|
|
13
|
+
from interloper.rest import JSONCursorPaginator, RESTClient
|
|
14
|
+
from pydantic_settings import SettingsConfigDict
|
|
15
|
+
|
|
16
|
+
FABRIC_API = "https://api.fabric.microsoft.com/v1"
|
|
17
|
+
FABRIC_SCOPE = "https://api.fabric.microsoft.com/.default"
|
|
18
|
+
|
|
19
|
+
# Every REST call here serves an operator waiting on a form or a check.
|
|
20
|
+
_TIMEOUT = 30.0
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class _CredentialAuth(httpx2.Auth):
|
|
24
|
+
"""Bearer auth drawing the token from the connection on every request.
|
|
25
|
+
|
|
26
|
+
The credential caches the token and renews it before it expires, so a
|
|
27
|
+
long-lived client never sends a stale one.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, owner: AzureConnection, scope: str) -> None:
|
|
31
|
+
"""Bind the auth to the connection whose credential signs the requests.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
owner: The connection holding the credential.
|
|
35
|
+
scope: The scope the token is requested for.
|
|
36
|
+
"""
|
|
37
|
+
self._owner = owner
|
|
38
|
+
self._scope = scope
|
|
39
|
+
|
|
40
|
+
def auth_flow(self, request: httpx2.Request) -> Generator[httpx2.Request, httpx2.Response, None]:
|
|
41
|
+
"""Authenticate the request with a bearer token for the scope.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
request: The request to authenticate.
|
|
45
|
+
|
|
46
|
+
Yields:
|
|
47
|
+
The authenticated request.
|
|
48
|
+
"""
|
|
49
|
+
request.headers["Authorization"] = f"Bearer {self._owner.token(self._scope)}"
|
|
50
|
+
yield request
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@connection(
|
|
54
|
+
key="azure_connection",
|
|
55
|
+
name="Microsoft Azure",
|
|
56
|
+
icon="icon:azure",
|
|
57
|
+
tags=["Cloud"],
|
|
58
|
+
)
|
|
59
|
+
class AzureConnection(Connection):
|
|
60
|
+
"""Connection resource holding a Microsoft Entra service principal.
|
|
61
|
+
|
|
62
|
+
The principal is an app registration's client ID and secret in a tenant.
|
|
63
|
+
``azure-identity`` only obtains the tokens; every REST call goes through
|
|
64
|
+
httpx2.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
model_config = SettingsConfigDict(env_prefix="azure_")
|
|
68
|
+
|
|
69
|
+
tenant_id: str = InputField(label="Tenant ID", description="Directory (tenant) ID of the Microsoft Entra tenant")
|
|
70
|
+
client_id: str = InputField(label="Client ID", description="Application (client) ID of the service principal")
|
|
71
|
+
client_secret: str = SecretField(
|
|
72
|
+
label="Client secret",
|
|
73
|
+
description="Client secret of the service principal",
|
|
74
|
+
info="From the app registration's Certificates & secrets page: the secret's Value, not its ID.",
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
@cached_property
|
|
78
|
+
def credential(self) -> ClientSecretCredential:
|
|
79
|
+
"""The credential every token is obtained through.
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
The client-secret credential, cached per connection instance so
|
|
83
|
+
its token cache is shared by every caller.
|
|
84
|
+
"""
|
|
85
|
+
return ClientSecretCredential(self.tenant_id, self.client_id, self.client_secret)
|
|
86
|
+
|
|
87
|
+
def token(self, scope: str) -> str:
|
|
88
|
+
"""Obtain an access token for a scope.
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
scope: The scope requested, e.g. ``https://api.fabric.microsoft.com/.default``.
|
|
92
|
+
|
|
93
|
+
Returns:
|
|
94
|
+
The bearer token.
|
|
95
|
+
"""
|
|
96
|
+
return self.credential.get_token(scope).token
|
|
97
|
+
|
|
98
|
+
@cached_property
|
|
99
|
+
def client(self) -> RESTClient:
|
|
100
|
+
"""The Fabric REST API client the check and the picker share.
|
|
101
|
+
|
|
102
|
+
Returns:
|
|
103
|
+
The client, authenticated for the Fabric API and cached per connection instance.
|
|
104
|
+
"""
|
|
105
|
+
return RESTClient(FABRIC_API, auth=_CredentialAuth(self, FABRIC_SCOPE), timeout=_TIMEOUT)
|
|
106
|
+
|
|
107
|
+
def _list(self, path: str) -> list[dict[str, Any]]:
|
|
108
|
+
"""List every item of a Fabric collection, following continuation tokens.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
path: The collection path, relative to the API root.
|
|
112
|
+
|
|
113
|
+
Returns:
|
|
114
|
+
The items of every page.
|
|
115
|
+
"""
|
|
116
|
+
pages = self.client.paginate(
|
|
117
|
+
path,
|
|
118
|
+
JSONCursorPaginator(cursor_path="continuationToken", cursor_param="continuationToken"),
|
|
119
|
+
data_selector="value",
|
|
120
|
+
)
|
|
121
|
+
return [item for page in pages for item in page]
|
|
122
|
+
|
|
123
|
+
@fetch_field_provider
|
|
124
|
+
def workspaces(self) -> list[dict[str, str]]:
|
|
125
|
+
"""List the workspaces holding a warehouse, with the SQL endpoint each serves.
|
|
126
|
+
|
|
127
|
+
Backs the warehouse destination's ``server`` ``FetchField``. Every
|
|
128
|
+
warehouse of a workspace is served by the workspace's one SQL
|
|
129
|
+
connection string, so the first warehouse listed names it, and a
|
|
130
|
+
workspace without a warehouse has nothing to write to and is left out.
|
|
131
|
+
|
|
132
|
+
Returns:
|
|
133
|
+
Workspace options with ``id``, ``name`` and ``server``, sorted
|
|
134
|
+
case-insensitively by name.
|
|
135
|
+
"""
|
|
136
|
+
options: list[dict[str, str]] = []
|
|
137
|
+
for workspace in self._list("/workspaces"):
|
|
138
|
+
response = self.client.get(f"/workspaces/{workspace['id']}/warehouses")
|
|
139
|
+
response.raise_for_status()
|
|
140
|
+
warehouses = response.json().get("value", [])
|
|
141
|
+
if not warehouses:
|
|
142
|
+
continue
|
|
143
|
+
options.append(
|
|
144
|
+
{
|
|
145
|
+
"id": workspace["id"],
|
|
146
|
+
"name": workspace["displayName"],
|
|
147
|
+
"server": warehouses[0]["properties"]["connectionString"],
|
|
148
|
+
}
|
|
149
|
+
)
|
|
150
|
+
return sorted(options, key=lambda option: option["name"].lower())
|
|
151
|
+
|
|
152
|
+
def check(self) -> bool:
|
|
153
|
+
"""Prove the principal can authenticate and reach Fabric by listing one page of workspaces.
|
|
154
|
+
|
|
155
|
+
Returns:
|
|
156
|
+
True; a rejected secret raises out of the credential, and a principal
|
|
157
|
+
Fabric refuses (the tenant does not let service principals use
|
|
158
|
+
Fabric APIs) raises an HTTP 401 or 403.
|
|
159
|
+
"""
|
|
160
|
+
self.client.get("/workspaces").raise_for_status()
|
|
161
|
+
return True
|
|
@@ -0,0 +1,554 @@
|
|
|
1
|
+
"""Microsoft Fabric Warehouse destination over the warehouse's SQL (TDS) endpoint."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
import threading
|
|
9
|
+
import warnings
|
|
10
|
+
from collections.abc import Iterator, Sequence
|
|
11
|
+
from contextlib import contextmanager
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
import mssql_python
|
|
16
|
+
from interloper.destination import IOContext, destination
|
|
17
|
+
from interloper.destination.database import DatabaseDestination, PartitionFilter
|
|
18
|
+
from interloper.errors import DataNotFoundError
|
|
19
|
+
from interloper.representation import Representation
|
|
20
|
+
from interloper.resource.fields import FetchField, InputField
|
|
21
|
+
from interloper.schema import FieldSpec
|
|
22
|
+
from interloper.utils.json import json_default, replace_non_finite
|
|
23
|
+
from pydantic import PrivateAttr
|
|
24
|
+
|
|
25
|
+
from interloper_azure.connection import AzureConnection
|
|
26
|
+
from interloper_azure.fabric.types import column_type
|
|
27
|
+
|
|
28
|
+
DEFAULT_SCHEMA = "dbo"
|
|
29
|
+
|
|
30
|
+
# SQL Server caps a request at 2100 parameters; stay one under it, as
|
|
31
|
+
# SQLAlchemy's mssql dialect does.
|
|
32
|
+
MAX_PARAMETERS = 2099
|
|
33
|
+
|
|
34
|
+
# SQL Server's row cap for an INSERT ... VALUES list; Fabric documents none, so keep it.
|
|
35
|
+
MAX_ROWS = 1000
|
|
36
|
+
|
|
37
|
+
# Two writes creating the same schema, or partitions of one asset creating its
|
|
38
|
+
# table, would both pass the catalog check and the second CREATE would fail.
|
|
39
|
+
_DDL_LOCK = threading.Lock()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class _Transaction:
|
|
44
|
+
"""One write's transaction: the session it runs on, and whether it has begun.
|
|
45
|
+
|
|
46
|
+
Attributes:
|
|
47
|
+
session: The driver connection every statement of the write runs on.
|
|
48
|
+
begun: Whether ``BEGIN TRANSACTION`` has been issued on it.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
session: mssql_python.Connection
|
|
52
|
+
begun: bool = False
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@destination(
|
|
56
|
+
key="fabric_warehouse_destination",
|
|
57
|
+
name="Microsoft Fabric Warehouse",
|
|
58
|
+
icon="icon:fabric",
|
|
59
|
+
tags=["Cloud"],
|
|
60
|
+
)
|
|
61
|
+
class FabricWarehouseDestination(DatabaseDestination):
|
|
62
|
+
"""Microsoft Fabric Warehouse destination.
|
|
63
|
+
|
|
64
|
+
A dataset is a schema of the warehouse: the asset's dataset, else
|
|
65
|
+
``default_dataset``, else ``dbo``. Tables are created on first write with
|
|
66
|
+
typed columns and are never altered afterwards. Only a Warehouse accepts
|
|
67
|
+
writes; a Lakehouse's SQL analytics endpoint is read-only.
|
|
68
|
+
|
|
69
|
+
Statements go through Microsoft's ``mssql-python`` driver, signed in as
|
|
70
|
+
the connection's service principal with a Microsoft Entra access token.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
connection: AzureConnection
|
|
74
|
+
|
|
75
|
+
server: str = FetchField(
|
|
76
|
+
provider="connection.workspaces",
|
|
77
|
+
label_key="name",
|
|
78
|
+
value_key="server",
|
|
79
|
+
label="Workspace",
|
|
80
|
+
description="Workspace holding the warehouse",
|
|
81
|
+
info=(
|
|
82
|
+
"Stores the workspace's SQL connection string (…datawarehouse.fabric.microsoft.com), the one "
|
|
83
|
+
"shown in the warehouse's settings; every warehouse of a workspace shares it."
|
|
84
|
+
),
|
|
85
|
+
)
|
|
86
|
+
warehouse: str = InputField(description="Warehouse name, as shown in the workspace", discriminator=True)
|
|
87
|
+
default_dataset: str | None = InputField(
|
|
88
|
+
default=None, description="Default schema for assets without a dataset; dbo when empty"
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
_transactions: dict[int, _Transaction] = PrivateAttr(default_factory=dict)
|
|
92
|
+
|
|
93
|
+
# -- Session ---------------------------------------------------------------
|
|
94
|
+
|
|
95
|
+
def connect(self) -> mssql_python.Connection:
|
|
96
|
+
"""Open a session on the warehouse, signed in as the connection's service principal.
|
|
97
|
+
|
|
98
|
+
Autocommit stays on so a lone statement commits by itself; a write
|
|
99
|
+
opens an explicit ``BEGIN TRANSACTION`` (see :meth:`transaction`). The
|
|
100
|
+
driver takes its token from the connection's credential, which caches
|
|
101
|
+
and renews it, so every new session signs in with a valid one.
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
The new driver connection.
|
|
105
|
+
"""
|
|
106
|
+
return mssql_python.connect(
|
|
107
|
+
f"Server={_odbc(self.server)};Database={_odbc(self.warehouse)};Encrypt=yes;TrustServerCertificate=no",
|
|
108
|
+
autocommit=True,
|
|
109
|
+
token_provider=self.connection.credential,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
@contextmanager
|
|
113
|
+
def _session(self) -> Iterator[mssql_python.Connection]:
|
|
114
|
+
"""Yield the session an operation runs on.
|
|
115
|
+
|
|
116
|
+
Inside :meth:`transaction` that is the thread's transaction session,
|
|
117
|
+
so the delete and the insert commit together; otherwise a fresh
|
|
118
|
+
session, closed after use.
|
|
119
|
+
|
|
120
|
+
Yields:
|
|
121
|
+
The driver connection.
|
|
122
|
+
"""
|
|
123
|
+
held = self._transactions.get(threading.get_ident())
|
|
124
|
+
if held is not None:
|
|
125
|
+
yield held.session
|
|
126
|
+
return
|
|
127
|
+
session = self.connect()
|
|
128
|
+
try:
|
|
129
|
+
yield session
|
|
130
|
+
finally:
|
|
131
|
+
session.close()
|
|
132
|
+
|
|
133
|
+
def _begin(self) -> None:
|
|
134
|
+
"""Begin the calling thread's transaction, if it holds one that has not begun.
|
|
135
|
+
|
|
136
|
+
Called right before the first data change rather than on entering
|
|
137
|
+
:meth:`transaction`, so a table that write creates is created, and
|
|
138
|
+
committed, before the transaction opens. DDL inside a transaction is
|
|
139
|
+
allowed by the warehouse but holds locks on its system catalog views
|
|
140
|
+
until commit, blocking every other write's existence checks.
|
|
141
|
+
"""
|
|
142
|
+
held = self._transactions.get(threading.get_ident())
|
|
143
|
+
if held is not None and not held.begun:
|
|
144
|
+
_run(held.session, "BEGIN TRANSACTION")
|
|
145
|
+
held.begun = True
|
|
146
|
+
|
|
147
|
+
# -- Naming ----------------------------------------------------------------
|
|
148
|
+
|
|
149
|
+
def _schema(self, dataset: str | None) -> str:
|
|
150
|
+
"""Return the schema a dataset resolves to.
|
|
151
|
+
|
|
152
|
+
Args:
|
|
153
|
+
dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
|
|
154
|
+
|
|
155
|
+
Returns:
|
|
156
|
+
The schema name.
|
|
157
|
+
"""
|
|
158
|
+
return dataset or self.default_dataset or DEFAULT_SCHEMA
|
|
159
|
+
|
|
160
|
+
@staticmethod
|
|
161
|
+
def _ref(table: str, schema: str) -> str:
|
|
162
|
+
"""Build the quoted, schema-qualified table reference.
|
|
163
|
+
|
|
164
|
+
Args:
|
|
165
|
+
table: Table name.
|
|
166
|
+
schema: The resolved schema.
|
|
167
|
+
|
|
168
|
+
Returns:
|
|
169
|
+
``[schema].[table]``.
|
|
170
|
+
"""
|
|
171
|
+
return f"{_quote(schema)}.{_quote(table)}"
|
|
172
|
+
|
|
173
|
+
@staticmethod
|
|
174
|
+
def _not_found(table: str, schema: str) -> str:
|
|
175
|
+
"""Word the error a read raises on a table that does not exist.
|
|
176
|
+
|
|
177
|
+
Args:
|
|
178
|
+
table: Table name.
|
|
179
|
+
schema: The resolved schema.
|
|
180
|
+
|
|
181
|
+
Returns:
|
|
182
|
+
The error message.
|
|
183
|
+
"""
|
|
184
|
+
return f"Table '{schema}.{table}' does not exist. Has the asset been materialized?"
|
|
185
|
+
|
|
186
|
+
# -- Catalog ---------------------------------------------------------------
|
|
187
|
+
|
|
188
|
+
@staticmethod
|
|
189
|
+
def _columns(session: mssql_python.Connection, table: str, schema: str) -> list[str]:
|
|
190
|
+
"""Read a table's column names from the catalog views.
|
|
191
|
+
|
|
192
|
+
The query reads catalog views only: the warehouse refuses a query
|
|
193
|
+
that mixes system and user tables.
|
|
194
|
+
|
|
195
|
+
Args:
|
|
196
|
+
session: The session to query through.
|
|
197
|
+
table: Table name.
|
|
198
|
+
schema: The resolved schema.
|
|
199
|
+
|
|
200
|
+
Returns:
|
|
201
|
+
The column names in table order, empty when the table does not exist.
|
|
202
|
+
"""
|
|
203
|
+
_, rows = _fetch(
|
|
204
|
+
session,
|
|
205
|
+
"SELECT c.name FROM sys.columns AS c "
|
|
206
|
+
"JOIN sys.tables AS t ON c.object_id = t.object_id "
|
|
207
|
+
"JOIN sys.schemas AS s ON t.schema_id = s.schema_id "
|
|
208
|
+
"WHERE s.name = ? AND t.name = ? ORDER BY c.column_id",
|
|
209
|
+
[schema, table],
|
|
210
|
+
)
|
|
211
|
+
return [row[0] for row in rows]
|
|
212
|
+
|
|
213
|
+
def _create_table(
|
|
214
|
+
self, session: mssql_python.Connection, table: str, schema: str, specs: Sequence[FieldSpec]
|
|
215
|
+
) -> None:
|
|
216
|
+
"""Create the schema and the table, unless they already exist.
|
|
217
|
+
|
|
218
|
+
T-SQL has no ``CREATE SCHEMA IF NOT EXISTS``, so each creation is
|
|
219
|
+
guarded by a catalog check, one creation at a time in the process.
|
|
220
|
+
``CREATE SCHEMA`` must be alone in its batch, so it runs as its own
|
|
221
|
+
statement. Runs before the write's transaction begins, so each
|
|
222
|
+
statement commits by itself.
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
session: The session to run the statements on, outside a transaction.
|
|
226
|
+
table: Table name.
|
|
227
|
+
schema: The resolved schema.
|
|
228
|
+
specs: The table's field specs.
|
|
229
|
+
"""
|
|
230
|
+
with _DDL_LOCK:
|
|
231
|
+
if self._columns(session, table, schema):
|
|
232
|
+
return
|
|
233
|
+
_, found = _fetch(session, "SELECT 1 FROM sys.schemas WHERE name = ?", [schema])
|
|
234
|
+
if not found:
|
|
235
|
+
_run(session, f"CREATE SCHEMA {_quote(schema)}")
|
|
236
|
+
_run(session, f"CREATE TABLE {self._ref(table, schema)} ({_columns_ddl(specs)})")
|
|
237
|
+
|
|
238
|
+
# -- DatabaseDestination hooks ---------------------------------------------
|
|
239
|
+
|
|
240
|
+
@contextmanager
|
|
241
|
+
def transaction(self) -> Iterator[None]:
|
|
242
|
+
"""Run one write, a delete followed by an insert, as ``BEGIN ... COMMIT``.
|
|
243
|
+
|
|
244
|
+
The transaction lives on one session held for the calling thread,
|
|
245
|
+
which the hooks pick up through :meth:`_session`: one instance serves
|
|
246
|
+
every asset of a run, each write running whole on its own worker
|
|
247
|
+
thread, so concurrent writes each hold their own. It begins at the
|
|
248
|
+
first data change (see :meth:`_begin`). The warehouse runs every
|
|
249
|
+
transaction under snapshot isolation, so readers keep seeing the old
|
|
250
|
+
rows until the commit.
|
|
251
|
+
|
|
252
|
+
Yields:
|
|
253
|
+
``None``; the write runs inside the block, committed on success and
|
|
254
|
+
rolled back on any exception.
|
|
255
|
+
"""
|
|
256
|
+
ident = threading.get_ident()
|
|
257
|
+
held = _Transaction(self.connect())
|
|
258
|
+
self._transactions[ident] = held
|
|
259
|
+
try:
|
|
260
|
+
yield
|
|
261
|
+
except BaseException:
|
|
262
|
+
if held.begun:
|
|
263
|
+
_run(held.session, "ROLLBACK TRANSACTION")
|
|
264
|
+
raise
|
|
265
|
+
else:
|
|
266
|
+
if held.begun:
|
|
267
|
+
_run(held.session, "COMMIT TRANSACTION")
|
|
268
|
+
finally:
|
|
269
|
+
del self._transactions[ident]
|
|
270
|
+
held.session.close()
|
|
271
|
+
|
|
272
|
+
def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
|
|
273
|
+
"""Insert rows in multi-row ``INSERT ... VALUES`` batches, creating the table on first write.
|
|
274
|
+
|
|
275
|
+
A missing table is created from the effective schema (declared on the
|
|
276
|
+
asset, or inferred during conform), else from a schema inferred from
|
|
277
|
+
the data here, so the table is always typed. Columns the table does
|
|
278
|
+
not have are dropped with a warning. Each batch carries as many rows
|
|
279
|
+
as the parameter and row caps allow (see :func:`batch_size`).
|
|
280
|
+
|
|
281
|
+
Args:
|
|
282
|
+
table: Target table name.
|
|
283
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
284
|
+
data: The data in its native representation.
|
|
285
|
+
context: IO context carrying the asset and effective schema.
|
|
286
|
+
"""
|
|
287
|
+
schema = self._schema(dataset)
|
|
288
|
+
view = Representation.of(data)
|
|
289
|
+
with self._session() as session:
|
|
290
|
+
columns = self._columns(session, table, schema)
|
|
291
|
+
if not columns:
|
|
292
|
+
specs = (context.schema or view.infer()).field_specs()
|
|
293
|
+
self._create_table(session, table, schema, specs)
|
|
294
|
+
columns = self._columns(session, table, schema)
|
|
295
|
+
keys, rows = _align(view.records, columns, f"{schema}.{table}")
|
|
296
|
+
if not keys or not rows:
|
|
297
|
+
return
|
|
298
|
+
self._begin()
|
|
299
|
+
names = ", ".join(_quote(key) for key in keys)
|
|
300
|
+
row = "(" + ", ".join("?" for _ in keys) + ")"
|
|
301
|
+
size = batch_size(len(keys))
|
|
302
|
+
for start in range(0, len(rows), size):
|
|
303
|
+
batch = rows[start : start + size]
|
|
304
|
+
_run(
|
|
305
|
+
session,
|
|
306
|
+
f"INSERT INTO {self._ref(table, schema)} ({names}) VALUES {', '.join(row for _ in batch)}",
|
|
307
|
+
[value for values in batch for value in values],
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
|
|
311
|
+
"""Delete the rows a filter selects, or every row.
|
|
312
|
+
|
|
313
|
+
A whole-table replace deletes rather than truncates: ``TRUNCATE``
|
|
314
|
+
takes a schema-modification lock that blocks readers until the
|
|
315
|
+
commit, a ``DELETE`` does not. A table that does not exist has nothing
|
|
316
|
+
to delete.
|
|
317
|
+
|
|
318
|
+
Args:
|
|
319
|
+
table: Target table name.
|
|
320
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
321
|
+
where: The rows to delete; ``None`` for the whole table.
|
|
322
|
+
"""
|
|
323
|
+
schema = self._schema(dataset)
|
|
324
|
+
with self._session() as session:
|
|
325
|
+
if not self._columns(session, table, schema):
|
|
326
|
+
return
|
|
327
|
+
self._begin()
|
|
328
|
+
ref = self._ref(table, schema)
|
|
329
|
+
if where is None:
|
|
330
|
+
_run(session, f"DELETE FROM {ref}")
|
|
331
|
+
return
|
|
332
|
+
predicate, parameters = _predicate(where)
|
|
333
|
+
_run(session, f"DELETE FROM {ref} WHERE {predicate}", parameters)
|
|
334
|
+
|
|
335
|
+
def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> list[dict[str, Any]]:
|
|
336
|
+
"""Select the rows a filter selects, or every row, as records.
|
|
337
|
+
|
|
338
|
+
Nested and repeated values come back as the JSON text they are stored as.
|
|
339
|
+
|
|
340
|
+
Args:
|
|
341
|
+
table: Target table name.
|
|
342
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
343
|
+
where: The rows to select; ``None`` for the whole table.
|
|
344
|
+
|
|
345
|
+
Returns:
|
|
346
|
+
The selected rows.
|
|
347
|
+
|
|
348
|
+
Raises:
|
|
349
|
+
DataNotFoundError: If the table does not exist yet.
|
|
350
|
+
"""
|
|
351
|
+
schema = self._schema(dataset)
|
|
352
|
+
with self._session() as session:
|
|
353
|
+
if not self._columns(session, table, schema):
|
|
354
|
+
raise DataNotFoundError(self._not_found(table, schema))
|
|
355
|
+
ref = self._ref(table, schema)
|
|
356
|
+
if where is None:
|
|
357
|
+
names, rows = _fetch(session, f"SELECT * FROM {ref}")
|
|
358
|
+
else:
|
|
359
|
+
predicate, parameters = _predicate(where)
|
|
360
|
+
names, rows = _fetch(session, f"SELECT * FROM {ref} WHERE {predicate}", parameters)
|
|
361
|
+
return [dict(zip(names, row, strict=True)) for row in rows]
|
|
362
|
+
|
|
363
|
+
def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
|
|
364
|
+
"""Return row counts grouped by a column's values, cast to text.
|
|
365
|
+
|
|
366
|
+
Args:
|
|
367
|
+
table: Target table name.
|
|
368
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
369
|
+
column: Column to group by.
|
|
370
|
+
|
|
371
|
+
Returns:
|
|
372
|
+
Mapping from the column's value (as a string) to its row count.
|
|
373
|
+
|
|
374
|
+
Raises:
|
|
375
|
+
DataNotFoundError: If the table does not exist.
|
|
376
|
+
"""
|
|
377
|
+
schema = self._schema(dataset)
|
|
378
|
+
value = f"CAST({_quote(column)} AS varchar(max))"
|
|
379
|
+
with self._session() as session:
|
|
380
|
+
if not self._columns(session, table, schema):
|
|
381
|
+
raise DataNotFoundError(self._not_found(table, schema))
|
|
382
|
+
_, rows = _fetch(
|
|
383
|
+
session,
|
|
384
|
+
f"SELECT {value} AS partition_value, COUNT(*) AS cnt FROM {self._ref(table, schema)} GROUP BY {value}",
|
|
385
|
+
)
|
|
386
|
+
return {row[0]: row[1] for row in rows}
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
# -- Utility functions ---------------------------------------------------------
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def batch_size(columns: int) -> int:
|
|
393
|
+
"""Return how many rows one ``INSERT ... VALUES`` statement carries.
|
|
394
|
+
|
|
395
|
+
Every value is one parameter, so a batch is bounded by
|
|
396
|
+
:data:`MAX_PARAMETERS` divided by the column count, and by
|
|
397
|
+
:data:`MAX_ROWS` for narrow tables.
|
|
398
|
+
|
|
399
|
+
Args:
|
|
400
|
+
columns: The number of columns each row binds.
|
|
401
|
+
|
|
402
|
+
Returns:
|
|
403
|
+
The rows per statement, at least one.
|
|
404
|
+
"""
|
|
405
|
+
return max(1, min(MAX_ROWS, MAX_PARAMETERS // max(columns, 1)))
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _quote(identifier: str) -> str:
|
|
409
|
+
"""Quote an identifier in T-SQL brackets, escaping embedded closing brackets.
|
|
410
|
+
|
|
411
|
+
Args:
|
|
412
|
+
identifier: A schema, table or column name.
|
|
413
|
+
|
|
414
|
+
Returns:
|
|
415
|
+
The bracketed identifier.
|
|
416
|
+
"""
|
|
417
|
+
return "[" + identifier.replace("]", "]]") + "]"
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _odbc(value: str) -> str:
|
|
421
|
+
"""Brace a connection string value so separators inside it are taken literally.
|
|
422
|
+
|
|
423
|
+
Args:
|
|
424
|
+
value: A server or database name.
|
|
425
|
+
|
|
426
|
+
Returns:
|
|
427
|
+
The value in braces, embedded closing braces doubled.
|
|
428
|
+
"""
|
|
429
|
+
return "{" + value.replace("}", "}}") + "}"
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _columns_ddl(specs: Sequence[FieldSpec]) -> str:
|
|
433
|
+
"""Render field specs as the column list of a ``CREATE TABLE``.
|
|
434
|
+
|
|
435
|
+
Args:
|
|
436
|
+
specs: The table's field specs.
|
|
437
|
+
|
|
438
|
+
Returns:
|
|
439
|
+
Comma-separated bracketed column definitions. Every column is
|
|
440
|
+
nullable: conform already enforces the schema's nullability.
|
|
441
|
+
"""
|
|
442
|
+
return ", ".join(f"{_quote(spec.name)} {column_type(spec)} NULL" for spec in specs)
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def _run(session: mssql_python.Connection, sql: str, parameters: Sequence[Any] | None = None) -> None:
|
|
446
|
+
"""Run one statement that returns no rows.
|
|
447
|
+
|
|
448
|
+
Args:
|
|
449
|
+
session: The session to run it on.
|
|
450
|
+
sql: The statement, with ``?`` placeholders for *parameters*.
|
|
451
|
+
parameters: The statement's parameters; defaults to none.
|
|
452
|
+
"""
|
|
453
|
+
cursor = session.cursor()
|
|
454
|
+
try:
|
|
455
|
+
if parameters is None:
|
|
456
|
+
cursor.execute(sql)
|
|
457
|
+
else:
|
|
458
|
+
cursor.execute(sql, list(parameters))
|
|
459
|
+
finally:
|
|
460
|
+
cursor.close()
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _fetch(
|
|
464
|
+
session: mssql_python.Connection, sql: str, parameters: Sequence[Any] | None = None
|
|
465
|
+
) -> tuple[list[str], list[tuple[Any, ...]]]:
|
|
466
|
+
"""Run one query and fetch its rows.
|
|
467
|
+
|
|
468
|
+
Args:
|
|
469
|
+
session: The session to run it on.
|
|
470
|
+
sql: The query, with ``?`` placeholders for *parameters*.
|
|
471
|
+
parameters: The query's parameters; defaults to none.
|
|
472
|
+
|
|
473
|
+
Returns:
|
|
474
|
+
The result's column names and its rows, as plain tuples.
|
|
475
|
+
"""
|
|
476
|
+
cursor = session.cursor()
|
|
477
|
+
try:
|
|
478
|
+
if parameters is None:
|
|
479
|
+
cursor.execute(sql)
|
|
480
|
+
else:
|
|
481
|
+
cursor.execute(sql, list(parameters))
|
|
482
|
+
names = [column[0] for column in cursor.description or []]
|
|
483
|
+
return names, [tuple(row) for row in cursor.fetchall()]
|
|
484
|
+
finally:
|
|
485
|
+
cursor.close()
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _predicate(where: PartitionFilter) -> tuple[str, list[Any]]:
|
|
489
|
+
"""Render a partition filter as a parameterised predicate.
|
|
490
|
+
|
|
491
|
+
No cast is needed: T-SQL converts a parameter to the column's type when
|
|
492
|
+
that type ranks higher, so a ``'2024-01-01'`` partition id compares as a
|
|
493
|
+
``date``.
|
|
494
|
+
|
|
495
|
+
Args:
|
|
496
|
+
where: The filter to render.
|
|
497
|
+
|
|
498
|
+
Returns:
|
|
499
|
+
The predicate text and the parameters its ``?`` placeholders name.
|
|
500
|
+
"""
|
|
501
|
+
column = _quote(where.column)
|
|
502
|
+
if where.bounds is None:
|
|
503
|
+
return f"{column} = ?", [_bind(where.value)]
|
|
504
|
+
start, end = where.bounds
|
|
505
|
+
return f"{column} >= ? AND {column} < ?", [_bind(start), _bind(end)]
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def _align(records: list[dict[str, Any]], columns: list[str], ref: str) -> tuple[list[str], list[tuple[Any, ...]]]:
|
|
509
|
+
"""Shape records to the table's columns for one batch of inserts.
|
|
510
|
+
|
|
511
|
+
Every row binds the same columns, the table columns any record carries,
|
|
512
|
+
so a column no record mentions is left ``NULL``.
|
|
513
|
+
|
|
514
|
+
Args:
|
|
515
|
+
records: The rows to write.
|
|
516
|
+
columns: The table's column names, in table order.
|
|
517
|
+
ref: The table's name, for the warning.
|
|
518
|
+
|
|
519
|
+
Returns:
|
|
520
|
+
The columns written, and each row's values in that order, bound for the driver.
|
|
521
|
+
"""
|
|
522
|
+
present = dict.fromkeys(key for record in records for key in record)
|
|
523
|
+
extras = [str(key) for key in present if key not in columns]
|
|
524
|
+
if extras:
|
|
525
|
+
warnings.warn(
|
|
526
|
+
f"Columns {extras} are not in the schema for '{ref}' and will not be written.",
|
|
527
|
+
UserWarning,
|
|
528
|
+
stacklevel=3,
|
|
529
|
+
)
|
|
530
|
+
keys = [name for name in columns if name in present]
|
|
531
|
+
return keys, [tuple(_bind(record.get(key)) for key in keys) for record in records]
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def _bind(value: Any) -> Any:
|
|
535
|
+
"""Turn a conformed value into what the driver binds for its column.
|
|
536
|
+
|
|
537
|
+
A nested or repeated value becomes JSON text; a non-finite float becomes
|
|
538
|
+
``NULL``, which no column rejects; a timezone-aware datetime becomes naive
|
|
539
|
+
UTC, since ``datetime2`` holds no offset and the driver would otherwise
|
|
540
|
+
send a ``datetimeoffset`` that the conversion truncates to local time.
|
|
541
|
+
|
|
542
|
+
Args:
|
|
543
|
+
value: A cell value.
|
|
544
|
+
|
|
545
|
+
Returns:
|
|
546
|
+
The value to bind.
|
|
547
|
+
"""
|
|
548
|
+
if isinstance(value, float) and not math.isfinite(value):
|
|
549
|
+
return None
|
|
550
|
+
if isinstance(value, (dict, list)):
|
|
551
|
+
return json.dumps(replace_non_finite(value), default=json_default)
|
|
552
|
+
if isinstance(value, datetime.datetime) and value.tzinfo is not None:
|
|
553
|
+
return value.astimezone(datetime.timezone.utc).replace(tzinfo=None)
|
|
554
|
+
return value
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Fabric Warehouse's view of interloper's field types: one table, read in order."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
from decimal import Decimal
|
|
7
|
+
|
|
8
|
+
from interloper.schema import FieldSpec
|
|
9
|
+
|
|
10
|
+
# The warehouse has no json column type and recommends varchar instead.
|
|
11
|
+
JSON = "varchar(max)"
|
|
12
|
+
|
|
13
|
+
# Ordered: the first base class that matches wins, so bool (a subclass of int)
|
|
14
|
+
# and datetime (a subclass of date) must come before their parents.
|
|
15
|
+
_PYTHON_TO_FABRIC: dict[type, str] = {
|
|
16
|
+
bool: "bit",
|
|
17
|
+
int: "bigint",
|
|
18
|
+
float: "float",
|
|
19
|
+
Decimal: "decimal(38,9)",
|
|
20
|
+
datetime.datetime: "datetime2(6)",
|
|
21
|
+
datetime.date: "date",
|
|
22
|
+
bytes: "varbinary(max)",
|
|
23
|
+
str: "varchar(max)",
|
|
24
|
+
dict: JSON,
|
|
25
|
+
list: JSON,
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def column_type(spec: FieldSpec) -> str:
|
|
30
|
+
"""Return the Fabric Warehouse column type for a field spec.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
|
|
34
|
+
|
|
35
|
+
Returns:
|
|
36
|
+
:data:`JSON` for a nested or repeated field, the type the spec's Python
|
|
37
|
+
type maps to otherwise, and ``varchar(max)`` for anything unmapped
|
|
38
|
+
(``typing.Any``).
|
|
39
|
+
"""
|
|
40
|
+
if spec.fields is not None or spec.repeated:
|
|
41
|
+
return JSON
|
|
42
|
+
if isinstance(spec.type, type):
|
|
43
|
+
for base, name in _PYTHON_TO_FABRIC.items():
|
|
44
|
+
if issubclass(spec.type, base):
|
|
45
|
+
return name
|
|
46
|
+
return "varchar(max)"
|