interloper-duckdb 0.94.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interloper_duckdb-0.94.0/PKG-INFO +118 -0
- interloper_duckdb-0.94.0/README.md +106 -0
- interloper_duckdb-0.94.0/pyproject.toml +63 -0
- interloper_duckdb-0.94.0/pyproject.toml.orig +43 -0
- interloper_duckdb-0.94.0/src/interloper_duckdb/__init__.py +9 -0
- interloper_duckdb-0.94.0/src/interloper_duckdb/connection.py +91 -0
- interloper_duckdb-0.94.0/src/interloper_duckdb/destination.py +372 -0
- interloper_duckdb-0.94.0/src/interloper_duckdb/types.py +45 -0
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: interloper-duckdb
|
|
3
|
+
Version: 0.94.0
|
|
4
|
+
Summary: Interloper DuckDB integration: DuckDB and MotherDuck destination and connection
|
|
5
|
+
Author: Guillaume Onfroy
|
|
6
|
+
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
+
Requires-Dist: duckdb>=1.0
|
|
8
|
+
Requires-Dist: interloper-core
|
|
9
|
+
Requires-Dist: interloper-pandas
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# interloper-duckdb
|
|
14
|
+
|
|
15
|
+
DuckDB tables as an interloper destination: a `DuckDBDestination` writing to a
|
|
16
|
+
local `.duckdb` file or a MotherDuck database, and the `DuckDBConnection` that
|
|
17
|
+
opens it.
|
|
18
|
+
|
|
19
|
+
## Setup
|
|
20
|
+
|
|
21
|
+
A **local file** needs nothing but a path; DuckDB creates the file on first
|
|
22
|
+
write:
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
from interloper_duckdb import DuckDBConnection
|
|
26
|
+
|
|
27
|
+
connection = DuckDBConnection(database="./warehouse.duckdb")
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
A **MotherDuck** database is named `md:<database>` and authenticates with a
|
|
31
|
+
service token from the MotherDuck settings page:
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
connection = DuckDBConnection(database="md:my_db", motherduck_token="...")
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Both fields also load from the environment (`DUCKDB_DATABASE`,
|
|
38
|
+
`DUCKDB_MOTHERDUCK_TOKEN`), so `DuckDBConnection()` works with no arguments.
|
|
39
|
+
|
|
40
|
+
## Usage
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
import interloper as il
|
|
44
|
+
from interloper_duckdb import DuckDBConnection, DuckDBDestination
|
|
45
|
+
|
|
46
|
+
destination = DuckDBDestination(
|
|
47
|
+
connection=DuckDBConnection(database="./warehouse.duckdb"),
|
|
48
|
+
default_dataset="raw",
|
|
49
|
+
)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
In a deployed instance you configure this through the UI instead: add a DuckDB
|
|
53
|
+
connection, then a DuckDB destination using it.
|
|
54
|
+
|
|
55
|
+
## Datasets are schemas
|
|
56
|
+
|
|
57
|
+
An asset's dataset is a DuckDB schema. A table lands in the asset's dataset,
|
|
58
|
+
else in the destination's `default_dataset`, else in `main`. The schema and
|
|
59
|
+
the table are created on the first write, the columns typed from the asset's
|
|
60
|
+
schema (or from one inferred from the data), and never altered afterwards: a
|
|
61
|
+
column the table does not have is dropped from the write with a warning.
|
|
62
|
+
|
|
63
|
+
| Field type | Column type |
|
|
64
|
+
|------------|-------------|
|
|
65
|
+
| `bool` | `BOOLEAN` |
|
|
66
|
+
| `int` | `BIGINT` |
|
|
67
|
+
| `float` | `DOUBLE` |
|
|
68
|
+
| `Decimal` | `DECIMAL(38,9)` |
|
|
69
|
+
| `datetime` | `TIMESTAMP` |
|
|
70
|
+
| `date` | `DATE` |
|
|
71
|
+
| `bytes` | `BLOB` |
|
|
72
|
+
| `str`, `Any` | `VARCHAR` |
|
|
73
|
+
| nested model, `list[...]` | `JSON` |
|
|
74
|
+
|
|
75
|
+
## Partitions
|
|
76
|
+
|
|
77
|
+
A write replaces what it covers, in one transaction: the whole table for an
|
|
78
|
+
unpartitioned asset, the rows inside a time partition's bounds
|
|
79
|
+
(`day >= start AND day < end`), or the rows equal to a partition's id for any
|
|
80
|
+
other partitioning. A window deletes each partition it covers and inserts the
|
|
81
|
+
whole batch once. If the insert fails, the delete is rolled back and the table
|
|
82
|
+
keeps its previous rows.
|
|
83
|
+
|
|
84
|
+
## One writer per file
|
|
85
|
+
|
|
86
|
+
A local DuckDB file admits one writing process at a time. Concurrent assets in
|
|
87
|
+
one process are fine (each write runs on its own cursor), but two processes
|
|
88
|
+
writing to the same file, such as two pods or a scheduler and a notebook,
|
|
89
|
+
fail to open it. A connection holds the file from its first use until its
|
|
90
|
+
process exits, and that includes the connection check the app runs from the
|
|
91
|
+
API process. Run the instance's writes in one process, or use MotherDuck,
|
|
92
|
+
which serves many writers.
|
|
93
|
+
|
|
94
|
+
## Querying the tables
|
|
95
|
+
|
|
96
|
+
The tables are plain DuckDB tables, so any DuckDB client reads them:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
duckdb warehouse.duckdb -c 'SELECT * FROM raw.ads_stats LIMIT 10'
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
import duckdb
|
|
104
|
+
|
|
105
|
+
duckdb.connect("warehouse.duckdb", read_only=True).sql("SELECT * FROM raw.ads_stats").df()
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Another process can open the file only while no process holds it for
|
|
109
|
+
writing. Open it `read_only`, so the reader does not lock the instance out in
|
|
110
|
+
turn.
|
|
111
|
+
|
|
112
|
+
## Docker images
|
|
113
|
+
|
|
114
|
+
The published interloper images do not ship this package. They are built on
|
|
115
|
+
Alpine, and DuckDB publishes no musllinux wheels, so installing it there means
|
|
116
|
+
compiling DuckDB from source. Run it from a glibc-based image (for example a
|
|
117
|
+
`python:3.12-slim` base with `pip install interloper-duckdb`), or anywhere
|
|
118
|
+
outside the images: the CLI, a notebook, a local scheduler.
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# interloper-duckdb
|
|
2
|
+
|
|
3
|
+
DuckDB tables as an interloper destination: a `DuckDBDestination` writing to a
|
|
4
|
+
local `.duckdb` file or a MotherDuck database, and the `DuckDBConnection` that
|
|
5
|
+
opens it.
|
|
6
|
+
|
|
7
|
+
## Setup
|
|
8
|
+
|
|
9
|
+
A **local file** needs nothing but a path; DuckDB creates the file on first
|
|
10
|
+
write:
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from interloper_duckdb import DuckDBConnection
|
|
14
|
+
|
|
15
|
+
connection = DuckDBConnection(database="./warehouse.duckdb")
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
A **MotherDuck** database is named `md:<database>` and authenticates with a
|
|
19
|
+
service token from the MotherDuck settings page:
|
|
20
|
+
|
|
21
|
+
```python
|
|
22
|
+
connection = DuckDBConnection(database="md:my_db", motherduck_token="...")
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Both fields also load from the environment (`DUCKDB_DATABASE`,
|
|
26
|
+
`DUCKDB_MOTHERDUCK_TOKEN`), so `DuckDBConnection()` works with no arguments.
|
|
27
|
+
|
|
28
|
+
## Usage
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
import interloper as il
|
|
32
|
+
from interloper_duckdb import DuckDBConnection, DuckDBDestination
|
|
33
|
+
|
|
34
|
+
destination = DuckDBDestination(
|
|
35
|
+
connection=DuckDBConnection(database="./warehouse.duckdb"),
|
|
36
|
+
default_dataset="raw",
|
|
37
|
+
)
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
In a deployed instance you configure this through the UI instead: add a DuckDB
|
|
41
|
+
connection, then a DuckDB destination using it.
|
|
42
|
+
|
|
43
|
+
## Datasets are schemas
|
|
44
|
+
|
|
45
|
+
An asset's dataset is a DuckDB schema. A table lands in the asset's dataset,
|
|
46
|
+
else in the destination's `default_dataset`, else in `main`. The schema and
|
|
47
|
+
the table are created on the first write, the columns typed from the asset's
|
|
48
|
+
schema (or from one inferred from the data), and never altered afterwards: a
|
|
49
|
+
column the table does not have is dropped from the write with a warning.
|
|
50
|
+
|
|
51
|
+
| Field type | Column type |
|
|
52
|
+
|------------|-------------|
|
|
53
|
+
| `bool` | `BOOLEAN` |
|
|
54
|
+
| `int` | `BIGINT` |
|
|
55
|
+
| `float` | `DOUBLE` |
|
|
56
|
+
| `Decimal` | `DECIMAL(38,9)` |
|
|
57
|
+
| `datetime` | `TIMESTAMP` |
|
|
58
|
+
| `date` | `DATE` |
|
|
59
|
+
| `bytes` | `BLOB` |
|
|
60
|
+
| `str`, `Any` | `VARCHAR` |
|
|
61
|
+
| nested model, `list[...]` | `JSON` |
|
|
62
|
+
|
|
63
|
+
## Partitions
|
|
64
|
+
|
|
65
|
+
A write replaces what it covers, in one transaction: the whole table for an
|
|
66
|
+
unpartitioned asset, the rows inside a time partition's bounds
|
|
67
|
+
(`day >= start AND day < end`), or the rows equal to a partition's id for any
|
|
68
|
+
other partitioning. A window deletes each partition it covers and inserts the
|
|
69
|
+
whole batch once. If the insert fails, the delete is rolled back and the table
|
|
70
|
+
keeps its previous rows.
|
|
71
|
+
|
|
72
|
+
## One writer per file
|
|
73
|
+
|
|
74
|
+
A local DuckDB file admits one writing process at a time. Concurrent assets in
|
|
75
|
+
one process are fine (each write runs on its own cursor), but two processes
|
|
76
|
+
writing to the same file, such as two pods or a scheduler and a notebook,
|
|
77
|
+
fail to open it. A connection holds the file from its first use until its
|
|
78
|
+
process exits, and that includes the connection check the app runs from the
|
|
79
|
+
API process. Run the instance's writes in one process, or use MotherDuck,
|
|
80
|
+
which serves many writers.
|
|
81
|
+
|
|
82
|
+
## Querying the tables
|
|
83
|
+
|
|
84
|
+
The tables are plain DuckDB tables, so any DuckDB client reads them:
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
duckdb warehouse.duckdb -c 'SELECT * FROM raw.ads_stats LIMIT 10'
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
import duckdb
|
|
92
|
+
|
|
93
|
+
duckdb.connect("warehouse.duckdb", read_only=True).sql("SELECT * FROM raw.ads_stats").df()
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Another process can open the file only while no process holds it for
|
|
97
|
+
writing. Open it `read_only`, so the reader does not lock the instance out in
|
|
98
|
+
turn.
|
|
99
|
+
|
|
100
|
+
## Docker images
|
|
101
|
+
|
|
102
|
+
The published interloper images do not ship this package. They are built on
|
|
103
|
+
Alpine, and DuckDB publishes no musllinux wheels, so installing it there means
|
|
104
|
+
compiling DuckDB from source. Run it from a glibc-based image (for example a
|
|
105
|
+
`python:3.12-slim` base with `pip install interloper-duckdb`), or anywhere
|
|
106
|
+
outside the images: the CLI, a notebook, a local scheduler.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "interloper-duckdb"
|
|
3
|
+
version = "0.94.0"
|
|
4
|
+
description = "Interloper DuckDB integration: DuckDB and MotherDuck destination and connection"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"duckdb>=1.0",
|
|
9
|
+
"interloper-core",
|
|
10
|
+
"interloper-pandas",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[[project.authors]]
|
|
14
|
+
name = "Guillaume Onfroy"
|
|
15
|
+
email = "guillaume@digitlcloud.com"
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
duckdb = "interloper_duckdb"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources.interloper-core]
|
|
25
|
+
workspace = true
|
|
26
|
+
|
|
27
|
+
[tool.uv.sources.interloper-pandas]
|
|
28
|
+
workspace = true
|
|
29
|
+
|
|
30
|
+
[tool.ruff]
|
|
31
|
+
line-length = 120
|
|
32
|
+
|
|
33
|
+
[tool.ruff.lint]
|
|
34
|
+
preview = true
|
|
35
|
+
extend-select = [
|
|
36
|
+
"E",
|
|
37
|
+
"I",
|
|
38
|
+
"UP",
|
|
39
|
+
"ANN001",
|
|
40
|
+
"ANN201",
|
|
41
|
+
"ANN202",
|
|
42
|
+
"DOC",
|
|
43
|
+
"D",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint.pydocstyle]
|
|
47
|
+
convention = "google"
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint.per-file-ignores]
|
|
50
|
+
"__init__.py" = [
|
|
51
|
+
"F401",
|
|
52
|
+
"F403",
|
|
53
|
+
]
|
|
54
|
+
"tests/**" = [
|
|
55
|
+
"ANN",
|
|
56
|
+
"F811",
|
|
57
|
+
"D101",
|
|
58
|
+
"D102",
|
|
59
|
+
"D103",
|
|
60
|
+
"D104",
|
|
61
|
+
"RUF069",
|
|
62
|
+
"PLW0108",
|
|
63
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# ###############
|
|
2
|
+
# PROJECT / UV
|
|
3
|
+
# ###############
|
|
4
|
+
[project]
|
|
5
|
+
name = "interloper-duckdb"
|
|
6
|
+
version = "0.94.0"
|
|
7
|
+
description = "Interloper DuckDB integration: DuckDB and MotherDuck destination and connection"
|
|
8
|
+
readme = "README.md"
|
|
9
|
+
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"duckdb>=1.0",
|
|
13
|
+
"interloper-core",
|
|
14
|
+
"interloper-pandas",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
duckdb = "interloper_duckdb"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources]
|
|
25
|
+
interloper-core = { workspace = true }
|
|
26
|
+
interloper-pandas = { workspace = true }
|
|
27
|
+
|
|
28
|
+
# ###############
|
|
29
|
+
# RUFF
|
|
30
|
+
# ###############
|
|
31
|
+
[tool.ruff]
|
|
32
|
+
line-length = 120
|
|
33
|
+
|
|
34
|
+
[tool.ruff.lint]
|
|
35
|
+
preview = true
|
|
36
|
+
extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
|
|
37
|
+
|
|
38
|
+
[tool.ruff.lint.pydocstyle]
|
|
39
|
+
convention = "google"
|
|
40
|
+
|
|
41
|
+
[tool.ruff.lint.per-file-ignores]
|
|
42
|
+
"__init__.py" = ["F401", "F403"]
|
|
43
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Interloper DuckDB integration: DuckDB and MotherDuck destination and connection."""
|
|
2
|
+
|
|
3
|
+
from interloper_duckdb.connection import DuckDBConnection
|
|
4
|
+
from interloper_duckdb.destination import DuckDBDestination
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"DuckDBConnection",
|
|
8
|
+
"DuckDBDestination",
|
|
9
|
+
]
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""DuckDB connection resource: a local database file or a MotherDuck database."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import cached_property
|
|
6
|
+
|
|
7
|
+
import duckdb
|
|
8
|
+
from interloper.connection import Connection, connection
|
|
9
|
+
from interloper.resource.fields import InputField, SecretField, fetch_field_provider
|
|
10
|
+
from pydantic_settings import SettingsConfigDict
|
|
11
|
+
|
|
12
|
+
_SYSTEM_SCHEMAS = ("information_schema", "pg_catalog")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@connection(
|
|
16
|
+
key="duckdb_connection",
|
|
17
|
+
name="DuckDB",
|
|
18
|
+
icon="icon:duckdb",
|
|
19
|
+
tags=["Database"],
|
|
20
|
+
)
|
|
21
|
+
class DuckDBConnection(Connection):
|
|
22
|
+
"""Connection resource opening a DuckDB database.
|
|
23
|
+
|
|
24
|
+
``database`` is a path to a ``.duckdb`` file, or ``md:<database>`` for a
|
|
25
|
+
MotherDuck database, which authenticates with ``motherduck_token``.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
model_config = SettingsConfigDict(env_prefix="duckdb_")
|
|
29
|
+
|
|
30
|
+
database: str = InputField(
|
|
31
|
+
label="Database",
|
|
32
|
+
description="Path to a .duckdb file, or md:<database> for MotherDuck",
|
|
33
|
+
info=(
|
|
34
|
+
"A local file is created on first use and admits one writing process at a time. "
|
|
35
|
+
"A MotherDuck database (md:my_db) needs the token below."
|
|
36
|
+
),
|
|
37
|
+
)
|
|
38
|
+
motherduck_token: str | None = SecretField(
|
|
39
|
+
default=None,
|
|
40
|
+
label="MotherDuck token",
|
|
41
|
+
description="Service token; only for md: databases",
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
@cached_property
|
|
45
|
+
def client(self) -> duckdb.DuckDBPyConnection:
|
|
46
|
+
"""The DuckDB connection every operation derives a cursor from.
|
|
47
|
+
|
|
48
|
+
One connection is opened per instance and never used directly: each
|
|
49
|
+
operation takes its own ``client.cursor()``, a duplicate connection to
|
|
50
|
+
the same database, so threads writing concurrently never share one
|
|
51
|
+
DuckDB connection object (which is not thread-safe).
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
The connection, cached per connection instance.
|
|
55
|
+
"""
|
|
56
|
+
config = {"motherduck_token": self.motherduck_token} if self.motherduck_token else {}
|
|
57
|
+
return duckdb.connect(self.database, config=config)
|
|
58
|
+
|
|
59
|
+
@fetch_field_provider
|
|
60
|
+
def schemas(self) -> list[dict[str, str]]:
|
|
61
|
+
"""List the schemas of the database, system schemas excluded.
|
|
62
|
+
|
|
63
|
+
Nothing binds it yet: it exists so a future destination field can
|
|
64
|
+
offer the schemas as a picker instead of free text.
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
Schema options with ``name``, sorted case-insensitively.
|
|
68
|
+
"""
|
|
69
|
+
cursor = self.client.cursor()
|
|
70
|
+
try:
|
|
71
|
+
rows = cursor.execute(
|
|
72
|
+
"SELECT schema_name FROM information_schema.schemata "
|
|
73
|
+
"WHERE catalog_name = current_database() AND schema_name NOT IN (?, ?)",
|
|
74
|
+
list(_SYSTEM_SCHEMAS),
|
|
75
|
+
).fetchall()
|
|
76
|
+
finally:
|
|
77
|
+
cursor.close()
|
|
78
|
+
return sorted(({"name": name} for (name,) in rows), key=lambda s: s["name"].lower())
|
|
79
|
+
|
|
80
|
+
def check(self) -> bool:
|
|
81
|
+
"""Prove the database opens and answers a query.
|
|
82
|
+
|
|
83
|
+
Returns:
|
|
84
|
+
True; a database that cannot be opened or queried raises instead.
|
|
85
|
+
"""
|
|
86
|
+
cursor = self.client.cursor()
|
|
87
|
+
try:
|
|
88
|
+
cursor.execute("SELECT 1").fetchall()
|
|
89
|
+
finally:
|
|
90
|
+
cursor.close()
|
|
91
|
+
return True
|
|
@@ -0,0 +1,372 @@
|
|
|
1
|
+
"""DuckDB destination: tables in a DuckDB file or a MotherDuck database."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
import threading
|
|
7
|
+
import warnings
|
|
8
|
+
from collections.abc import Iterator
|
|
9
|
+
from contextlib import contextmanager
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import duckdb
|
|
14
|
+
import pandas as pd
|
|
15
|
+
from interloper.destination import IOContext, destination
|
|
16
|
+
from interloper.destination.database import DatabaseDestination, PartitionFilter
|
|
17
|
+
from interloper.errors import DataNotFoundError
|
|
18
|
+
from interloper.representation import Representation
|
|
19
|
+
from interloper.resource.fields import InputField
|
|
20
|
+
from interloper.schema import FieldSpec
|
|
21
|
+
from pydantic import PrivateAttr
|
|
22
|
+
|
|
23
|
+
from interloper_duckdb.connection import DuckDBConnection
|
|
24
|
+
from interloper_duckdb.types import column_type
|
|
25
|
+
|
|
26
|
+
DEFAULT_SCHEMA = "main"
|
|
27
|
+
|
|
28
|
+
_BATCH = "interloper_batch"
|
|
29
|
+
|
|
30
|
+
# DuckDB reports two overlapping creates of one schema or table as a
|
|
31
|
+
# write-write conflict even with IF NOT EXISTS, and table creation is rare.
|
|
32
|
+
_DDL_LOCK = threading.Lock()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class _Transaction:
|
|
37
|
+
"""One write's transaction: the cursor it runs on, and whether it has begun.
|
|
38
|
+
|
|
39
|
+
Attributes:
|
|
40
|
+
cursor: The cursor every statement of the write runs on.
|
|
41
|
+
begun: Whether ``BEGIN TRANSACTION`` has been issued on it.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
cursor: duckdb.DuckDBPyConnection
|
|
45
|
+
begun: bool = False
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@destination(
|
|
49
|
+
key="duckdb_destination",
|
|
50
|
+
name="DuckDB",
|
|
51
|
+
icon="icon:duckdb",
|
|
52
|
+
tags=["Database"],
|
|
53
|
+
)
|
|
54
|
+
class DuckDBDestination(DatabaseDestination):
|
|
55
|
+
"""DuckDB destination.
|
|
56
|
+
|
|
57
|
+
A dataset is a DuckDB schema: the asset's dataset, else ``default_dataset``,
|
|
58
|
+
else ``main``. Tables are created on first write with typed columns and
|
|
59
|
+
are never altered afterwards.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
connection: DuckDBConnection
|
|
63
|
+
|
|
64
|
+
default_dataset: str | None = InputField(default=None, description="Default schema for assets without a dataset")
|
|
65
|
+
|
|
66
|
+
_transactions: dict[int, _Transaction] = PrivateAttr(default_factory=dict)
|
|
67
|
+
|
|
68
|
+
# -- Helpers ---------------------------------------------------------------
|
|
69
|
+
|
|
70
|
+
def _schema(self, dataset: str | None) -> str:
|
|
71
|
+
"""Return the schema a dataset resolves to.
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
|
|
75
|
+
|
|
76
|
+
Returns:
|
|
77
|
+
The schema name.
|
|
78
|
+
"""
|
|
79
|
+
return dataset or self.default_dataset or DEFAULT_SCHEMA
|
|
80
|
+
|
|
81
|
+
def _ref(self, table: str, dataset: str | None) -> str:
|
|
82
|
+
"""Build the quoted, schema-qualified table reference.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
table: Table name.
|
|
86
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
``"schema"."table"``.
|
|
90
|
+
"""
|
|
91
|
+
return f"{_quote(self._schema(dataset))}.{_quote(table)}"
|
|
92
|
+
|
|
93
|
+
@contextmanager
|
|
94
|
+
def _cursor(self) -> Iterator[duckdb.DuckDBPyConnection]:
|
|
95
|
+
"""Yield the cursor an operation runs on.
|
|
96
|
+
|
|
97
|
+
Inside :meth:`transaction` that is the thread's transaction cursor, so
|
|
98
|
+
the delete and the insert commit together; otherwise a fresh cursor,
|
|
99
|
+
closed after use.
|
|
100
|
+
|
|
101
|
+
Yields:
|
|
102
|
+
The cursor.
|
|
103
|
+
"""
|
|
104
|
+
held = self._transactions.get(threading.get_ident())
|
|
105
|
+
if held is not None:
|
|
106
|
+
yield held.cursor
|
|
107
|
+
return
|
|
108
|
+
cursor = self.connection.client.cursor()
|
|
109
|
+
try:
|
|
110
|
+
yield cursor
|
|
111
|
+
finally:
|
|
112
|
+
cursor.close()
|
|
113
|
+
|
|
114
|
+
def _begin(self) -> None:
|
|
115
|
+
"""Begin the calling thread's transaction, if it holds one that has not begun.
|
|
116
|
+
|
|
117
|
+
Called right before the first data change rather than on entering
|
|
118
|
+
:meth:`transaction`: a DuckDB transaction reads the catalog as of its
|
|
119
|
+
first statement, so a table created before it begins is visible to it.
|
|
120
|
+
"""
|
|
121
|
+
held = self._transactions.get(threading.get_ident())
|
|
122
|
+
if held is not None and not held.begun:
|
|
123
|
+
held.cursor.execute("BEGIN TRANSACTION")
|
|
124
|
+
held.begun = True
|
|
125
|
+
|
|
126
|
+
def _columns(self, cursor: duckdb.DuckDBPyConnection, table: str, dataset: str | None) -> dict[str, str]:
|
|
127
|
+
"""Read a table's columns and their types.
|
|
128
|
+
|
|
129
|
+
Args:
|
|
130
|
+
cursor: The cursor to query through.
|
|
131
|
+
table: Table name.
|
|
132
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
133
|
+
|
|
134
|
+
Returns:
|
|
135
|
+
Column name to DuckDB type in table order, empty when the table does not exist.
|
|
136
|
+
"""
|
|
137
|
+
rows = cursor.execute(
|
|
138
|
+
"SELECT column_name, data_type FROM information_schema.columns "
|
|
139
|
+
"WHERE table_catalog = current_database() AND table_schema = ? AND table_name = ? "
|
|
140
|
+
"ORDER BY ordinal_position",
|
|
141
|
+
[self._schema(dataset), table],
|
|
142
|
+
).fetchall()
|
|
143
|
+
return dict(rows)
|
|
144
|
+
|
|
145
|
+
def _create_table(
|
|
146
|
+
self, cursor: duckdb.DuckDBPyConnection, table: str, dataset: str | None, specs: list[FieldSpec]
|
|
147
|
+
) -> None:
|
|
148
|
+
"""Create the schema and the table, unless they already exist.
|
|
149
|
+
|
|
150
|
+
Runs before the write's transaction begins, each statement committing
|
|
151
|
+
on its own, and one creation at a time in the process, so concurrent
|
|
152
|
+
assets writing to a new schema (or partitions to a new table) never
|
|
153
|
+
race on it.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
cursor: The cursor to run the statements on, outside a transaction.
|
|
157
|
+
table: Table name.
|
|
158
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
159
|
+
specs: The table's field specs.
|
|
160
|
+
"""
|
|
161
|
+
columns = ", ".join(
|
|
162
|
+
f"{_quote(spec.name)} {column_type(spec)}{'' if spec.nullable else ' NOT NULL'}" for spec in specs
|
|
163
|
+
)
|
|
164
|
+
statements = (
|
|
165
|
+
f"CREATE SCHEMA IF NOT EXISTS {_quote(self._schema(dataset))}",
|
|
166
|
+
f"CREATE TABLE IF NOT EXISTS {self._ref(table, dataset)} ({columns})",
|
|
167
|
+
)
|
|
168
|
+
with _DDL_LOCK:
|
|
169
|
+
for statement in statements:
|
|
170
|
+
cursor.execute(statement)
|
|
171
|
+
|
|
172
|
+
def _predicate(self, where: PartitionFilter, types: dict[str, str]) -> tuple[str, list[Any]]:
|
|
173
|
+
"""Render a partition filter as a parameterised predicate.
|
|
174
|
+
|
|
175
|
+
Each parameter is cast to the column's type, so a partition id that
|
|
176
|
+
arrives as a string compares against a ``DATE`` or ``BIGINT`` column.
|
|
177
|
+
Against a ``VARCHAR`` column, date and datetime bounds are rendered in
|
|
178
|
+
ISO 8601 (``T`` separator), the form the rows carry, since DuckDB's
|
|
179
|
+
own cast to text separates with a space and would compare out of order.
|
|
180
|
+
|
|
181
|
+
Args:
|
|
182
|
+
where: The filter to render.
|
|
183
|
+
types: The table's column types.
|
|
184
|
+
|
|
185
|
+
Returns:
|
|
186
|
+
The predicate text and its parameters.
|
|
187
|
+
"""
|
|
188
|
+
column = _quote(where.column)
|
|
189
|
+
column_type = types.get(where.column)
|
|
190
|
+
placeholder = f"CAST(? AS {column_type})" if column_type is not None else "?"
|
|
191
|
+
if where.bounds is None:
|
|
192
|
+
return f"{column} = {placeholder}", [where.value]
|
|
193
|
+
start, end = where.bounds
|
|
194
|
+
if column_type == "VARCHAR":
|
|
195
|
+
start, end = (_iso(bound) for bound in (start, end))
|
|
196
|
+
return f"{column} >= {placeholder} AND {column} < {placeholder}", [start, end]
|
|
197
|
+
|
|
198
|
+
# -- DatabaseDestination hooks ---------------------------------------------
|
|
199
|
+
|
|
200
|
+
def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
|
|
201
|
+
"""Insert data through a registered DataFrame, creating the table on first write.
|
|
202
|
+
|
|
203
|
+
A missing table is created from the effective schema (declared on the
|
|
204
|
+
asset, or inferred during conform), else from a schema inferred from
|
|
205
|
+
the data here, so the table is always typed. Columns the table does
|
|
206
|
+
not have are dropped with a warning, and the rest are inserted by name.
|
|
207
|
+
|
|
208
|
+
Args:
|
|
209
|
+
table: Target table name.
|
|
210
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
211
|
+
data: The data in its native representation.
|
|
212
|
+
context: IO context carrying the asset and effective schema.
|
|
213
|
+
|
|
214
|
+
Raises:
|
|
215
|
+
RuntimeError: If the table is still missing after creating it.
|
|
216
|
+
"""
|
|
217
|
+
with self._cursor() as cursor:
|
|
218
|
+
columns = self._columns(cursor, table, dataset)
|
|
219
|
+
if not columns:
|
|
220
|
+
schema = context.schema or Representation.of(data).infer()
|
|
221
|
+
self._create_table(cursor, table, dataset, schema.field_specs())
|
|
222
|
+
columns = self._columns(cursor, table, dataset)
|
|
223
|
+
if not columns:
|
|
224
|
+
raise RuntimeError(f"Table '{self._schema(dataset)}.{table}' could not be created.")
|
|
225
|
+
|
|
226
|
+
ref = self._ref(table, dataset)
|
|
227
|
+
frame = Representation.of(data).to("dataframe")
|
|
228
|
+
extras = [str(c) for c in frame.columns if str(c) not in columns]
|
|
229
|
+
if extras:
|
|
230
|
+
warnings.warn(
|
|
231
|
+
f"Columns {extras} are not in the schema for '{self._schema(dataset)}.{table}' "
|
|
232
|
+
"and will not be written.",
|
|
233
|
+
UserWarning,
|
|
234
|
+
stacklevel=2,
|
|
235
|
+
)
|
|
236
|
+
present = ", ".join(_quote(c) for c in columns if c in frame.columns)
|
|
237
|
+
if not present:
|
|
238
|
+
return
|
|
239
|
+
self._begin()
|
|
240
|
+
cursor.register(_BATCH, frame)
|
|
241
|
+
try:
|
|
242
|
+
cursor.execute(f"INSERT INTO {ref} ({present}) SELECT {present} FROM {_BATCH}")
|
|
243
|
+
finally:
|
|
244
|
+
cursor.unregister(_BATCH)
|
|
245
|
+
|
|
246
|
+
def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
|
|
247
|
+
"""Delete the rows a filter selects, or every row.
|
|
248
|
+
|
|
249
|
+
A table that does not exist has nothing to delete.
|
|
250
|
+
|
|
251
|
+
Args:
|
|
252
|
+
table: Target table name.
|
|
253
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
254
|
+
where: The rows to delete; ``None`` for the whole table.
|
|
255
|
+
"""
|
|
256
|
+
with self._cursor() as cursor:
|
|
257
|
+
types = self._columns(cursor, table, dataset)
|
|
258
|
+
if not types:
|
|
259
|
+
return
|
|
260
|
+
self._begin()
|
|
261
|
+
ref = self._ref(table, dataset)
|
|
262
|
+
if where is None:
|
|
263
|
+
cursor.execute(f"DELETE FROM {ref}")
|
|
264
|
+
return
|
|
265
|
+
predicate, parameters = self._predicate(where, types)
|
|
266
|
+
cursor.execute(f"DELETE FROM {ref} WHERE {predicate}", parameters)
|
|
267
|
+
|
|
268
|
+
def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
|
|
269
|
+
"""Select the rows a filter selects, or every row, as a DataFrame.
|
|
270
|
+
|
|
271
|
+
Args:
|
|
272
|
+
table: Target table name.
|
|
273
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
274
|
+
where: The rows to select; ``None`` for the whole table.
|
|
275
|
+
|
|
276
|
+
Returns:
|
|
277
|
+
The selected rows.
|
|
278
|
+
|
|
279
|
+
Raises:
|
|
280
|
+
DataNotFoundError: If the table does not exist yet.
|
|
281
|
+
"""
|
|
282
|
+
with self._cursor() as cursor:
|
|
283
|
+
types = self._columns(cursor, table, dataset)
|
|
284
|
+
if not types:
|
|
285
|
+
raise DataNotFoundError(
|
|
286
|
+
f"Table '{self._schema(dataset)}.{table}' does not exist. Has the asset been materialized?"
|
|
287
|
+
)
|
|
288
|
+
ref = self._ref(table, dataset)
|
|
289
|
+
if where is None:
|
|
290
|
+
return cursor.execute(f"SELECT * FROM {ref}").df()
|
|
291
|
+
predicate, parameters = self._predicate(where, types)
|
|
292
|
+
return cursor.execute(f"SELECT * FROM {ref} WHERE {predicate}", parameters).df()
|
|
293
|
+
|
|
294
|
+
def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
|
|
295
|
+
"""Return row counts grouped by a column.
|
|
296
|
+
|
|
297
|
+
Args:
|
|
298
|
+
table: Target table name.
|
|
299
|
+
dataset: The schema, or ``None`` for the destination's default.
|
|
300
|
+
column: Column to group by.
|
|
301
|
+
|
|
302
|
+
Returns:
|
|
303
|
+
Mapping from the column's value (as a string) to its row count.
|
|
304
|
+
|
|
305
|
+
Raises:
|
|
306
|
+
DataNotFoundError: If the table does not exist.
|
|
307
|
+
"""
|
|
308
|
+
with self._cursor() as cursor:
|
|
309
|
+
if not self._columns(cursor, table, dataset):
|
|
310
|
+
raise DataNotFoundError(
|
|
311
|
+
f"Table '{self._schema(dataset)}.{table}' does not exist. Has the asset been materialized?"
|
|
312
|
+
)
|
|
313
|
+
rows = cursor.execute(
|
|
314
|
+
f"SELECT CAST({_quote(column)} AS VARCHAR) AS partition_value, COUNT(*) AS cnt "
|
|
315
|
+
f"FROM {self._ref(table, dataset)} GROUP BY 1"
|
|
316
|
+
).fetchall()
|
|
317
|
+
return dict(rows)
|
|
318
|
+
|
|
319
|
+
@contextmanager
|
|
320
|
+
def transaction(self) -> Iterator[None]:
|
|
321
|
+
"""Run one write, a delete followed by an insert, as one DuckDB transaction.
|
|
322
|
+
|
|
323
|
+
The transaction lives on one cursor held for the calling thread, which
|
|
324
|
+
the hooks pick up through :meth:`_cursor`; the engine runs a whole
|
|
325
|
+
write on one thread, and concurrent writes each hold their own. It
|
|
326
|
+
begins at the first data change (see :meth:`_begin`), so creating a
|
|
327
|
+
missing table is not part of it and survives a rollback.
|
|
328
|
+
|
|
329
|
+
Yields:
|
|
330
|
+
``None``; the write runs inside the block, committed on success
|
|
331
|
+
and rolled back on any exception.
|
|
332
|
+
"""
|
|
333
|
+
ident = threading.get_ident()
|
|
334
|
+
held = _Transaction(self.connection.client.cursor())
|
|
335
|
+
self._transactions[ident] = held
|
|
336
|
+
try:
|
|
337
|
+
yield
|
|
338
|
+
except BaseException:
|
|
339
|
+
if held.begun:
|
|
340
|
+
held.cursor.execute("ROLLBACK")
|
|
341
|
+
raise
|
|
342
|
+
else:
|
|
343
|
+
if held.begun:
|
|
344
|
+
held.cursor.execute("COMMIT")
|
|
345
|
+
finally:
|
|
346
|
+
del self._transactions[ident]
|
|
347
|
+
held.cursor.close()
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _iso(value: Any) -> Any:
|
|
351
|
+
"""Render a date or datetime as ISO 8601, leaving any other value as is.
|
|
352
|
+
|
|
353
|
+
Args:
|
|
354
|
+
value: A partition bound.
|
|
355
|
+
|
|
356
|
+
Returns:
|
|
357
|
+
The ISO string for a date or datetime, else *value*.
|
|
358
|
+
"""
|
|
359
|
+
return value.isoformat() if isinstance(value, datetime.date) else value
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _quote(identifier: str) -> str:
|
|
363
|
+
"""Quote an identifier for DuckDB, escaping embedded double quotes.
|
|
364
|
+
|
|
365
|
+
Args:
|
|
366
|
+
identifier: A schema, table or column name.
|
|
367
|
+
|
|
368
|
+
Returns:
|
|
369
|
+
The double-quoted identifier.
|
|
370
|
+
"""
|
|
371
|
+
escaped = identifier.replace('"', '""')
|
|
372
|
+
return f'"{escaped}"'
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""DuckDB's view of interloper's field types."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
from decimal import Decimal
|
|
7
|
+
|
|
8
|
+
from interloper.schema import FieldSpec
|
|
9
|
+
|
|
10
|
+
JSON = "JSON"
|
|
11
|
+
|
|
12
|
+
# Ordered: the first base class that matches wins, so bool (a subclass of int)
|
|
13
|
+
# and datetime (a subclass of date) must come before their parents.
|
|
14
|
+
_PYTHON_TO_DUCKDB: dict[type, str] = {
|
|
15
|
+
bool: "BOOLEAN",
|
|
16
|
+
int: "BIGINT",
|
|
17
|
+
float: "DOUBLE",
|
|
18
|
+
Decimal: "DECIMAL(38,9)",
|
|
19
|
+
datetime.datetime: "TIMESTAMP",
|
|
20
|
+
datetime.date: "DATE",
|
|
21
|
+
bytes: "BLOB",
|
|
22
|
+
str: "VARCHAR",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def column_type(spec: FieldSpec) -> str:
|
|
27
|
+
"""Return the DuckDB column type for a field spec.
|
|
28
|
+
|
|
29
|
+
Nested models and repeated fields are stored as ``JSON``; a scalar maps
|
|
30
|
+
through its Python type, and anything that is not a class
|
|
31
|
+
(``typing.Any``) or that matches no known type is a ``VARCHAR``.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
spec: The field spec, from :meth:`Schema.field_specs`.
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
A DuckDB type name.
|
|
38
|
+
"""
|
|
39
|
+
if spec.fields is not None or spec.repeated:
|
|
40
|
+
return JSON
|
|
41
|
+
if isinstance(spec.type, type):
|
|
42
|
+
for base, name in _PYTHON_TO_DUCKDB.items():
|
|
43
|
+
if issubclass(spec.type, base):
|
|
44
|
+
return name
|
|
45
|
+
return "VARCHAR"
|