interloper-duckdb 0.94.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,118 @@
1
+ Metadata-Version: 2.3
2
+ Name: interloper-duckdb
3
+ Version: 0.94.0
4
+ Summary: Interloper DuckDB integration: DuckDB and MotherDuck destination and connection
5
+ Author: Guillaume Onfroy
6
+ Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
7
+ Requires-Dist: duckdb>=1.0
8
+ Requires-Dist: interloper-core
9
+ Requires-Dist: interloper-pandas
10
+ Requires-Python: >=3.10
11
+ Description-Content-Type: text/markdown
12
+
13
+ # interloper-duckdb
14
+
15
+ DuckDB tables as an interloper destination: a `DuckDBDestination` writing to a
16
+ local `.duckdb` file or a MotherDuck database, and the `DuckDBConnection` that
17
+ opens it.
18
+
19
+ ## Setup
20
+
21
+ A **local file** needs nothing but a path; DuckDB creates the file on first
22
+ write:
23
+
24
+ ```python
25
+ from interloper_duckdb import DuckDBConnection
26
+
27
+ connection = DuckDBConnection(database="./warehouse.duckdb")
28
+ ```
29
+
30
+ A **MotherDuck** database is named `md:<database>` and authenticates with a
31
+ service token from the MotherDuck settings page:
32
+
33
+ ```python
34
+ connection = DuckDBConnection(database="md:my_db", motherduck_token="...")
35
+ ```
36
+
37
+ Both fields also load from the environment (`DUCKDB_DATABASE`,
38
+ `DUCKDB_MOTHERDUCK_TOKEN`), so `DuckDBConnection()` works with no arguments.
39
+
40
+ ## Usage
41
+
42
+ ```python
43
+ import interloper as il
44
+ from interloper_duckdb import DuckDBConnection, DuckDBDestination
45
+
46
+ destination = DuckDBDestination(
47
+ connection=DuckDBConnection(database="./warehouse.duckdb"),
48
+ default_dataset="raw",
49
+ )
50
+ ```
51
+
52
+ In a deployed instance you configure this through the UI instead: add a DuckDB
53
+ connection, then a DuckDB destination using it.
54
+
55
+ ## Datasets are schemas
56
+
57
+ An asset's dataset is a DuckDB schema. A table lands in the asset's dataset,
58
+ else in the destination's `default_dataset`, else in `main`. The schema and
59
+ the table are created on the first write, the columns typed from the asset's
60
+ schema (or from one inferred from the data), and never altered afterwards: a
61
+ column the table does not have is dropped from the write with a warning.
62
+
63
+ | Field type | Column type |
64
+ |------------|-------------|
65
+ | `bool` | `BOOLEAN` |
66
+ | `int` | `BIGINT` |
67
+ | `float` | `DOUBLE` |
68
+ | `Decimal` | `DECIMAL(38,9)` |
69
+ | `datetime` | `TIMESTAMP` |
70
+ | `date` | `DATE` |
71
+ | `bytes` | `BLOB` |
72
+ | `str`, `Any` | `VARCHAR` |
73
+ | nested model, `list[...]` | `JSON` |
74
+
75
+ ## Partitions
76
+
77
+ A write replaces what it covers, in one transaction: the whole table for an
78
+ unpartitioned asset, the rows inside a time partition's bounds
79
+ (`day >= start AND day < end`), or the rows equal to a partition's id for any
80
+ other partitioning. A window deletes each partition it covers and inserts the
81
+ whole batch once. If the insert fails, the delete is rolled back and the table
82
+ keeps its previous rows.
83
+
84
+ ## One writer per file
85
+
86
+ A local DuckDB file admits one writing process at a time. Concurrent assets in
87
+ one process are fine (each write runs on its own cursor), but two processes
88
+ writing to the same file, such as two pods or a scheduler and a notebook,
89
+ fail to open it. A connection holds the file from its first use until its
90
+ process exits, and that includes the connection check the app runs from the
91
+ API process. Run the instance's writes in one process, or use MotherDuck,
92
+ which serves many writers.
93
+
94
+ ## Querying the tables
95
+
96
+ The tables are plain DuckDB tables, so any DuckDB client reads them:
97
+
98
+ ```bash
99
+ duckdb warehouse.duckdb -c 'SELECT * FROM raw.ads_stats LIMIT 10'
100
+ ```
101
+
102
+ ```python
103
+ import duckdb
104
+
105
+ duckdb.connect("warehouse.duckdb", read_only=True).sql("SELECT * FROM raw.ads_stats").df()
106
+ ```
107
+
108
+ Another process can open the file only while no process holds it for
109
+ writing. Open it `read_only`, so the reader does not lock the instance out in
110
+ turn.
111
+
112
+ ## Docker images
113
+
114
+ The published interloper images do not ship this package. They are built on
115
+ Alpine, and DuckDB publishes no musllinux wheels, so installing it there means
116
+ compiling DuckDB from source. Run it from a glibc-based image (for example a
117
+ `python:3.12-slim` base with `pip install interloper-duckdb`), or anywhere
118
+ outside the images: the CLI, a notebook, a local scheduler.
@@ -0,0 +1,106 @@
1
+ # interloper-duckdb
2
+
3
+ DuckDB tables as an interloper destination: a `DuckDBDestination` writing to a
4
+ local `.duckdb` file or a MotherDuck database, and the `DuckDBConnection` that
5
+ opens it.
6
+
7
+ ## Setup
8
+
9
+ A **local file** needs nothing but a path; DuckDB creates the file on first
10
+ write:
11
+
12
+ ```python
13
+ from interloper_duckdb import DuckDBConnection
14
+
15
+ connection = DuckDBConnection(database="./warehouse.duckdb")
16
+ ```
17
+
18
+ A **MotherDuck** database is named `md:<database>` and authenticates with a
19
+ service token from the MotherDuck settings page:
20
+
21
+ ```python
22
+ connection = DuckDBConnection(database="md:my_db", motherduck_token="...")
23
+ ```
24
+
25
+ Both fields also load from the environment (`DUCKDB_DATABASE`,
26
+ `DUCKDB_MOTHERDUCK_TOKEN`), so `DuckDBConnection()` works with no arguments.
27
+
28
+ ## Usage
29
+
30
+ ```python
31
+ import interloper as il
32
+ from interloper_duckdb import DuckDBConnection, DuckDBDestination
33
+
34
+ destination = DuckDBDestination(
35
+ connection=DuckDBConnection(database="./warehouse.duckdb"),
36
+ default_dataset="raw",
37
+ )
38
+ ```
39
+
40
+ In a deployed instance you configure this through the UI instead: add a DuckDB
41
+ connection, then a DuckDB destination using it.
42
+
43
+ ## Datasets are schemas
44
+
45
+ An asset's dataset is a DuckDB schema. A table lands in the asset's dataset,
46
+ else in the destination's `default_dataset`, else in `main`. The schema and
47
+ the table are created on the first write, the columns typed from the asset's
48
+ schema (or from one inferred from the data), and never altered afterwards: a
49
+ column the table does not have is dropped from the write with a warning.
50
+
51
+ | Field type | Column type |
52
+ |------------|-------------|
53
+ | `bool` | `BOOLEAN` |
54
+ | `int` | `BIGINT` |
55
+ | `float` | `DOUBLE` |
56
+ | `Decimal` | `DECIMAL(38,9)` |
57
+ | `datetime` | `TIMESTAMP` |
58
+ | `date` | `DATE` |
59
+ | `bytes` | `BLOB` |
60
+ | `str`, `Any` | `VARCHAR` |
61
+ | nested model, `list[...]` | `JSON` |
62
+
63
+ ## Partitions
64
+
65
+ A write replaces what it covers, in one transaction: the whole table for an
66
+ unpartitioned asset, the rows inside a time partition's bounds
67
+ (`day >= start AND day < end`), or the rows equal to a partition's id for any
68
+ other partitioning. A window deletes each partition it covers and inserts the
69
+ whole batch once. If the insert fails, the delete is rolled back and the table
70
+ keeps its previous rows.
71
+
72
+ ## One writer per file
73
+
74
+ A local DuckDB file admits one writing process at a time. Concurrent assets in
75
+ one process are fine (each write runs on its own cursor), but two processes
76
+ writing to the same file, such as two pods or a scheduler and a notebook,
77
+ fail to open it. A connection holds the file from its first use until its
78
+ process exits, and that includes the connection check the app runs from the
79
+ API process. Run the instance's writes in one process, or use MotherDuck,
80
+ which serves many writers.
81
+
82
+ ## Querying the tables
83
+
84
+ The tables are plain DuckDB tables, so any DuckDB client reads them:
85
+
86
+ ```bash
87
+ duckdb warehouse.duckdb -c 'SELECT * FROM raw.ads_stats LIMIT 10'
88
+ ```
89
+
90
+ ```python
91
+ import duckdb
92
+
93
+ duckdb.connect("warehouse.duckdb", read_only=True).sql("SELECT * FROM raw.ads_stats").df()
94
+ ```
95
+
96
+ Another process can open the file only while no process holds it for
97
+ writing. Open it `read_only`, so the reader does not lock the instance out in
98
+ turn.
99
+
100
+ ## Docker images
101
+
102
+ The published interloper images do not ship this package. They are built on
103
+ Alpine, and DuckDB publishes no musllinux wheels, so installing it there means
104
+ compiling DuckDB from source. Run it from a glibc-based image (for example a
105
+ `python:3.12-slim` base with `pip install interloper-duckdb`), or anywhere
106
+ outside the images: the CLI, a notebook, a local scheduler.
@@ -0,0 +1,63 @@
1
+ [project]
2
+ name = "interloper-duckdb"
3
+ version = "0.94.0"
4
+ description = "Interloper DuckDB integration: DuckDB and MotherDuck destination and connection"
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ dependencies = [
8
+ "duckdb>=1.0",
9
+ "interloper-core",
10
+ "interloper-pandas",
11
+ ]
12
+
13
+ [[project.authors]]
14
+ name = "Guillaume Onfroy"
15
+ email = "guillaume@digitlcloud.com"
16
+
17
+ [project.entry-points."interloper.components"]
18
+ duckdb = "interloper_duckdb"
19
+
20
+ [build-system]
21
+ requires = ["uv_build>=0.12.9,<0.13"]
22
+ build-backend = "uv_build"
23
+
24
+ [tool.uv.sources.interloper-core]
25
+ workspace = true
26
+
27
+ [tool.uv.sources.interloper-pandas]
28
+ workspace = true
29
+
30
+ [tool.ruff]
31
+ line-length = 120
32
+
33
+ [tool.ruff.lint]
34
+ preview = true
35
+ extend-select = [
36
+ "E",
37
+ "I",
38
+ "UP",
39
+ "ANN001",
40
+ "ANN201",
41
+ "ANN202",
42
+ "DOC",
43
+ "D",
44
+ ]
45
+
46
+ [tool.ruff.lint.pydocstyle]
47
+ convention = "google"
48
+
49
+ [tool.ruff.lint.per-file-ignores]
50
+ "__init__.py" = [
51
+ "F401",
52
+ "F403",
53
+ ]
54
+ "tests/**" = [
55
+ "ANN",
56
+ "F811",
57
+ "D101",
58
+ "D102",
59
+ "D103",
60
+ "D104",
61
+ "RUF069",
62
+ "PLW0108",
63
+ ]
@@ -0,0 +1,43 @@
1
+ # ###############
2
+ # PROJECT / UV
3
+ # ###############
4
+ [project]
5
+ name = "interloper-duckdb"
6
+ version = "0.94.0"
7
+ description = "Interloper DuckDB integration: DuckDB and MotherDuck destination and connection"
8
+ readme = "README.md"
9
+ authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
10
+ requires-python = ">=3.10"
11
+ dependencies = [
12
+ "duckdb>=1.0",
13
+ "interloper-core",
14
+ "interloper-pandas",
15
+ ]
16
+
17
+ [project.entry-points."interloper.components"]
18
+ duckdb = "interloper_duckdb"
19
+
20
+ [build-system]
21
+ requires = ["uv_build>=0.12.9,<0.13"]
22
+ build-backend = "uv_build"
23
+
24
+ [tool.uv.sources]
25
+ interloper-core = { workspace = true }
26
+ interloper-pandas = { workspace = true }
27
+
28
+ # ###############
29
+ # RUFF
30
+ # ###############
31
+ [tool.ruff]
32
+ line-length = 120
33
+
34
+ [tool.ruff.lint]
35
+ preview = true
36
+ extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
37
+
38
+ [tool.ruff.lint.pydocstyle]
39
+ convention = "google"
40
+
41
+ [tool.ruff.lint.per-file-ignores]
42
+ "__init__.py" = ["F401", "F403"]
43
+ "tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
@@ -0,0 +1,9 @@
1
+ """Interloper DuckDB integration: DuckDB and MotherDuck destination and connection."""
2
+
3
+ from interloper_duckdb.connection import DuckDBConnection
4
+ from interloper_duckdb.destination import DuckDBDestination
5
+
6
+ __all__ = [
7
+ "DuckDBConnection",
8
+ "DuckDBDestination",
9
+ ]
@@ -0,0 +1,91 @@
1
+ """DuckDB connection resource: a local database file or a MotherDuck database."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from functools import cached_property
6
+
7
+ import duckdb
8
+ from interloper.connection import Connection, connection
9
+ from interloper.resource.fields import InputField, SecretField, fetch_field_provider
10
+ from pydantic_settings import SettingsConfigDict
11
+
12
+ _SYSTEM_SCHEMAS = ("information_schema", "pg_catalog")
13
+
14
+
15
+ @connection(
16
+ key="duckdb_connection",
17
+ name="DuckDB",
18
+ icon="icon:duckdb",
19
+ tags=["Database"],
20
+ )
21
+ class DuckDBConnection(Connection):
22
+ """Connection resource opening a DuckDB database.
23
+
24
+ ``database`` is a path to a ``.duckdb`` file, or ``md:<database>`` for a
25
+ MotherDuck database, which authenticates with ``motherduck_token``.
26
+ """
27
+
28
+ model_config = SettingsConfigDict(env_prefix="duckdb_")
29
+
30
+ database: str = InputField(
31
+ label="Database",
32
+ description="Path to a .duckdb file, or md:<database> for MotherDuck",
33
+ info=(
34
+ "A local file is created on first use and admits one writing process at a time. "
35
+ "A MotherDuck database (md:my_db) needs the token below."
36
+ ),
37
+ )
38
+ motherduck_token: str | None = SecretField(
39
+ default=None,
40
+ label="MotherDuck token",
41
+ description="Service token; only for md: databases",
42
+ )
43
+
44
+ @cached_property
45
+ def client(self) -> duckdb.DuckDBPyConnection:
46
+ """The DuckDB connection every operation derives a cursor from.
47
+
48
+ One connection is opened per instance and never used directly: each
49
+ operation takes its own ``client.cursor()``, a duplicate connection to
50
+ the same database, so threads writing concurrently never share one
51
+ DuckDB connection object (which is not thread-safe).
52
+
53
+ Returns:
54
+ The connection, cached per connection instance.
55
+ """
56
+ config = {"motherduck_token": self.motherduck_token} if self.motherduck_token else {}
57
+ return duckdb.connect(self.database, config=config)
58
+
59
+ @fetch_field_provider
60
+ def schemas(self) -> list[dict[str, str]]:
61
+ """List the schemas of the database, system schemas excluded.
62
+
63
+ Nothing binds it yet: it exists so a future destination field can
64
+ offer the schemas as a picker instead of free text.
65
+
66
+ Returns:
67
+ Schema options with ``name``, sorted case-insensitively.
68
+ """
69
+ cursor = self.client.cursor()
70
+ try:
71
+ rows = cursor.execute(
72
+ "SELECT schema_name FROM information_schema.schemata "
73
+ "WHERE catalog_name = current_database() AND schema_name NOT IN (?, ?)",
74
+ list(_SYSTEM_SCHEMAS),
75
+ ).fetchall()
76
+ finally:
77
+ cursor.close()
78
+ return sorted(({"name": name} for (name,) in rows), key=lambda s: s["name"].lower())
79
+
80
+ def check(self) -> bool:
81
+ """Prove the database opens and answers a query.
82
+
83
+ Returns:
84
+ True; a database that cannot be opened or queried raises instead.
85
+ """
86
+ cursor = self.client.cursor()
87
+ try:
88
+ cursor.execute("SELECT 1").fetchall()
89
+ finally:
90
+ cursor.close()
91
+ return True
@@ -0,0 +1,372 @@
1
+ """DuckDB destination: tables in a DuckDB file or a MotherDuck database."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ import threading
7
+ import warnings
8
+ from collections.abc import Iterator
9
+ from contextlib import contextmanager
10
+ from dataclasses import dataclass
11
+ from typing import Any
12
+
13
+ import duckdb
14
+ import pandas as pd
15
+ from interloper.destination import IOContext, destination
16
+ from interloper.destination.database import DatabaseDestination, PartitionFilter
17
+ from interloper.errors import DataNotFoundError
18
+ from interloper.representation import Representation
19
+ from interloper.resource.fields import InputField
20
+ from interloper.schema import FieldSpec
21
+ from pydantic import PrivateAttr
22
+
23
+ from interloper_duckdb.connection import DuckDBConnection
24
+ from interloper_duckdb.types import column_type
25
+
26
+ DEFAULT_SCHEMA = "main"
27
+
28
+ _BATCH = "interloper_batch"
29
+
30
+ # DuckDB reports two overlapping creates of one schema or table as a
31
+ # write-write conflict even with IF NOT EXISTS, and table creation is rare.
32
+ _DDL_LOCK = threading.Lock()
33
+
34
+
35
+ @dataclass
36
+ class _Transaction:
37
+ """One write's transaction: the cursor it runs on, and whether it has begun.
38
+
39
+ Attributes:
40
+ cursor: The cursor every statement of the write runs on.
41
+ begun: Whether ``BEGIN TRANSACTION`` has been issued on it.
42
+ """
43
+
44
+ cursor: duckdb.DuckDBPyConnection
45
+ begun: bool = False
46
+
47
+
48
+ @destination(
49
+ key="duckdb_destination",
50
+ name="DuckDB",
51
+ icon="icon:duckdb",
52
+ tags=["Database"],
53
+ )
54
+ class DuckDBDestination(DatabaseDestination):
55
+ """DuckDB destination.
56
+
57
+ A dataset is a DuckDB schema: the asset's dataset, else ``default_dataset``,
58
+ else ``main``. Tables are created on first write with typed columns and
59
+ are never altered afterwards.
60
+ """
61
+
62
+ connection: DuckDBConnection
63
+
64
+ default_dataset: str | None = InputField(default=None, description="Default schema for assets without a dataset")
65
+
66
+ _transactions: dict[int, _Transaction] = PrivateAttr(default_factory=dict)
67
+
68
+ # -- Helpers ---------------------------------------------------------------
69
+
70
+ def _schema(self, dataset: str | None) -> str:
71
+ """Return the schema a dataset resolves to.
72
+
73
+ Args:
74
+ dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
75
+
76
+ Returns:
77
+ The schema name.
78
+ """
79
+ return dataset or self.default_dataset or DEFAULT_SCHEMA
80
+
81
+ def _ref(self, table: str, dataset: str | None) -> str:
82
+ """Build the quoted, schema-qualified table reference.
83
+
84
+ Args:
85
+ table: Table name.
86
+ dataset: The schema, or ``None`` for the destination's default.
87
+
88
+ Returns:
89
+ ``"schema"."table"``.
90
+ """
91
+ return f"{_quote(self._schema(dataset))}.{_quote(table)}"
92
+
93
+ @contextmanager
94
+ def _cursor(self) -> Iterator[duckdb.DuckDBPyConnection]:
95
+ """Yield the cursor an operation runs on.
96
+
97
+ Inside :meth:`transaction` that is the thread's transaction cursor, so
98
+ the delete and the insert commit together; otherwise a fresh cursor,
99
+ closed after use.
100
+
101
+ Yields:
102
+ The cursor.
103
+ """
104
+ held = self._transactions.get(threading.get_ident())
105
+ if held is not None:
106
+ yield held.cursor
107
+ return
108
+ cursor = self.connection.client.cursor()
109
+ try:
110
+ yield cursor
111
+ finally:
112
+ cursor.close()
113
+
114
+ def _begin(self) -> None:
115
+ """Begin the calling thread's transaction, if it holds one that has not begun.
116
+
117
+ Called right before the first data change rather than on entering
118
+ :meth:`transaction`: a DuckDB transaction reads the catalog as of its
119
+ first statement, so a table created before it begins is visible to it.
120
+ """
121
+ held = self._transactions.get(threading.get_ident())
122
+ if held is not None and not held.begun:
123
+ held.cursor.execute("BEGIN TRANSACTION")
124
+ held.begun = True
125
+
126
+ def _columns(self, cursor: duckdb.DuckDBPyConnection, table: str, dataset: str | None) -> dict[str, str]:
127
+ """Read a table's columns and their types.
128
+
129
+ Args:
130
+ cursor: The cursor to query through.
131
+ table: Table name.
132
+ dataset: The schema, or ``None`` for the destination's default.
133
+
134
+ Returns:
135
+ Column name to DuckDB type in table order, empty when the table does not exist.
136
+ """
137
+ rows = cursor.execute(
138
+ "SELECT column_name, data_type FROM information_schema.columns "
139
+ "WHERE table_catalog = current_database() AND table_schema = ? AND table_name = ? "
140
+ "ORDER BY ordinal_position",
141
+ [self._schema(dataset), table],
142
+ ).fetchall()
143
+ return dict(rows)
144
+
145
+ def _create_table(
146
+ self, cursor: duckdb.DuckDBPyConnection, table: str, dataset: str | None, specs: list[FieldSpec]
147
+ ) -> None:
148
+ """Create the schema and the table, unless they already exist.
149
+
150
+ Runs before the write's transaction begins, each statement committing
151
+ on its own, and one creation at a time in the process, so concurrent
152
+ assets writing to a new schema (or partitions to a new table) never
153
+ race on it.
154
+
155
+ Args:
156
+ cursor: The cursor to run the statements on, outside a transaction.
157
+ table: Table name.
158
+ dataset: The schema, or ``None`` for the destination's default.
159
+ specs: The table's field specs.
160
+ """
161
+ columns = ", ".join(
162
+ f"{_quote(spec.name)} {column_type(spec)}{'' if spec.nullable else ' NOT NULL'}" for spec in specs
163
+ )
164
+ statements = (
165
+ f"CREATE SCHEMA IF NOT EXISTS {_quote(self._schema(dataset))}",
166
+ f"CREATE TABLE IF NOT EXISTS {self._ref(table, dataset)} ({columns})",
167
+ )
168
+ with _DDL_LOCK:
169
+ for statement in statements:
170
+ cursor.execute(statement)
171
+
172
+ def _predicate(self, where: PartitionFilter, types: dict[str, str]) -> tuple[str, list[Any]]:
173
+ """Render a partition filter as a parameterised predicate.
174
+
175
+ Each parameter is cast to the column's type, so a partition id that
176
+ arrives as a string compares against a ``DATE`` or ``BIGINT`` column.
177
+ Against a ``VARCHAR`` column, date and datetime bounds are rendered in
178
+ ISO 8601 (``T`` separator), the form the rows carry, since DuckDB's
179
+ own cast to text separates with a space and would compare out of order.
180
+
181
+ Args:
182
+ where: The filter to render.
183
+ types: The table's column types.
184
+
185
+ Returns:
186
+ The predicate text and its parameters.
187
+ """
188
+ column = _quote(where.column)
189
+ column_type = types.get(where.column)
190
+ placeholder = f"CAST(? AS {column_type})" if column_type is not None else "?"
191
+ if where.bounds is None:
192
+ return f"{column} = {placeholder}", [where.value]
193
+ start, end = where.bounds
194
+ if column_type == "VARCHAR":
195
+ start, end = (_iso(bound) for bound in (start, end))
196
+ return f"{column} >= {placeholder} AND {column} < {placeholder}", [start, end]
197
+
198
+ # -- DatabaseDestination hooks ---------------------------------------------
199
+
200
+ def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
201
+ """Insert data through a registered DataFrame, creating the table on first write.
202
+
203
+ A missing table is created from the effective schema (declared on the
204
+ asset, or inferred during conform), else from a schema inferred from
205
+ the data here, so the table is always typed. Columns the table does
206
+ not have are dropped with a warning, and the rest are inserted by name.
207
+
208
+ Args:
209
+ table: Target table name.
210
+ dataset: The schema, or ``None`` for the destination's default.
211
+ data: The data in its native representation.
212
+ context: IO context carrying the asset and effective schema.
213
+
214
+ Raises:
215
+ RuntimeError: If the table is still missing after creating it.
216
+ """
217
+ with self._cursor() as cursor:
218
+ columns = self._columns(cursor, table, dataset)
219
+ if not columns:
220
+ schema = context.schema or Representation.of(data).infer()
221
+ self._create_table(cursor, table, dataset, schema.field_specs())
222
+ columns = self._columns(cursor, table, dataset)
223
+ if not columns:
224
+ raise RuntimeError(f"Table '{self._schema(dataset)}.{table}' could not be created.")
225
+
226
+ ref = self._ref(table, dataset)
227
+ frame = Representation.of(data).to("dataframe")
228
+ extras = [str(c) for c in frame.columns if str(c) not in columns]
229
+ if extras:
230
+ warnings.warn(
231
+ f"Columns {extras} are not in the schema for '{self._schema(dataset)}.{table}' "
232
+ "and will not be written.",
233
+ UserWarning,
234
+ stacklevel=2,
235
+ )
236
+ present = ", ".join(_quote(c) for c in columns if c in frame.columns)
237
+ if not present:
238
+ return
239
+ self._begin()
240
+ cursor.register(_BATCH, frame)
241
+ try:
242
+ cursor.execute(f"INSERT INTO {ref} ({present}) SELECT {present} FROM {_BATCH}")
243
+ finally:
244
+ cursor.unregister(_BATCH)
245
+
246
+ def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
247
+ """Delete the rows a filter selects, or every row.
248
+
249
+ A table that does not exist has nothing to delete.
250
+
251
+ Args:
252
+ table: Target table name.
253
+ dataset: The schema, or ``None`` for the destination's default.
254
+ where: The rows to delete; ``None`` for the whole table.
255
+ """
256
+ with self._cursor() as cursor:
257
+ types = self._columns(cursor, table, dataset)
258
+ if not types:
259
+ return
260
+ self._begin()
261
+ ref = self._ref(table, dataset)
262
+ if where is None:
263
+ cursor.execute(f"DELETE FROM {ref}")
264
+ return
265
+ predicate, parameters = self._predicate(where, types)
266
+ cursor.execute(f"DELETE FROM {ref} WHERE {predicate}", parameters)
267
+
268
+ def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
269
+ """Select the rows a filter selects, or every row, as a DataFrame.
270
+
271
+ Args:
272
+ table: Target table name.
273
+ dataset: The schema, or ``None`` for the destination's default.
274
+ where: The rows to select; ``None`` for the whole table.
275
+
276
+ Returns:
277
+ The selected rows.
278
+
279
+ Raises:
280
+ DataNotFoundError: If the table does not exist yet.
281
+ """
282
+ with self._cursor() as cursor:
283
+ types = self._columns(cursor, table, dataset)
284
+ if not types:
285
+ raise DataNotFoundError(
286
+ f"Table '{self._schema(dataset)}.{table}' does not exist. Has the asset been materialized?"
287
+ )
288
+ ref = self._ref(table, dataset)
289
+ if where is None:
290
+ return cursor.execute(f"SELECT * FROM {ref}").df()
291
+ predicate, parameters = self._predicate(where, types)
292
+ return cursor.execute(f"SELECT * FROM {ref} WHERE {predicate}", parameters).df()
293
+
294
+ def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
295
+ """Return row counts grouped by a column.
296
+
297
+ Args:
298
+ table: Target table name.
299
+ dataset: The schema, or ``None`` for the destination's default.
300
+ column: Column to group by.
301
+
302
+ Returns:
303
+ Mapping from the column's value (as a string) to its row count.
304
+
305
+ Raises:
306
+ DataNotFoundError: If the table does not exist.
307
+ """
308
+ with self._cursor() as cursor:
309
+ if not self._columns(cursor, table, dataset):
310
+ raise DataNotFoundError(
311
+ f"Table '{self._schema(dataset)}.{table}' does not exist. Has the asset been materialized?"
312
+ )
313
+ rows = cursor.execute(
314
+ f"SELECT CAST({_quote(column)} AS VARCHAR) AS partition_value, COUNT(*) AS cnt "
315
+ f"FROM {self._ref(table, dataset)} GROUP BY 1"
316
+ ).fetchall()
317
+ return dict(rows)
318
+
319
+ @contextmanager
320
+ def transaction(self) -> Iterator[None]:
321
+ """Run one write, a delete followed by an insert, as one DuckDB transaction.
322
+
323
+ The transaction lives on one cursor held for the calling thread, which
324
+ the hooks pick up through :meth:`_cursor`; the engine runs a whole
325
+ write on one thread, and concurrent writes each hold their own. It
326
+ begins at the first data change (see :meth:`_begin`), so creating a
327
+ missing table is not part of it and survives a rollback.
328
+
329
+ Yields:
330
+ ``None``; the write runs inside the block, committed on success
331
+ and rolled back on any exception.
332
+ """
333
+ ident = threading.get_ident()
334
+ held = _Transaction(self.connection.client.cursor())
335
+ self._transactions[ident] = held
336
+ try:
337
+ yield
338
+ except BaseException:
339
+ if held.begun:
340
+ held.cursor.execute("ROLLBACK")
341
+ raise
342
+ else:
343
+ if held.begun:
344
+ held.cursor.execute("COMMIT")
345
+ finally:
346
+ del self._transactions[ident]
347
+ held.cursor.close()
348
+
349
+
350
+ def _iso(value: Any) -> Any:
351
+ """Render a date or datetime as ISO 8601, leaving any other value as is.
352
+
353
+ Args:
354
+ value: A partition bound.
355
+
356
+ Returns:
357
+ The ISO string for a date or datetime, else *value*.
358
+ """
359
+ return value.isoformat() if isinstance(value, datetime.date) else value
360
+
361
+
362
+ def _quote(identifier: str) -> str:
363
+ """Quote an identifier for DuckDB, escaping embedded double quotes.
364
+
365
+ Args:
366
+ identifier: A schema, table or column name.
367
+
368
+ Returns:
369
+ The double-quoted identifier.
370
+ """
371
+ escaped = identifier.replace('"', '""')
372
+ return f'"{escaped}"'
@@ -0,0 +1,45 @@
1
+ """DuckDB's view of interloper's field types."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ from decimal import Decimal
7
+
8
+ from interloper.schema import FieldSpec
9
+
10
+ JSON = "JSON"
11
+
12
+ # Ordered: the first base class that matches wins, so bool (a subclass of int)
13
+ # and datetime (a subclass of date) must come before their parents.
14
+ _PYTHON_TO_DUCKDB: dict[type, str] = {
15
+ bool: "BOOLEAN",
16
+ int: "BIGINT",
17
+ float: "DOUBLE",
18
+ Decimal: "DECIMAL(38,9)",
19
+ datetime.datetime: "TIMESTAMP",
20
+ datetime.date: "DATE",
21
+ bytes: "BLOB",
22
+ str: "VARCHAR",
23
+ }
24
+
25
+
26
+ def column_type(spec: FieldSpec) -> str:
27
+ """Return the DuckDB column type for a field spec.
28
+
29
+ Nested models and repeated fields are stored as ``JSON``; a scalar maps
30
+ through its Python type, and anything that is not a class
31
+ (``typing.Any``) or that matches no known type is a ``VARCHAR``.
32
+
33
+ Args:
34
+ spec: The field spec, from :meth:`Schema.field_specs`.
35
+
36
+ Returns:
37
+ A DuckDB type name.
38
+ """
39
+ if spec.fields is not None or spec.repeated:
40
+ return JSON
41
+ if isinstance(spec.type, type):
42
+ for base, name in _PYTHON_TO_DUCKDB.items():
43
+ if issubclass(spec.type, base):
44
+ return name
45
+ return "VARCHAR"