interloper-clickhouse 0.96.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interloper_clickhouse-0.96.0/PKG-INFO +197 -0
- interloper_clickhouse-0.96.0/README.md +185 -0
- interloper_clickhouse-0.96.0/pyproject.toml +66 -0
- interloper_clickhouse-0.96.0/pyproject.toml.orig +46 -0
- interloper_clickhouse-0.96.0/src/interloper_clickhouse/__init__.py +9 -0
- interloper_clickhouse-0.96.0/src/interloper_clickhouse/connection.py +82 -0
- interloper_clickhouse-0.96.0/src/interloper_clickhouse/destination.py +599 -0
- interloper_clickhouse-0.96.0/src/interloper_clickhouse/partitioning.py +92 -0
- interloper_clickhouse-0.96.0/src/interloper_clickhouse/types.py +86 -0
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: interloper-clickhouse
|
|
3
|
+
Version: 0.96.0
|
|
4
|
+
Summary: Interloper ClickHouse integration: connection and destination
|
|
5
|
+
Author: Guillaume Onfroy
|
|
6
|
+
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
+
Requires-Dist: clickhouse-connect[pandas]>=0.8
|
|
8
|
+
Requires-Dist: interloper-core
|
|
9
|
+
Requires-Dist: interloper-pandas
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# interloper-clickhouse
|
|
14
|
+
|
|
15
|
+
ClickHouse integration for interloper: a `ClickHouseDestination` that stores
|
|
16
|
+
assets as ClickHouse `MergeTree` tables, and the `ClickHouseConnection` that
|
|
17
|
+
holds the server address and credentials. It works against self-hosted
|
|
18
|
+
ClickHouse and ClickHouse Cloud, over ClickHouse's HTTP interface through the
|
|
19
|
+
official [`clickhouse-connect`](https://clickhouse.com/docs/integrations/python) client.
|
|
20
|
+
|
|
21
|
+
## Setup
|
|
22
|
+
|
|
23
|
+
The connection needs a host, a user (`default` unless you say otherwise) and
|
|
24
|
+
its password.
|
|
25
|
+
|
|
26
|
+
**ClickHouse Cloud** only accepts TLS, on port 8443. Copy the host from the
|
|
27
|
+
service's *Connect* dialog and keep the defaults (`secure` on, no port):
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from interloper_clickhouse import ClickHouseConnection
|
|
31
|
+
|
|
32
|
+
connection = ClickHouseConnection(
|
|
33
|
+
host="abc123.eu-central-1.aws.clickhouse.cloud",
|
|
34
|
+
username="interloper",
|
|
35
|
+
password="...",
|
|
36
|
+
)
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
**Self-hosted** servers listen on 8123 for plain HTTP and 8443 for HTTPS. Turn
|
|
40
|
+
`secure` off for a server without TLS; the port then defaults to 8123:
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
connection = ClickHouseConnection(host="clickhouse.internal", password="...", secure=False)
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Every field also loads from the environment (`CLICKHOUSE_HOST`,
|
|
47
|
+
`CLICKHOUSE_PORT`, `CLICKHOUSE_USERNAME`, `CLICKHOUSE_PASSWORD`,
|
|
48
|
+
`CLICKHOUSE_SECURE`), so `ClickHouseConnection()` works with no arguments.
|
|
49
|
+
|
|
50
|
+
### Grants
|
|
51
|
+
|
|
52
|
+
The user needs, on the databases the assets write to:
|
|
53
|
+
|
|
54
|
+
- `CREATE DATABASE`, unless you create the databases yourself
|
|
55
|
+
- `CREATE TABLE` and `DROP TABLE`: each write creates a staging table next to
|
|
56
|
+
its target and drops it afterwards
|
|
57
|
+
- `INSERT` and `SELECT`
|
|
58
|
+
- `ALTER DELETE`, the privilege ClickHouse checks for `DROP PARTITION` and,
|
|
59
|
+
with `INSERT`, for `REPLACE PARTITION` (it also covers the `delete` hook's
|
|
60
|
+
lightweight `DELETE`), and `TRUNCATE` for that hook's whole-table case
|
|
61
|
+
|
|
62
|
+
A sketch for a dedicated user:
|
|
63
|
+
|
|
64
|
+
```sql
|
|
65
|
+
CREATE USER interloper IDENTIFIED BY '...';
|
|
66
|
+
GRANT CREATE DATABASE, CREATE TABLE, DROP TABLE, INSERT, SELECT, ALTER DELETE, TRUNCATE
|
|
67
|
+
ON marts.* TO interloper;
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Each staging load sets `max_partitions_per_insert_block`, so the user must be
|
|
71
|
+
allowed to change settings (not `readonly = 1`).
|
|
72
|
+
|
|
73
|
+
## Usage
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
import interloper as il
|
|
77
|
+
from interloper_clickhouse import ClickHouseConnection, ClickHouseDestination
|
|
78
|
+
|
|
79
|
+
destination = ClickHouseDestination(
|
|
80
|
+
connection=ClickHouseConnection(host="abc123.eu-central-1.aws.clickhouse.cloud", password="..."),
|
|
81
|
+
default_dataset="raw",
|
|
82
|
+
)
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
In a deployed instance you configure this through the UI instead: add a
|
|
86
|
+
ClickHouse connection, then a ClickHouse destination.
|
|
87
|
+
|
|
88
|
+
## Datasets are databases
|
|
89
|
+
|
|
90
|
+
An asset's `dataset` is the ClickHouse database its table lives in. An asset
|
|
91
|
+
without a dataset falls back to `default_dataset`, then to ClickHouse's
|
|
92
|
+
`default` database. A missing database is created on the first write
|
|
93
|
+
(`CREATE DATABASE IF NOT EXISTS`), and a missing table is created typed from
|
|
94
|
+
the asset's schema (or one inferred from the data):
|
|
95
|
+
|
|
96
|
+
| Field type | ClickHouse type |
|
|
97
|
+
|------------|-----------------|
|
|
98
|
+
| `bool` | `Bool` |
|
|
99
|
+
| `int` | `Int64` |
|
|
100
|
+
| `float` | `Float64` |
|
|
101
|
+
| `Decimal` | `Decimal(38, 9)` |
|
|
102
|
+
| `datetime` | `DateTime64(6, 'UTC')` |
|
|
103
|
+
| `date` | `Date32` |
|
|
104
|
+
| `bytes`, `str` | `String` |
|
|
105
|
+
| nested models, lists, dicts, `Any` | `String` holding JSON |
|
|
106
|
+
|
|
107
|
+
A nullable field (and an `Any` field) becomes `Nullable(T)`; every other
|
|
108
|
+
column is a plain `T`. Nested and repeated fields are JSON text
|
|
109
|
+
rather than the `JSON` type: that type is production-ready only from
|
|
110
|
+
ClickHouse 25.3, and it holds JSON objects, so a list field could not be
|
|
111
|
+
stored in it.
|
|
112
|
+
|
|
113
|
+
An existing table is never altered: a column the data carries but the table
|
|
114
|
+
does not is dropped with a warning.
|
|
115
|
+
|
|
116
|
+
## Table layout
|
|
117
|
+
|
|
118
|
+
Tables use the `MergeTree` engine (ClickHouse Cloud turns it into
|
|
119
|
+
`SharedMergeTree` by itself). The layout follows the asset's partitioning, so
|
|
120
|
+
that **one interloper partition is exactly one ClickHouse partition**:
|
|
121
|
+
|
|
122
|
+
| Asset partitioning | `PARTITION BY` | `ORDER BY` |
|
|
123
|
+
|--------------------|----------------|------------|
|
|
124
|
+
| none | (none) | `tuple()` |
|
|
125
|
+
| daily, on a `date` | `day` | `day` |
|
|
126
|
+
| daily, on a `datetime` | `toDate(at)` | `at` |
|
|
127
|
+
| hourly | `toStartOfHour(at)` | `at` |
|
|
128
|
+
| monthly | `toStartOfMonth(day)` | `day` |
|
|
129
|
+
| yearly | `toStartOfYear(day)` | `day` |
|
|
130
|
+
| any other `PartitionConfig` | `region` | `region` |
|
|
131
|
+
|
|
132
|
+
A time partition column held as text (ISO dates in a `String`) is parsed with
|
|
133
|
+
`parseDateTime64BestEffort` inside the key. Hourly partitions need a datetime
|
|
134
|
+
column; a `date` cannot hold them and the write fails with a `ConfigError`.
|
|
135
|
+
|
|
136
|
+
The partition column is the one column never made `Nullable`, since ClickHouse
|
|
137
|
+
keeps nullable columns out of partition and sorting keys. A row without a
|
|
138
|
+
partition value fails the load.
|
|
139
|
+
|
|
140
|
+
ClickHouse recommends keeping a table under about a thousand partitions, and
|
|
141
|
+
daily partitions reach that in under three years. Prefer monthly partitioning
|
|
142
|
+
for long histories.
|
|
143
|
+
|
|
144
|
+
If a table already exists with a different partition key (say the asset moved
|
|
145
|
+
from daily to monthly), the write fails with a `ConfigError` instead of
|
|
146
|
+
replacing the wrong rows: recreate the table, or partition the asset to match.
|
|
147
|
+
|
|
148
|
+
## Atomic replaces without transactions
|
|
149
|
+
|
|
150
|
+
ClickHouse has no production multi-statement transactions (they are
|
|
151
|
+
experimental and unavailable on ClickHouse Cloud), and `DELETE` is a mutation
|
|
152
|
+
rather than a cheap row operation. So the destination does not use core's
|
|
153
|
+
delete-then-insert. Each write:
|
|
154
|
+
|
|
155
|
+
1. creates a staging table `_interloper_staging_<table>_<random>` with
|
|
156
|
+
`CREATE TABLE ... AS <target>`, the same columns, engine and keys;
|
|
157
|
+
2. loads the whole batch into it in a single `insert_df` (a window is one
|
|
158
|
+
insert, as ClickHouse wants few large inserts);
|
|
159
|
+
3. for each partition the write covers, runs
|
|
160
|
+
`ALTER TABLE <target> REPLACE PARTITION ID '<id>' FROM <staging>`, or
|
|
161
|
+
`DROP PARTITION ID '<id>'` when the batch holds no rows for it. The ids
|
|
162
|
+
come from ClickHouse's own `partitionId` over the table's partition key, so
|
|
163
|
+
they match the parts whatever the column type;
|
|
164
|
+
4. drops the staging table (`DROP TABLE IF EXISTS ... SYNC`), whether or not
|
|
165
|
+
the steps before it succeeded.
|
|
166
|
+
|
|
167
|
+
An unpartitioned table is one partition, `tuple()`, replaced the same way. A
|
|
168
|
+
whole write to a partitioned table replaces every partition the table or the
|
|
169
|
+
batch holds.
|
|
170
|
+
|
|
171
|
+
What this guarantees:
|
|
172
|
+
|
|
173
|
+
- **Each partition is replaced atomically.** A reader sees a partition's old
|
|
174
|
+
rows or its new rows, never a mix and never neither.
|
|
175
|
+
- **A failed load leaves the target untouched**: nothing reaches the target
|
|
176
|
+
before the staging table holds the whole batch.
|
|
177
|
+
- **A window is not atomic as a whole.** It replaces its partitions one after
|
|
178
|
+
the other; a failure part way leaves the earlier ones replaced and the
|
|
179
|
+
later ones as they were. Running the write again converges.
|
|
180
|
+
- **A partition in a window that the data does not cover is cleared**, as in
|
|
181
|
+
every other destination.
|
|
182
|
+
- Rows whose partition value falls outside the partitions being written are
|
|
183
|
+
not written, with a warning.
|
|
184
|
+
|
|
185
|
+
Reads select by the asset's partition column with server-side bound
|
|
186
|
+
parameters typed as the column (`` `day` >= {start:Date32} AND `day` < {end:Date32} ``),
|
|
187
|
+
and come back as pandas DataFrames.
|
|
188
|
+
|
|
189
|
+
## Notes
|
|
190
|
+
|
|
191
|
+
One client serves every destination on a connection, across threads: each
|
|
192
|
+
statement is its own HTTP request, and session ids are turned off because a
|
|
193
|
+
ClickHouse session admits one query at a time.
|
|
194
|
+
|
|
195
|
+
The destination does not issue `ON CLUSTER` statements. On a self-hosted
|
|
196
|
+
cluster, use a `Replicated` database engine so tables and partition changes
|
|
197
|
+
reach every replica.
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
# interloper-clickhouse
|
|
2
|
+
|
|
3
|
+
ClickHouse integration for interloper: a `ClickHouseDestination` that stores
|
|
4
|
+
assets as ClickHouse `MergeTree` tables, and the `ClickHouseConnection` that
|
|
5
|
+
holds the server address and credentials. It works against self-hosted
|
|
6
|
+
ClickHouse and ClickHouse Cloud, over ClickHouse's HTTP interface through the
|
|
7
|
+
official [`clickhouse-connect`](https://clickhouse.com/docs/integrations/python) client.
|
|
8
|
+
|
|
9
|
+
## Setup
|
|
10
|
+
|
|
11
|
+
The connection needs a host, a user (`default` unless you say otherwise) and
|
|
12
|
+
its password.
|
|
13
|
+
|
|
14
|
+
**ClickHouse Cloud** only accepts TLS, on port 8443. Copy the host from the
|
|
15
|
+
service's *Connect* dialog and keep the defaults (`secure` on, no port):
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from interloper_clickhouse import ClickHouseConnection
|
|
19
|
+
|
|
20
|
+
connection = ClickHouseConnection(
|
|
21
|
+
host="abc123.eu-central-1.aws.clickhouse.cloud",
|
|
22
|
+
username="interloper",
|
|
23
|
+
password="...",
|
|
24
|
+
)
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
**Self-hosted** servers listen on 8123 for plain HTTP and 8443 for HTTPS. Turn
|
|
28
|
+
`secure` off for a server without TLS; the port then defaults to 8123:
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
connection = ClickHouseConnection(host="clickhouse.internal", password="...", secure=False)
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Every field also loads from the environment (`CLICKHOUSE_HOST`,
|
|
35
|
+
`CLICKHOUSE_PORT`, `CLICKHOUSE_USERNAME`, `CLICKHOUSE_PASSWORD`,
|
|
36
|
+
`CLICKHOUSE_SECURE`), so `ClickHouseConnection()` works with no arguments.
|
|
37
|
+
|
|
38
|
+
### Grants
|
|
39
|
+
|
|
40
|
+
The user needs, on the databases the assets write to:
|
|
41
|
+
|
|
42
|
+
- `CREATE DATABASE`, unless you create the databases yourself
|
|
43
|
+
- `CREATE TABLE` and `DROP TABLE`: each write creates a staging table next to
|
|
44
|
+
its target and drops it afterwards
|
|
45
|
+
- `INSERT` and `SELECT`
|
|
46
|
+
- `ALTER DELETE`, the privilege ClickHouse checks for `DROP PARTITION` and,
|
|
47
|
+
with `INSERT`, for `REPLACE PARTITION` (it also covers the `delete` hook's
|
|
48
|
+
lightweight `DELETE`), and `TRUNCATE` for that hook's whole-table case
|
|
49
|
+
|
|
50
|
+
A sketch for a dedicated user:
|
|
51
|
+
|
|
52
|
+
```sql
|
|
53
|
+
CREATE USER interloper IDENTIFIED BY '...';
|
|
54
|
+
GRANT CREATE DATABASE, CREATE TABLE, DROP TABLE, INSERT, SELECT, ALTER DELETE, TRUNCATE
|
|
55
|
+
ON marts.* TO interloper;
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Each staging load sets `max_partitions_per_insert_block`, so the user must be
|
|
59
|
+
allowed to change settings (not `readonly = 1`).
|
|
60
|
+
|
|
61
|
+
## Usage
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
import interloper as il
|
|
65
|
+
from interloper_clickhouse import ClickHouseConnection, ClickHouseDestination
|
|
66
|
+
|
|
67
|
+
destination = ClickHouseDestination(
|
|
68
|
+
connection=ClickHouseConnection(host="abc123.eu-central-1.aws.clickhouse.cloud", password="..."),
|
|
69
|
+
default_dataset="raw",
|
|
70
|
+
)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
In a deployed instance you configure this through the UI instead: add a
|
|
74
|
+
ClickHouse connection, then a ClickHouse destination.
|
|
75
|
+
|
|
76
|
+
## Datasets are databases
|
|
77
|
+
|
|
78
|
+
An asset's `dataset` is the ClickHouse database its table lives in. An asset
|
|
79
|
+
without a dataset falls back to `default_dataset`, then to ClickHouse's
|
|
80
|
+
`default` database. A missing database is created on the first write
|
|
81
|
+
(`CREATE DATABASE IF NOT EXISTS`), and a missing table is created typed from
|
|
82
|
+
the asset's schema (or one inferred from the data):
|
|
83
|
+
|
|
84
|
+
| Field type | ClickHouse type |
|
|
85
|
+
|------------|-----------------|
|
|
86
|
+
| `bool` | `Bool` |
|
|
87
|
+
| `int` | `Int64` |
|
|
88
|
+
| `float` | `Float64` |
|
|
89
|
+
| `Decimal` | `Decimal(38, 9)` |
|
|
90
|
+
| `datetime` | `DateTime64(6, 'UTC')` |
|
|
91
|
+
| `date` | `Date32` |
|
|
92
|
+
| `bytes`, `str` | `String` |
|
|
93
|
+
| nested models, lists, dicts, `Any` | `String` holding JSON |
|
|
94
|
+
|
|
95
|
+
A nullable field (and an `Any` field) becomes `Nullable(T)`; every other
|
|
96
|
+
column is a plain `T`. Nested and repeated fields are JSON text
|
|
97
|
+
rather than the `JSON` type: that type is production-ready only from
|
|
98
|
+
ClickHouse 25.3, and it holds JSON objects, so a list field could not be
|
|
99
|
+
stored in it.
|
|
100
|
+
|
|
101
|
+
An existing table is never altered: a column the data carries but the table
|
|
102
|
+
does not is dropped with a warning.
|
|
103
|
+
|
|
104
|
+
## Table layout
|
|
105
|
+
|
|
106
|
+
Tables use the `MergeTree` engine (ClickHouse Cloud turns it into
|
|
107
|
+
`SharedMergeTree` by itself). The layout follows the asset's partitioning, so
|
|
108
|
+
that **one interloper partition is exactly one ClickHouse partition**:
|
|
109
|
+
|
|
110
|
+
| Asset partitioning | `PARTITION BY` | `ORDER BY` |
|
|
111
|
+
|--------------------|----------------|------------|
|
|
112
|
+
| none | (none) | `tuple()` |
|
|
113
|
+
| daily, on a `date` | `day` | `day` |
|
|
114
|
+
| daily, on a `datetime` | `toDate(at)` | `at` |
|
|
115
|
+
| hourly | `toStartOfHour(at)` | `at` |
|
|
116
|
+
| monthly | `toStartOfMonth(day)` | `day` |
|
|
117
|
+
| yearly | `toStartOfYear(day)` | `day` |
|
|
118
|
+
| any other `PartitionConfig` | `region` | `region` |
|
|
119
|
+
|
|
120
|
+
A time partition column held as text (ISO dates in a `String`) is parsed with
|
|
121
|
+
`parseDateTime64BestEffort` inside the key. Hourly partitions need a datetime
|
|
122
|
+
column; a `date` cannot hold them and the write fails with a `ConfigError`.
|
|
123
|
+
|
|
124
|
+
The partition column is the one column never made `Nullable`, since ClickHouse
|
|
125
|
+
keeps nullable columns out of partition and sorting keys. A row without a
|
|
126
|
+
partition value fails the load.
|
|
127
|
+
|
|
128
|
+
ClickHouse recommends keeping a table under about a thousand partitions, and
|
|
129
|
+
daily partitions reach that in under three years. Prefer monthly partitioning
|
|
130
|
+
for long histories.
|
|
131
|
+
|
|
132
|
+
If a table already exists with a different partition key (say the asset moved
|
|
133
|
+
from daily to monthly), the write fails with a `ConfigError` instead of
|
|
134
|
+
replacing the wrong rows: recreate the table, or partition the asset to match.
|
|
135
|
+
|
|
136
|
+
## Atomic replaces without transactions
|
|
137
|
+
|
|
138
|
+
ClickHouse has no production multi-statement transactions (they are
|
|
139
|
+
experimental and unavailable on ClickHouse Cloud), and `DELETE` is a mutation
|
|
140
|
+
rather than a cheap row operation. So the destination does not use core's
|
|
141
|
+
delete-then-insert. Each write:
|
|
142
|
+
|
|
143
|
+
1. creates a staging table `_interloper_staging_<table>_<random>` with
|
|
144
|
+
`CREATE TABLE ... AS <target>`, the same columns, engine and keys;
|
|
145
|
+
2. loads the whole batch into it in a single `insert_df` (a window is one
|
|
146
|
+
insert, as ClickHouse wants few large inserts);
|
|
147
|
+
3. for each partition the write covers, runs
|
|
148
|
+
`ALTER TABLE <target> REPLACE PARTITION ID '<id>' FROM <staging>`, or
|
|
149
|
+
`DROP PARTITION ID '<id>'` when the batch holds no rows for it. The ids
|
|
150
|
+
come from ClickHouse's own `partitionId` over the table's partition key, so
|
|
151
|
+
they match the parts whatever the column type;
|
|
152
|
+
4. drops the staging table (`DROP TABLE IF EXISTS ... SYNC`), whether or not
|
|
153
|
+
the steps before it succeeded.
|
|
154
|
+
|
|
155
|
+
An unpartitioned table is one partition, `tuple()`, replaced the same way. A
|
|
156
|
+
whole write to a partitioned table replaces every partition the table or the
|
|
157
|
+
batch holds.
|
|
158
|
+
|
|
159
|
+
What this guarantees:
|
|
160
|
+
|
|
161
|
+
- **Each partition is replaced atomically.** A reader sees a partition's old
|
|
162
|
+
rows or its new rows, never a mix and never neither.
|
|
163
|
+
- **A failed load leaves the target untouched**: nothing reaches the target
|
|
164
|
+
before the staging table holds the whole batch.
|
|
165
|
+
- **A window is not atomic as a whole.** It replaces its partitions one after
|
|
166
|
+
the other; a failure part way leaves the earlier ones replaced and the
|
|
167
|
+
later ones as they were. Running the write again converges.
|
|
168
|
+
- **A partition in a window that the data does not cover is cleared**, as in
|
|
169
|
+
every other destination.
|
|
170
|
+
- Rows whose partition value falls outside the partitions being written are
|
|
171
|
+
not written, with a warning.
|
|
172
|
+
|
|
173
|
+
Reads select by the asset's partition column with server-side bound
|
|
174
|
+
parameters typed as the column (`` `day` >= {start:Date32} AND `day` < {end:Date32} ``),
|
|
175
|
+
and come back as pandas DataFrames.
|
|
176
|
+
|
|
177
|
+
## Notes
|
|
178
|
+
|
|
179
|
+
One client serves every destination on a connection, across threads: each
|
|
180
|
+
statement is its own HTTP request, and session ids are turned off because a
|
|
181
|
+
ClickHouse session admits one query at a time.
|
|
182
|
+
|
|
183
|
+
The destination does not issue `ON CLUSTER` statements. On a self-hosted
|
|
184
|
+
cluster, use a `Replicated` database engine so tables and partition changes
|
|
185
|
+
reach every replica.
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "interloper-clickhouse"
|
|
3
|
+
version = "0.96.0"
|
|
4
|
+
description = "Interloper ClickHouse integration: connection and destination"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"clickhouse-connect[pandas]>=0.8",
|
|
9
|
+
"interloper-core",
|
|
10
|
+
"interloper-pandas",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[[project.authors]]
|
|
14
|
+
name = "Guillaume Onfroy"
|
|
15
|
+
email = "guillaume@digitlcloud.com"
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
clickhouse = "interloper_clickhouse"
|
|
19
|
+
|
|
20
|
+
[dependency-groups]
|
|
21
|
+
dev = ["chdb>=3.0"]
|
|
22
|
+
|
|
23
|
+
[build-system]
|
|
24
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
25
|
+
build-backend = "uv_build"
|
|
26
|
+
|
|
27
|
+
[tool.uv.sources.interloper-core]
|
|
28
|
+
workspace = true
|
|
29
|
+
|
|
30
|
+
[tool.uv.sources.interloper-pandas]
|
|
31
|
+
workspace = true
|
|
32
|
+
|
|
33
|
+
[tool.ruff]
|
|
34
|
+
line-length = 120
|
|
35
|
+
|
|
36
|
+
[tool.ruff.lint]
|
|
37
|
+
preview = true
|
|
38
|
+
extend-select = [
|
|
39
|
+
"E",
|
|
40
|
+
"I",
|
|
41
|
+
"UP",
|
|
42
|
+
"ANN001",
|
|
43
|
+
"ANN201",
|
|
44
|
+
"ANN202",
|
|
45
|
+
"DOC",
|
|
46
|
+
"D",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint.pydocstyle]
|
|
50
|
+
convention = "google"
|
|
51
|
+
|
|
52
|
+
[tool.ruff.lint.per-file-ignores]
|
|
53
|
+
"__init__.py" = [
|
|
54
|
+
"F401",
|
|
55
|
+
"F403",
|
|
56
|
+
]
|
|
57
|
+
"tests/**" = [
|
|
58
|
+
"ANN",
|
|
59
|
+
"F811",
|
|
60
|
+
"D101",
|
|
61
|
+
"D102",
|
|
62
|
+
"D103",
|
|
63
|
+
"D104",
|
|
64
|
+
"RUF069",
|
|
65
|
+
"PLW0108",
|
|
66
|
+
]
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# ###############
|
|
2
|
+
# PROJECT / UV
|
|
3
|
+
# ###############
|
|
4
|
+
[project]
|
|
5
|
+
name = "interloper-clickhouse"
|
|
6
|
+
version = "0.96.0"
|
|
7
|
+
description = "Interloper ClickHouse integration: connection and destination"
|
|
8
|
+
readme = "README.md"
|
|
9
|
+
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"clickhouse-connect[pandas]>=0.8",
|
|
13
|
+
"interloper-core",
|
|
14
|
+
"interloper-pandas",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
clickhouse = "interloper_clickhouse"
|
|
19
|
+
|
|
20
|
+
[dependency-groups]
|
|
21
|
+
dev = ["chdb>=3.0"]
|
|
22
|
+
|
|
23
|
+
[build-system]
|
|
24
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
25
|
+
build-backend = "uv_build"
|
|
26
|
+
|
|
27
|
+
[tool.uv.sources]
|
|
28
|
+
interloper-core = { workspace = true }
|
|
29
|
+
interloper-pandas = { workspace = true }
|
|
30
|
+
|
|
31
|
+
# ###############
|
|
32
|
+
# RUFF
|
|
33
|
+
# ###############
|
|
34
|
+
[tool.ruff]
|
|
35
|
+
line-length = 120
|
|
36
|
+
|
|
37
|
+
[tool.ruff.lint]
|
|
38
|
+
preview = true
|
|
39
|
+
extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
|
|
40
|
+
|
|
41
|
+
[tool.ruff.lint.pydocstyle]
|
|
42
|
+
convention = "google"
|
|
43
|
+
|
|
44
|
+
[tool.ruff.lint.per-file-ignores]
|
|
45
|
+
"__init__.py" = ["F401", "F403"]
|
|
46
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Interloper ClickHouse integration: connection and destination."""
|
|
2
|
+
|
|
3
|
+
from interloper_clickhouse.connection import ClickHouseConnection
|
|
4
|
+
from interloper_clickhouse.destination import ClickHouseDestination
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"ClickHouseConnection",
|
|
8
|
+
"ClickHouseDestination",
|
|
9
|
+
]
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""ClickHouse connection resource holding server address and user credentials."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import cached_property
|
|
6
|
+
|
|
7
|
+
import clickhouse_connect
|
|
8
|
+
from clickhouse_connect.driver.client import Client
|
|
9
|
+
from interloper.connection import Connection, connection
|
|
10
|
+
from interloper.resource.fields import InputField, SecretField, fetch_field_provider
|
|
11
|
+
from pydantic import Field
|
|
12
|
+
from pydantic_settings import SettingsConfigDict
|
|
13
|
+
|
|
14
|
+
_SYSTEM_DATABASES = ("INFORMATION_SCHEMA", "information_schema", "system")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@connection(
|
|
18
|
+
key="clickhouse_connection",
|
|
19
|
+
name="ClickHouse",
|
|
20
|
+
icon="icon:clickhouse",
|
|
21
|
+
tags=["Database"],
|
|
22
|
+
)
|
|
23
|
+
class ClickHouseConnection(Connection):
|
|
24
|
+
"""Connection resource holding the address of a ClickHouse server and a user's credentials.
|
|
25
|
+
|
|
26
|
+
It talks to ClickHouse over its HTTP interface, which both self-hosted
|
|
27
|
+
servers and ClickHouse Cloud expose; Cloud only accepts TLS, on port 8443.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
model_config = SettingsConfigDict(env_prefix="clickhouse_")
|
|
31
|
+
|
|
32
|
+
host: str = InputField(description="Server host name, e.g. abc123.eu-central-1.aws.clickhouse.cloud")
|
|
33
|
+
port: int | None = Field(default=None, description="HTTP port; 8443 with TLS, 8123 without when empty")
|
|
34
|
+
username: str = InputField(default="default", description="ClickHouse user name")
|
|
35
|
+
password: str = SecretField(description="ClickHouse user password")
|
|
36
|
+
secure: bool = Field(default=True, title="TLS", description="Connect over HTTPS; ClickHouse Cloud requires it")
|
|
37
|
+
|
|
38
|
+
@cached_property
|
|
39
|
+
def client(self) -> Client:
|
|
40
|
+
"""The client every statement goes through, shared by every destination on this connection.
|
|
41
|
+
|
|
42
|
+
One client serves concurrent writes: each call is its own HTTP request
|
|
43
|
+
on a shared connection pool. Session ids are turned off because a
|
|
44
|
+
ClickHouse session admits one query at a time, and nothing here relies
|
|
45
|
+
on session state.
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
The client, cached per connection instance.
|
|
49
|
+
"""
|
|
50
|
+
return clickhouse_connect.get_client(
|
|
51
|
+
host=self.host,
|
|
52
|
+
port=self.port,
|
|
53
|
+
username=self.username,
|
|
54
|
+
password=self.password,
|
|
55
|
+
secure=self.secure,
|
|
56
|
+
autogenerate_session_id=False,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
@fetch_field_provider
|
|
60
|
+
def databases(self) -> list[dict[str, str]]:
|
|
61
|
+
"""List the databases this connection's user can see, system databases excluded.
|
|
62
|
+
|
|
63
|
+
Nothing binds it yet: it exists so a destination field can offer the
|
|
64
|
+
databases as a picker instead of free text.
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
Database options with ``name``, sorted case-insensitively.
|
|
68
|
+
"""
|
|
69
|
+
result = self.client.query(
|
|
70
|
+
"SELECT name FROM system.databases WHERE name NOT IN {system:Array(String)}",
|
|
71
|
+
parameters={"system": list(_SYSTEM_DATABASES)},
|
|
72
|
+
)
|
|
73
|
+
return sorted(({"name": name} for (name,) in result.result_rows), key=lambda option: option["name"].lower())
|
|
74
|
+
|
|
75
|
+
def check(self) -> bool:
|
|
76
|
+
"""Prove the server answers and the credentials work by running ``SELECT 1``.
|
|
77
|
+
|
|
78
|
+
Returns:
|
|
79
|
+
True; a network or authentication failure raises out of the client.
|
|
80
|
+
"""
|
|
81
|
+
self.client.command("SELECT 1")
|
|
82
|
+
return True
|
|
@@ -0,0 +1,599 @@
|
|
|
1
|
+
"""ClickHouse destination: MergeTree tables whose partitions are replaced atomically."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
import uuid
|
|
9
|
+
import warnings
|
|
10
|
+
from collections.abc import Sequence
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
from interloper.destination import IOContext, destination
|
|
15
|
+
from interloper.destination.database import DatabaseDestination, PartitionFilter
|
|
16
|
+
from interloper.errors import ConfigError, DataNotFoundError
|
|
17
|
+
from interloper.partitioning import PartitionConfig
|
|
18
|
+
from interloper.partitioning.base import Partition
|
|
19
|
+
from interloper.partitioning.time import TimePartition
|
|
20
|
+
from interloper.representation import Representation
|
|
21
|
+
from interloper.resource.fields import InputField
|
|
22
|
+
from interloper.schema import FieldSpec
|
|
23
|
+
from interloper.utils.data import is_empty
|
|
24
|
+
from interloper.utils.json import json_default, replace_non_finite
|
|
25
|
+
|
|
26
|
+
from interloper_clickhouse.connection import ClickHouseConnection
|
|
27
|
+
from interloper_clickhouse.partitioning import partition_key, same_key
|
|
28
|
+
from interloper_clickhouse.types import STRING, base_type, column_type, is_json, is_nullable
|
|
29
|
+
|
|
30
|
+
DEFAULT_DATABASE = "default"
|
|
31
|
+
|
|
32
|
+
_STAGING_PREFIX = "_interloper_staging_"
|
|
33
|
+
|
|
34
|
+
# A window's batch spans one partition per period; ClickHouse refuses an
|
|
35
|
+
# insert touching more than 100 partitions unless this is lifted (0 = no limit).
|
|
36
|
+
_STAGING_SETTINGS = {"max_partitions_per_insert_block": 0}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@destination(
|
|
40
|
+
key="clickhouse_destination",
|
|
41
|
+
name="ClickHouse",
|
|
42
|
+
icon="icon:clickhouse",
|
|
43
|
+
tags=["Database"],
|
|
44
|
+
)
|
|
45
|
+
class ClickHouseDestination(DatabaseDestination):
|
|
46
|
+
"""ClickHouse destination.
|
|
47
|
+
|
|
48
|
+
A dataset is a ClickHouse database: the asset's dataset, else
|
|
49
|
+
``default_dataset``, else ``default``. A table is created on first write
|
|
50
|
+
as a ``MergeTree`` partitioned so that one interloper partition is exactly
|
|
51
|
+
one ClickHouse partition (see :func:`~interloper_clickhouse.partitioning.partition_key`),
|
|
52
|
+
and is never altered afterwards.
|
|
53
|
+
|
|
54
|
+
ClickHouse has no multi-statement transactions, so the base's
|
|
55
|
+
delete-then-insert inside :meth:`transaction` would not be atomic, and a
|
|
56
|
+
``DELETE`` is a mutation rather than a cheap row operation. :meth:`write`
|
|
57
|
+
and :meth:`write_partition` are therefore overridden: the batch is loaded
|
|
58
|
+
into a staging table created ``AS`` the target, and each partition the
|
|
59
|
+
write covers is swapped in with ``ALTER TABLE ... REPLACE PARTITION``,
|
|
60
|
+
which is atomic per partition, or dropped when the batch holds no rows for
|
|
61
|
+
it. The staging table is dropped afterwards, also on failure.
|
|
62
|
+
:meth:`select`, :meth:`count` and :meth:`delete` remain the hooks the base
|
|
63
|
+
reads and deletes through.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
connection: ClickHouseConnection
|
|
67
|
+
|
|
68
|
+
default_dataset: str | None = InputField(
|
|
69
|
+
default=None, description="Default database for assets without a dataset; 'default' when empty"
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
# -- Destination interface ---------------------------------------------------
|
|
73
|
+
|
|
74
|
+
def write(self, context: IOContext, data: Any) -> None:
|
|
75
|
+
"""Replace every partition the context covers with the data, a window as one batch.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
context: IO context carrying the target asset, the partition or window,
|
|
79
|
+
and the effective schema.
|
|
80
|
+
data: The data to write, in its native representation.
|
|
81
|
+
"""
|
|
82
|
+
if is_empty(data):
|
|
83
|
+
return
|
|
84
|
+
self._warn_missing_partition_column(data, context)
|
|
85
|
+
self._replace(context, context.partitions, data)
|
|
86
|
+
|
|
87
|
+
def write_partition(self, context: IOContext, partition: Partition | None, data: Any) -> None:
|
|
88
|
+
"""Replace one partition, or the whole table, with the data.
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
context: IO context carrying the target asset and the effective schema.
|
|
92
|
+
partition: The partition being stored, or ``None`` for the whole table.
|
|
93
|
+
data: The data to store, in its native representation.
|
|
94
|
+
"""
|
|
95
|
+
self._replace(context, [partition], data)
|
|
96
|
+
|
|
97
|
+
# -- Naming ------------------------------------------------------------------
|
|
98
|
+
|
|
99
|
+
def _database(self, dataset: str | None) -> str:
|
|
100
|
+
"""Return the database a dataset resolves to.
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
The database name.
|
|
107
|
+
"""
|
|
108
|
+
return dataset or self.default_dataset or DEFAULT_DATABASE
|
|
109
|
+
|
|
110
|
+
@staticmethod
|
|
111
|
+
def _ref(table: str, database: str) -> str:
|
|
112
|
+
"""Build the quoted, database-qualified table reference.
|
|
113
|
+
|
|
114
|
+
Args:
|
|
115
|
+
table: Table name.
|
|
116
|
+
database: The resolved database name.
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
The database and the table, each backtick-quoted, joined by a dot.
|
|
120
|
+
"""
|
|
121
|
+
return f"{_quote(database)}.{_quote(table)}"
|
|
122
|
+
|
|
123
|
+
# -- Catalog -----------------------------------------------------------------
|
|
124
|
+
|
|
125
|
+
def _columns(self, table: str, database: str) -> dict[str, str]:
|
|
126
|
+
"""Read a table's columns and their types.
|
|
127
|
+
|
|
128
|
+
Args:
|
|
129
|
+
table: Table name.
|
|
130
|
+
database: The resolved database name.
|
|
131
|
+
|
|
132
|
+
Returns:
|
|
133
|
+
Column name to ClickHouse type in table order, empty when the table does not exist.
|
|
134
|
+
"""
|
|
135
|
+
result = self.connection.client.query(
|
|
136
|
+
"SELECT name, type FROM system.columns WHERE database = {database:String} AND table = {table:String} "
|
|
137
|
+
"ORDER BY position",
|
|
138
|
+
parameters={"database": database, "table": table},
|
|
139
|
+
)
|
|
140
|
+
return {name: type_name for name, type_name in result.result_rows}
|
|
141
|
+
|
|
142
|
+
def _existing(self, table: str, database: str) -> dict[str, str]:
|
|
143
|
+
"""Read the columns of a table that must exist.
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
table: Table name.
|
|
147
|
+
database: The resolved database name.
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
Column name to ClickHouse type in table order.
|
|
151
|
+
|
|
152
|
+
Raises:
|
|
153
|
+
DataNotFoundError: If the table does not exist yet.
|
|
154
|
+
"""
|
|
155
|
+
columns = self._columns(table, database)
|
|
156
|
+
if not columns:
|
|
157
|
+
raise DataNotFoundError(f"Table '{database}.{table}' does not exist. Has the asset been materialized?")
|
|
158
|
+
return columns
|
|
159
|
+
|
|
160
|
+
def _ensure_table(
|
|
161
|
+
self, table: str, database: str, specs: Sequence[FieldSpec], config: PartitionConfig | None
|
|
162
|
+
) -> dict[str, str]:
|
|
163
|
+
"""Create the database and the table when the table does not exist, and check its partition key.
|
|
164
|
+
|
|
165
|
+
Columns are ``Nullable`` when their field is, except the partition
|
|
166
|
+
column: ClickHouse keeps ``Nullable`` columns out of partition and
|
|
167
|
+
sorting keys, and a row without a partition value belongs to no
|
|
168
|
+
partition. An existing table is checked rather than altered: a
|
|
169
|
+
partition key other than the one the asset's partitioning needs would
|
|
170
|
+
make a partition replace touch the wrong rows.
|
|
171
|
+
|
|
172
|
+
Args:
|
|
173
|
+
table: Table name.
|
|
174
|
+
database: The resolved database name.
|
|
175
|
+
specs: The table's field specs.
|
|
176
|
+
config: The asset's partitioning, or ``None``.
|
|
177
|
+
|
|
178
|
+
Returns:
|
|
179
|
+
Column name to ClickHouse type in table order.
|
|
180
|
+
|
|
181
|
+
Raises:
|
|
182
|
+
ConfigError: If the partition column is not in the schema, or an
|
|
183
|
+
existing table is partitioned differently.
|
|
184
|
+
"""
|
|
185
|
+
columns = self._columns(table, database)
|
|
186
|
+
if not columns:
|
|
187
|
+
self._create_table(table, database, specs, config)
|
|
188
|
+
columns = self._columns(table, database)
|
|
189
|
+
if config is not None and config.column not in columns:
|
|
190
|
+
raise ConfigError(f"Partition column '{config.column}' is not a column of '{database}.{table}'.")
|
|
191
|
+
expected = None if config is None else partition_key(config, _quote(config.column), columns[config.column])
|
|
192
|
+
result = self.connection.client.query(
|
|
193
|
+
"SELECT partition_key FROM system.tables WHERE database = {database:String} AND name = {table:String}",
|
|
194
|
+
parameters={"database": database, "table": table},
|
|
195
|
+
)
|
|
196
|
+
actual = result.result_rows[0][0]
|
|
197
|
+
if not same_key(actual, expected):
|
|
198
|
+
raise ConfigError(
|
|
199
|
+
f"Table '{database}.{table}' is partitioned by '{actual or 'nothing'}', but the asset's "
|
|
200
|
+
f"partitioning needs '{expected or 'nothing'}'. Recreate the table, or partition the asset to match."
|
|
201
|
+
)
|
|
202
|
+
return columns
|
|
203
|
+
|
|
204
|
+
def _create_table(
|
|
205
|
+
self, table: str, database: str, specs: Sequence[FieldSpec], config: PartitionConfig | None
|
|
206
|
+
) -> None:
|
|
207
|
+
"""Create the database and a typed ``MergeTree`` table, unless they already exist.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
table: Table name.
|
|
211
|
+
database: The resolved database name.
|
|
212
|
+
specs: The table's field specs.
|
|
213
|
+
config: The asset's partitioning, or ``None``.
|
|
214
|
+
|
|
215
|
+
Raises:
|
|
216
|
+
ConfigError: If the partition column is not in the schema.
|
|
217
|
+
"""
|
|
218
|
+
column = None if config is None else config.column
|
|
219
|
+
if column is not None and column not in {spec.name for spec in specs}:
|
|
220
|
+
raise ConfigError(f"Partition column '{column}' is not in the schema of '{database}.{table}'.")
|
|
221
|
+
definitions = []
|
|
222
|
+
key = None
|
|
223
|
+
for spec in specs:
|
|
224
|
+
type_name = column_type(spec)
|
|
225
|
+
if spec.name == column:
|
|
226
|
+
key = partition_key(config, _quote(column), type_name)
|
|
227
|
+
elif is_nullable(spec):
|
|
228
|
+
type_name = f"Nullable({type_name})"
|
|
229
|
+
definitions.append(f"{_quote(spec.name)} {type_name}")
|
|
230
|
+
layout = "ENGINE = MergeTree"
|
|
231
|
+
if key is not None and column is not None:
|
|
232
|
+
layout += f" PARTITION BY {key} ORDER BY {_quote(column)}"
|
|
233
|
+
else:
|
|
234
|
+
layout += " ORDER BY tuple()"
|
|
235
|
+
client = self.connection.client
|
|
236
|
+
client.command(f"CREATE DATABASE IF NOT EXISTS {_quote(database)}")
|
|
237
|
+
client.command(f"CREATE TABLE IF NOT EXISTS {self._ref(table, database)} ({', '.join(definitions)}) {layout}")
|
|
238
|
+
|
|
239
|
+
# -- Loading -----------------------------------------------------------------
|
|
240
|
+
|
|
241
|
+
@staticmethod
|
|
242
|
+
def _specs(data: Any, context: IOContext) -> list[FieldSpec]:
|
|
243
|
+
"""Return the field specs a table is typed from.
|
|
244
|
+
|
|
245
|
+
Args:
|
|
246
|
+
data: The data being written.
|
|
247
|
+
context: IO context carrying the effective schema.
|
|
248
|
+
|
|
249
|
+
Returns:
|
|
250
|
+
The effective schema's specs, or those of a schema inferred from the data.
|
|
251
|
+
"""
|
|
252
|
+
return list((context.schema or Representation.of(data).infer()).field_specs())
|
|
253
|
+
|
|
254
|
+
def _load(
|
|
255
|
+
self,
|
|
256
|
+
ref: str,
|
|
257
|
+
columns: dict[str, str],
|
|
258
|
+
specs: Sequence[FieldSpec],
|
|
259
|
+
data: Any,
|
|
260
|
+
settings: dict[str, Any] | None = None,
|
|
261
|
+
) -> None:
|
|
262
|
+
"""Insert data into a table in one ``insert_df``, aligned to the table's columns.
|
|
263
|
+
|
|
264
|
+
A column the table does not have is dropped with a warning, since an
|
|
265
|
+
existing table is never altered. A field stored as JSON text is
|
|
266
|
+
encoded here; every other value is sent as conformed.
|
|
267
|
+
|
|
268
|
+
Args:
|
|
269
|
+
ref: The quoted table reference to insert into.
|
|
270
|
+
columns: The table's column name to ClickHouse type.
|
|
271
|
+
specs: The field specs the data was conformed to.
|
|
272
|
+
data: The data in its native representation.
|
|
273
|
+
settings: ClickHouse settings for the insert; defaults to none.
|
|
274
|
+
"""
|
|
275
|
+
frame = Representation.of(data).to("dataframe")
|
|
276
|
+
extras = [str(c) for c in frame.columns if str(c) not in columns]
|
|
277
|
+
if extras:
|
|
278
|
+
warnings.warn(
|
|
279
|
+
f"Columns {extras} are not in the schema for '{ref}' and will not be written.",
|
|
280
|
+
UserWarning,
|
|
281
|
+
stacklevel=4,
|
|
282
|
+
)
|
|
283
|
+
present = [c for c in columns if c in frame.columns]
|
|
284
|
+
if not present or frame.empty:
|
|
285
|
+
return
|
|
286
|
+
frame = frame[present]
|
|
287
|
+
encoded = [
|
|
288
|
+
spec.name
|
|
289
|
+
for spec in specs
|
|
290
|
+
if spec.name in present and is_json(spec) and base_type(columns[spec.name]) == STRING
|
|
291
|
+
]
|
|
292
|
+
if encoded:
|
|
293
|
+
frame = frame.assign(**{name: frame[name].map(_to_json) for name in encoded})
|
|
294
|
+
self.connection.client.insert_df(
|
|
295
|
+
table=ref,
|
|
296
|
+
df=frame,
|
|
297
|
+
column_names=present,
|
|
298
|
+
column_type_names=[columns[c] for c in present],
|
|
299
|
+
settings=settings,
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
# -- Replacing ---------------------------------------------------------------
|
|
303
|
+
|
|
304
|
+
def _replace(self, context: IOContext, partitions: Sequence[Partition | None], data: Any) -> None:
|
|
305
|
+
"""Replace partitions of the target with the data, through a staging table.
|
|
306
|
+
|
|
307
|
+
The data goes into a fresh staging table created ``AS`` the target, so
|
|
308
|
+
it shares its structure and partition key, as ``REPLACE PARTITION``
|
|
309
|
+
requires. Each partition then moves in with ``REPLACE PARTITION``,
|
|
310
|
+
atomic per partition; one the staging table holds no rows for is
|
|
311
|
+
dropped instead, since newer servers refuse to replace from an empty
|
|
312
|
+
partition. The staging table is dropped whatever happens.
|
|
313
|
+
|
|
314
|
+
Args:
|
|
315
|
+
context: IO context carrying the target asset and the effective schema.
|
|
316
|
+
partitions: The partitions to replace, ``[None]`` for the whole table.
|
|
317
|
+
data: The data to write, in its native representation.
|
|
318
|
+
"""
|
|
319
|
+
table, dataset = self._target(context)
|
|
320
|
+
database = self._database(dataset)
|
|
321
|
+
config = context.asset.partitioning
|
|
322
|
+
specs = self._specs(data, context)
|
|
323
|
+
columns = self._ensure_table(table, database, specs, config)
|
|
324
|
+
target = self._ref(table, database)
|
|
325
|
+
staging = self._ref(f"{_STAGING_PREFIX}{table}_{uuid.uuid4().hex[:16]}", database)
|
|
326
|
+
client = self.connection.client
|
|
327
|
+
client.command(f"CREATE TABLE {staging} AS {target}")
|
|
328
|
+
try:
|
|
329
|
+
self._load(staging, columns, specs, data, settings=_STAGING_SETTINGS)
|
|
330
|
+
loaded = self._loaded_ids(staging)
|
|
331
|
+
for clause, has_rows in self._clauses(target, config, columns, partitions, loaded):
|
|
332
|
+
if has_rows:
|
|
333
|
+
client.command(f"ALTER TABLE {target} REPLACE PARTITION {clause} FROM {staging}")
|
|
334
|
+
else:
|
|
335
|
+
client.command(f"ALTER TABLE {target} DROP PARTITION {clause}")
|
|
336
|
+
finally:
|
|
337
|
+
client.command(f"DROP TABLE IF EXISTS {staging} SYNC")
|
|
338
|
+
|
|
339
|
+
def _loaded_ids(self, ref: str) -> set[str]:
|
|
340
|
+
"""Read the ids of the partitions a table holds rows in.
|
|
341
|
+
|
|
342
|
+
Args:
|
|
343
|
+
ref: The quoted table reference.
|
|
344
|
+
|
|
345
|
+
Returns:
|
|
346
|
+
The partition ids, from the ``_partition_id`` virtual column.
|
|
347
|
+
"""
|
|
348
|
+
result = self.connection.client.query(f"SELECT DISTINCT _partition_id FROM {ref}")
|
|
349
|
+
return {partition_id for (partition_id,) in result.result_rows}
|
|
350
|
+
|
|
351
|
+
def _clauses(
|
|
352
|
+
self,
|
|
353
|
+
target: str,
|
|
354
|
+
config: PartitionConfig | None,
|
|
355
|
+
columns: dict[str, str],
|
|
356
|
+
partitions: Sequence[Partition | None],
|
|
357
|
+
loaded: set[str],
|
|
358
|
+
) -> list[tuple[str, bool]]:
|
|
359
|
+
"""Name the target partitions a write replaces, and whether the staging table holds rows for each.
|
|
360
|
+
|
|
361
|
+
An unpartitioned table is the single partition ``tuple()``. A whole
|
|
362
|
+
write to a partitioned table covers every partition the target or the
|
|
363
|
+
staging table holds. Otherwise the write covers its partitions, and
|
|
364
|
+
rows the staging table holds outside them are not moved, with a
|
|
365
|
+
warning.
|
|
366
|
+
|
|
367
|
+
Args:
|
|
368
|
+
target: The quoted target table reference.
|
|
369
|
+
config: The asset's partitioning, or ``None``.
|
|
370
|
+
columns: The target's column name to ClickHouse type.
|
|
371
|
+
partitions: The partitions to replace, ``[None]`` for the whole table.
|
|
372
|
+
loaded: The ids of the partitions the staging table holds rows in.
|
|
373
|
+
|
|
374
|
+
Returns:
|
|
375
|
+
``(PARTITION clause, has rows)`` pairs, in write order.
|
|
376
|
+
"""
|
|
377
|
+
if config is None:
|
|
378
|
+
return [("tuple()", bool(loaded))]
|
|
379
|
+
if list(partitions) == [None]:
|
|
380
|
+
ids = sorted(loaded | self._loaded_ids(target))
|
|
381
|
+
else:
|
|
382
|
+
ids = self._partition_ids(config, columns[config.column], partitions)
|
|
383
|
+
stray = loaded - set(ids)
|
|
384
|
+
if stray:
|
|
385
|
+
warnings.warn(
|
|
386
|
+
f"The data holds rows in {len(stray)} partition(s) outside the ones being written to "
|
|
387
|
+
f"'{target}'; those rows were not written.",
|
|
388
|
+
UserWarning,
|
|
389
|
+
stacklevel=4,
|
|
390
|
+
)
|
|
391
|
+
return [(f"ID {_literal(partition_id)}", partition_id in loaded) for partition_id in ids]
|
|
392
|
+
|
|
393
|
+
def _partition_ids(
|
|
394
|
+
self, config: PartitionConfig, column_type_name: str, partitions: Sequence[Partition | None]
|
|
395
|
+
) -> list[str]:
|
|
396
|
+
"""Compute the ClickHouse partition id of each interloper partition.
|
|
397
|
+
|
|
398
|
+
ClickHouse computes them itself (``partitionId``) over the table's own
|
|
399
|
+
partition key applied to each partition's value, so the ids match the
|
|
400
|
+
ones the parts carry whatever the key's type.
|
|
401
|
+
|
|
402
|
+
Args:
|
|
403
|
+
config: The asset's partitioning.
|
|
404
|
+
column_type_name: The partition column's ClickHouse type.
|
|
405
|
+
partitions: The partitions to identify.
|
|
406
|
+
|
|
407
|
+
Returns:
|
|
408
|
+
One id per distinct partition, in order.
|
|
409
|
+
"""
|
|
410
|
+
values = [_partition_value(p) for p in partitions if p is not None]
|
|
411
|
+
key = partition_key(config, f"CAST(v AS {column_type_name})", column_type_name)
|
|
412
|
+
result = self.connection.client.query(
|
|
413
|
+
f"SELECT arrayMap(v -> partitionId({key}), {{values:Array(String)}})",
|
|
414
|
+
parameters={"values": values},
|
|
415
|
+
)
|
|
416
|
+
return list(dict.fromkeys(result.result_rows[0][0]))
|
|
417
|
+
|
|
418
|
+
# -- DatabaseDestination hooks -------------------------------------------------
|
|
419
|
+
|
|
420
|
+
def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
|
|
421
|
+
"""Append data to the table, creating it on first write.
|
|
422
|
+
|
|
423
|
+
:meth:`write` replaces through a staging table instead; this hook adds
|
|
424
|
+
rows without removing any.
|
|
425
|
+
|
|
426
|
+
Args:
|
|
427
|
+
table: Target table name.
|
|
428
|
+
dataset: The database, or ``None`` for the destination's default.
|
|
429
|
+
data: The data in its native representation.
|
|
430
|
+
context: IO context carrying the asset and effective schema.
|
|
431
|
+
"""
|
|
432
|
+
database = self._database(dataset)
|
|
433
|
+
specs = self._specs(data, context)
|
|
434
|
+
columns = self._ensure_table(table, database, specs, context.asset.partitioning)
|
|
435
|
+
self._load(self._ref(table, database), columns, specs, data)
|
|
436
|
+
|
|
437
|
+
def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
|
|
438
|
+
"""Delete the rows a filter selects with a lightweight ``DELETE``, or truncate the table.
|
|
439
|
+
|
|
440
|
+
:meth:`write` never deletes row by row; this hook serves callers that
|
|
441
|
+
remove rows outside a write. A table that does not exist has nothing
|
|
442
|
+
to delete.
|
|
443
|
+
|
|
444
|
+
Args:
|
|
445
|
+
table: Target table name.
|
|
446
|
+
dataset: The database, or ``None`` for the destination's default.
|
|
447
|
+
where: The rows to delete; ``None`` for the whole table.
|
|
448
|
+
"""
|
|
449
|
+
database = self._database(dataset)
|
|
450
|
+
columns = self._columns(table, database)
|
|
451
|
+
if not columns:
|
|
452
|
+
return
|
|
453
|
+
ref = self._ref(table, database)
|
|
454
|
+
if where is None:
|
|
455
|
+
self.connection.client.command(f"TRUNCATE TABLE {ref}")
|
|
456
|
+
return
|
|
457
|
+
predicate, parameters = _predicate(where, columns)
|
|
458
|
+
self.connection.client.command(f"DELETE FROM {ref} WHERE {predicate}", parameters=parameters)
|
|
459
|
+
|
|
460
|
+
def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
|
|
461
|
+
"""Select the rows a filter selects, or every row, as a DataFrame.
|
|
462
|
+
|
|
463
|
+
Args:
|
|
464
|
+
table: Target table name.
|
|
465
|
+
dataset: The database, or ``None`` for the destination's default.
|
|
466
|
+
where: The rows to select; ``None`` for the whole table.
|
|
467
|
+
|
|
468
|
+
Returns:
|
|
469
|
+
The selected rows.
|
|
470
|
+
"""
|
|
471
|
+
database = self._database(dataset)
|
|
472
|
+
columns = self._existing(table, database)
|
|
473
|
+
ref = self._ref(table, database)
|
|
474
|
+
if where is None:
|
|
475
|
+
return self.connection.client.query_df(f"SELECT * FROM {ref}")
|
|
476
|
+
predicate, parameters = _predicate(where, columns)
|
|
477
|
+
return self.connection.client.query_df(f"SELECT * FROM {ref} WHERE {predicate}", parameters=parameters)
|
|
478
|
+
|
|
479
|
+
def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
|
|
480
|
+
"""Return row counts grouped by a column.
|
|
481
|
+
|
|
482
|
+
Args:
|
|
483
|
+
table: Target table name.
|
|
484
|
+
dataset: The database, or ``None`` for the destination's default.
|
|
485
|
+
column: Column to group by.
|
|
486
|
+
|
|
487
|
+
Returns:
|
|
488
|
+
Mapping from the column's value (as a string) to its row count.
|
|
489
|
+
"""
|
|
490
|
+
database = self._database(dataset)
|
|
491
|
+
self._existing(table, database)
|
|
492
|
+
result = self.connection.client.query(
|
|
493
|
+
f"SELECT toString({_quote(column)}) AS partition_value, count() AS cnt "
|
|
494
|
+
f"FROM {self._ref(table, database)} GROUP BY partition_value"
|
|
495
|
+
)
|
|
496
|
+
return {value: count for value, count in result.result_rows}
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
# -- Utility functions -------------------------------------------------------------
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _quote(identifier: str) -> str:
|
|
503
|
+
"""Quote an identifier for ClickHouse, escaping backslashes and backticks.
|
|
504
|
+
|
|
505
|
+
Args:
|
|
506
|
+
identifier: A database, table or column name.
|
|
507
|
+
|
|
508
|
+
Returns:
|
|
509
|
+
The backtick-quoted identifier.
|
|
510
|
+
"""
|
|
511
|
+
escaped = identifier.replace("\\", "\\\\").replace("`", "\\`")
|
|
512
|
+
return f"`{escaped}`"
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def _literal(value: str) -> str:
|
|
516
|
+
"""Render a string literal for ClickHouse, escaping backslashes and quotes.
|
|
517
|
+
|
|
518
|
+
Args:
|
|
519
|
+
value: The string.
|
|
520
|
+
|
|
521
|
+
Returns:
|
|
522
|
+
The single-quoted literal.
|
|
523
|
+
"""
|
|
524
|
+
escaped = value.replace("\\", "\\\\").replace("'", "\\'")
|
|
525
|
+
return f"'{escaped}'"
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _partition_value(partition: Partition) -> str:
|
|
529
|
+
"""Render the value a partition's rows carry in its partition column, as text ClickHouse casts.
|
|
530
|
+
|
|
531
|
+
Args:
|
|
532
|
+
partition: The partition.
|
|
533
|
+
|
|
534
|
+
Returns:
|
|
535
|
+
A time partition's period start, a datetime with a space separator;
|
|
536
|
+
any other partition's id.
|
|
537
|
+
"""
|
|
538
|
+
if isinstance(partition, TimePartition):
|
|
539
|
+
value = partition.value
|
|
540
|
+
if isinstance(value, datetime.datetime):
|
|
541
|
+
return value.strftime("%Y-%m-%d %H:%M:%S")
|
|
542
|
+
return value.isoformat()
|
|
543
|
+
return partition.id
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _predicate(where: PartitionFilter, columns: dict[str, str]) -> tuple[str, dict[str, Any]]:
|
|
547
|
+
"""Render a partition filter as a predicate with server-side bound parameters.
|
|
548
|
+
|
|
549
|
+
Each parameter is typed as the column, so ClickHouse parses a partition id
|
|
550
|
+
that arrives as text into a ``Date32`` or an ``Int64``. Against a
|
|
551
|
+
``String`` column, date and datetime bounds are rendered in ISO 8601, the
|
|
552
|
+
form the rows carry.
|
|
553
|
+
|
|
554
|
+
Args:
|
|
555
|
+
where: The filter to render.
|
|
556
|
+
columns: The table's column name to ClickHouse type.
|
|
557
|
+
|
|
558
|
+
Returns:
|
|
559
|
+
The predicate text and its parameters.
|
|
560
|
+
"""
|
|
561
|
+
column = _quote(where.column)
|
|
562
|
+
type_name = columns.get(where.column, STRING)
|
|
563
|
+
if where.bounds is None:
|
|
564
|
+
return f"{column} = {{value:{type_name}}}", {"value": where.value}
|
|
565
|
+
start, end = where.bounds
|
|
566
|
+
if base_type(type_name) == STRING:
|
|
567
|
+
start, end = (_iso(bound) for bound in (start, end))
|
|
568
|
+
return f"{column} >= {{start:{type_name}}} AND {column} < {{end:{type_name}}}", {"start": start, "end": end}
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def _iso(value: Any) -> Any:
|
|
572
|
+
"""Render a date or datetime as ISO 8601, leaving any other value as is.
|
|
573
|
+
|
|
574
|
+
Args:
|
|
575
|
+
value: A partition bound.
|
|
576
|
+
|
|
577
|
+
Returns:
|
|
578
|
+
The ISO string for a date or datetime, else *value*.
|
|
579
|
+
"""
|
|
580
|
+
return value.isoformat() if isinstance(value, datetime.date) else value
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def _to_json(value: Any) -> Any:
|
|
584
|
+
"""Encode a value as JSON text for a ``String`` column.
|
|
585
|
+
|
|
586
|
+
Args:
|
|
587
|
+
value: A cell of a field stored as JSON text.
|
|
588
|
+
|
|
589
|
+
Returns:
|
|
590
|
+
``None`` for a missing value, text and bytes unchanged, the JSON
|
|
591
|
+
encoding of anything else.
|
|
592
|
+
"""
|
|
593
|
+
if value is None or (isinstance(value, float) and math.isnan(value)):
|
|
594
|
+
return None
|
|
595
|
+
if isinstance(value, (str, bytes)):
|
|
596
|
+
return value
|
|
597
|
+
if hasattr(value, "tolist"):
|
|
598
|
+
value = value.tolist()
|
|
599
|
+
return json.dumps(replace_non_finite(value), default=json_default)
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""ClickHouse partition keys that line up with interloper partitions."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from interloper.errors import ConfigError
|
|
6
|
+
from interloper.partitioning import PartitionConfig, TimeGranularity, TimePartitionConfig
|
|
7
|
+
|
|
8
|
+
from interloper_clickhouse.types import base_type
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def partition_key(config: PartitionConfig | None, ref: str, column_type: str) -> str | None:
|
|
12
|
+
"""Return the ``PARTITION BY`` expression that makes one interloper partition one ClickHouse partition.
|
|
13
|
+
|
|
14
|
+
A time partition covers a period, so its rows share the period start the
|
|
15
|
+
expression computes: the day (the ``Date`` column itself, ``toDate`` of a
|
|
16
|
+
``DateTime``), ``toStartOfMonth``, ``toStartOfYear`` or ``toStartOfHour``.
|
|
17
|
+
A ``String`` column is parsed first, so ISO dates held as text partition
|
|
18
|
+
the same way. Any other partition is one value of the column.
|
|
19
|
+
|
|
20
|
+
The expression is rendered over *ref*, so the same function renders the
|
|
21
|
+
table's key (over the quoted column) and the key of a constant (over a
|
|
22
|
+
``CAST`` of it), which is how a partition's id is computed.
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
config: The asset's partitioning, or ``None`` for an unpartitioned asset.
|
|
26
|
+
ref: The SQL the expression applies to: a quoted column, or a constant.
|
|
27
|
+
column_type: The partition column's ClickHouse type.
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
The expression, or ``None`` for an unpartitioned asset.
|
|
31
|
+
|
|
32
|
+
Raises:
|
|
33
|
+
ConfigError: If the column's type cannot carry the asset's time
|
|
34
|
+
partitioning: hourly partitions on a ``Date``, or a time partition
|
|
35
|
+
on a column that is neither a date, a datetime nor text.
|
|
36
|
+
"""
|
|
37
|
+
if config is None:
|
|
38
|
+
return None
|
|
39
|
+
if not isinstance(config, TimePartitionConfig):
|
|
40
|
+
return ref
|
|
41
|
+
granularity = config.granularity
|
|
42
|
+
kind = base_type(column_type)
|
|
43
|
+
if kind in ("Date", "Date32"):
|
|
44
|
+
if granularity is TimeGranularity.HOUR:
|
|
45
|
+
raise ConfigError(
|
|
46
|
+
f"Partition column '{config.column}' is a {kind}, which cannot hold hourly partitions; "
|
|
47
|
+
"declare it as a datetime."
|
|
48
|
+
)
|
|
49
|
+
moment, day = ref, ref
|
|
50
|
+
elif kind.startswith("DateTime"):
|
|
51
|
+
moment, day = ref, f"toDate({ref})"
|
|
52
|
+
elif kind == "String":
|
|
53
|
+
moment = f"parseDateTime64BestEffort({ref}, 6, 'UTC')"
|
|
54
|
+
day = f"toDate({moment})"
|
|
55
|
+
else:
|
|
56
|
+
raise ConfigError(
|
|
57
|
+
f"Partition column '{config.column}' is a {kind}; time partitioning needs a date, a datetime or text."
|
|
58
|
+
)
|
|
59
|
+
return {
|
|
60
|
+
TimeGranularity.HOUR: f"toStartOfHour({moment})",
|
|
61
|
+
TimeGranularity.DAY: day,
|
|
62
|
+
TimeGranularity.MONTH: f"toStartOfMonth({moment})",
|
|
63
|
+
TimeGranularity.YEAR: f"toStartOfYear({moment})",
|
|
64
|
+
}[granularity]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def same_key(actual: str, expected: str | None) -> bool:
|
|
68
|
+
"""Compare a table's partition key with the one the asset needs, ignoring quoting and spacing.
|
|
69
|
+
|
|
70
|
+
``system.tables`` reports the key as ClickHouse formats it, without the
|
|
71
|
+
backticks this package renders and with its own spacing.
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
actual: The table's ``partition_key``, empty for an unpartitioned table.
|
|
75
|
+
expected: The expression :func:`partition_key` renders, or ``None``.
|
|
76
|
+
|
|
77
|
+
Returns:
|
|
78
|
+
True when both name the same expression.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
def normalize(expression: str) -> str:
|
|
82
|
+
"""Drop backticks and whitespace.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
expression: A partition key expression.
|
|
86
|
+
|
|
87
|
+
Returns:
|
|
88
|
+
The expression without backticks or whitespace.
|
|
89
|
+
"""
|
|
90
|
+
return "".join(expression.replace("`", "").split())
|
|
91
|
+
|
|
92
|
+
return normalize(actual) == normalize(expected or "")
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""ClickHouse's view of interloper's field types."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
from decimal import Decimal
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from interloper.schema import FieldSpec
|
|
10
|
+
|
|
11
|
+
STRING = "String"
|
|
12
|
+
|
|
13
|
+
# Ordered: the first base class that matches wins, so bool (a subclass of int)
|
|
14
|
+
# and datetime (a subclass of date) must come before their parents.
|
|
15
|
+
_PYTHON_TO_CLICKHOUSE: dict[type, str] = {
|
|
16
|
+
bool: "Bool",
|
|
17
|
+
int: "Int64",
|
|
18
|
+
float: "Float64",
|
|
19
|
+
Decimal: "Decimal(38, 9)",
|
|
20
|
+
datetime.datetime: "DateTime64(6, 'UTC')",
|
|
21
|
+
datetime.date: "Date32",
|
|
22
|
+
bytes: STRING,
|
|
23
|
+
str: STRING,
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def is_json(spec: FieldSpec) -> bool:
|
|
28
|
+
"""Return whether a field is stored as JSON text in a ``String`` column.
|
|
29
|
+
|
|
30
|
+
Nested models, repeated fields, plain ``dict``/``list`` fields and fields
|
|
31
|
+
of no concrete class (``typing.Any``) have no scalar ClickHouse type.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
True when the field's values are written as JSON text.
|
|
38
|
+
"""
|
|
39
|
+
if spec.fields is not None or spec.repeated or spec.type is Any or not isinstance(spec.type, type):
|
|
40
|
+
return True
|
|
41
|
+
return issubclass(spec.type, (dict, list))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def is_nullable(spec: FieldSpec) -> bool:
|
|
45
|
+
"""Return whether a field's column is ``Nullable``.
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
True for a nullable field, and for a ``typing.Any`` field, which admits ``None``.
|
|
52
|
+
"""
|
|
53
|
+
return spec.nullable or spec.type is Any
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def column_type(spec: FieldSpec) -> str:
|
|
57
|
+
"""Return the ClickHouse column type for a field spec, before nullability.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
spec: The field spec, from :meth:`Schema.field_specs` or an inferred schema.
|
|
61
|
+
|
|
62
|
+
Returns:
|
|
63
|
+
``String`` for a field stored as JSON text (see :func:`is_json`), the
|
|
64
|
+
type the spec's Python type maps to otherwise, and ``String`` for any
|
|
65
|
+
class that maps to none.
|
|
66
|
+
"""
|
|
67
|
+
if is_json(spec):
|
|
68
|
+
return STRING
|
|
69
|
+
for base, name in _PYTHON_TO_CLICKHOUSE.items():
|
|
70
|
+
if issubclass(spec.type, base):
|
|
71
|
+
return name
|
|
72
|
+
return STRING
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def base_type(type_name: str) -> str:
|
|
76
|
+
"""Strip a ``Nullable`` wrapper from a ClickHouse type name.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
type_name: A ClickHouse type name, as ``system.columns`` reports it.
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
The wrapped type for ``Nullable(T)``, the name unchanged otherwise.
|
|
83
|
+
"""
|
|
84
|
+
if type_name.startswith("Nullable(") and type_name.endswith(")"):
|
|
85
|
+
return type_name[len("Nullable(") : -1]
|
|
86
|
+
return type_name
|