smart-data-engine-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +318 -0
- sde/_cutover_project.py +179 -0
- sde/_local_state.py +188 -0
- sde/_operator_deadline.py +50 -0
- sde/_usage.py +314 -0
- sde/bulk.py +79 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/cutover.py +286 -0
- sde/engines/__init__.py +0 -0
- sde/engines/_clickhouse_connection.py +224 -0
- sde/engines/_index_build.py +294 -0
- sde/engines/_operator.py +394 -0
- sde/engines/_staging.py +222 -0
- sde/engines/_storage.py +22 -0
- sde/engines/_write_fences.py +271 -0
- sde/engines/clickhouse.py +1115 -0
- sde/engines/orderbook.py +457 -0
- sde/engines/postgres.py +967 -0
- sde/entity.py +170 -0
- sde/errors.py +103 -0
- sde/explain.py +300 -0
- sde/frozen_verification.py +152 -0
- sde/generation.py +131 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/index_build.py +313 -0
- sde/index_operator.py +347 -0
- sde/infer.py +461 -0
- sde/inspection.py +62 -0
- sde/internal.py +90 -0
- sde/layout.py +669 -0
- sde/local_cutover.py +801 -0
- sde/logging.py +143 -0
- sde/migration.py +856 -0
- sde/model.py +482 -0
- sde/physical.py +531 -0
- sde/placement.py +1010 -0
- sde/provisioning.py +63 -0
- sde/py.typed +0 -0
- sde/query.py +521 -0
- sde/routing.py +85 -0
- sde/schema.py +466 -0
- sde/session.py +993 -0
- sde/shapes.py +153 -0
- sde/staging.py +264 -0
- sde/staging_operator.py +393 -0
- sde/telemetry.py +1087 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +331 -0
- sde/types.py +228 -0
- sde/verification.py +220 -0
- sde/watermark.py +222 -0
- sde/write_fence.py +283 -0
- sde_demo/__init__.py +1 -0
- sde_demo/__main__.py +183 -0
- sde_demo/diagnostics.py +92 -0
- sde_demo/model.py +75 -0
- sde_demo/project.py +312 -0
- sde_demo/py.typed +0 -0
- sde_demo/query_count.py +301 -0
- sde_demo/resources.py +969 -0
- sde_demo/runtime.py +419 -0
- sde_demo/verification.py +242 -0
- sde_operator/__init__.py +1 -0
- sde_operator/__main__.py +210 -0
- smart_data_engine_sdk-0.1.0.dist-info/METADATA +174 -0
- smart_data_engine_sdk-0.1.0.dist-info/RECORD +73 -0
- smart_data_engine_sdk-0.1.0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dist-info/entry_points.txt +3 -0
- smart_data_engine_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dist-info/licenses/NOTICE +13 -0
|
@@ -0,0 +1,1115 @@
|
|
|
1
|
+
"""ClickHouse adapter.
|
|
2
|
+
|
|
3
|
+
The second engine, and the reason there is a product at all: between PostgreSQL and ClickHouse
|
|
4
|
+
lies the decision clients get wrong once, at the start, and never revisit. With one adapter the
|
|
5
|
+
planner chooses from a set of size one.
|
|
6
|
+
|
|
7
|
+
Three things here are genuinely different from the PostgreSQL adapter, and none of them is a
|
|
8
|
+
detail. Each is a place where "the same operation" means something else because the engine is
|
|
9
|
+
different, and the product's position is that such a difference must be *stated* rather than
|
|
10
|
+
smoothed over.
|
|
11
|
+
|
|
12
|
+
**A naive datetime is UTC. Always, explicitly, here.**
|
|
13
|
+
|
|
14
|
+
A fix for a divergence that was measured rather than imagined. `datetime(2026, 8, 27, 12, 0)` with
|
|
15
|
+
no tzinfo, written through this library:
|
|
16
|
+
|
|
17
|
+
- into PostgreSQL `timestamptz`, it arrives as ``12:00:00+00:00`` - into ClickHouse
|
|
18
|
+
``DateTime64(3, 'UTC')`` with the driver left to its own devices, it arrived as
|
|
19
|
+
``10:00:00`` - because `clickhouse-connect` reads a naive datetime as *local* time and converts
|
|
20
|
+
|
|
21
|
+
Two hours apart, from one call, in one library, with no error anywhere. After a migration from one
|
|
22
|
+
engine to the other every timestamp in the client's analytics would shift by the offset of
|
|
23
|
+
whichever machine happened to write the row. So naive datetimes are given `timezone.utc` before
|
|
24
|
+
they reach the driver, which is what the PostgreSQL path already does in effect. If you mean a
|
|
25
|
+
different zone, pass an aware datetime and it is respected.
|
|
26
|
+
|
|
27
|
+
**There are no transactions, and this adapter says so instead of pretending.**
|
|
28
|
+
|
|
29
|
+
`transaction()` raises. Not a no-op context manager: a caller who believes they are in a
|
|
30
|
+
transaction and is not has been lied to at the worst possible moment. A client who needs two
|
|
31
|
+
entities to commit together declares that, and the planner then places them in the same group and
|
|
32
|
+
therefore the same engine - the requirement becomes a placement constraint. This method existing
|
|
33
|
+
and refusing is how that constraint is discovered at the point of use rather than inferred from a
|
|
34
|
+
rollback that never happened.
|
|
35
|
+
|
|
36
|
+
**Keys are not enforced, so the table is a `ReplacingMergeTree` and point reads use `FINAL`.**
|
|
37
|
+
|
|
38
|
+
A MergeTree does not enforce uniqueness on its `ORDER BY`. So a second `save()` of the same key
|
|
39
|
+
does not raise the way PostgreSQL's primary key does - and that divergence cannot be removed, only
|
|
40
|
+
chosen. Plain `MergeTree` would leave two rows and make every aggregate over that group quietly
|
|
41
|
+
wrong; `ReplacingMergeTree` keeps the newest and collapses the rest at merge time. Of the two,
|
|
42
|
+
"the newest row for a key wins" is something a client can reason about. `FINAL` on point reads and
|
|
43
|
+
counts is what makes the collapse visible immediately rather than eventually, and it is not free -
|
|
44
|
+
which is the trade, stated.
|
|
45
|
+
|
|
46
|
+
The residue is worth naming plainly, because a client will meet it: in PostgreSQL, saving the same
|
|
47
|
+
key twice is an error; here it is an overwrite. That belongs in a placement decision rather than
|
|
48
|
+
in an adapter, and there is no declaration for it yet.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
from __future__ import annotations
|
|
52
|
+
|
|
53
|
+
import datetime as _dt
|
|
54
|
+
import math
|
|
55
|
+
import os
|
|
56
|
+
import re
|
|
57
|
+
import ssl
|
|
58
|
+
import stat
|
|
59
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
60
|
+
from importlib.metadata import version
|
|
61
|
+
from typing import Any, cast
|
|
62
|
+
|
|
63
|
+
from .._usage import UsageGate, guarded
|
|
64
|
+
from ..bulk import batch_columns
|
|
65
|
+
from ..errors import EngineError
|
|
66
|
+
from ..explain import (
|
|
67
|
+
Cost,
|
|
68
|
+
QueryPlan,
|
|
69
|
+
QueryPlanRefused,
|
|
70
|
+
replacing_merge_tree_finding,
|
|
71
|
+
)
|
|
72
|
+
from ..logging import log
|
|
73
|
+
from ..migration import key_columns, same_width
|
|
74
|
+
from ..physical import (
|
|
75
|
+
PhysicalFinding,
|
|
76
|
+
declared_tables,
|
|
77
|
+
parse_identifier_list,
|
|
78
|
+
parse_partition_key,
|
|
79
|
+
)
|
|
80
|
+
from ..placement import BACKFILL_TABLE, WATERMARK_TABLE, PhysicalLayout
|
|
81
|
+
from ..query import ReadColumn, ReadPlan, read_row, read_sql, summary_sql
|
|
82
|
+
from ..schema import QUOTE, schema_statements
|
|
83
|
+
from ..write_fence import WriteFence
|
|
84
|
+
from ._clickhouse_connection import (
|
|
85
|
+
CONNECT_TIMEOUT_SECONDS as CONNECT_TIMEOUT_SECONDS,
|
|
86
|
+
)
|
|
87
|
+
from ._clickhouse_connection import (
|
|
88
|
+
HANDSHAKE_TIMEOUT_SECONDS as HANDSHAKE_TIMEOUT_SECONDS,
|
|
89
|
+
)
|
|
90
|
+
from ._clickhouse_connection import (
|
|
91
|
+
ConnectionParameters,
|
|
92
|
+
parse_dsn,
|
|
93
|
+
)
|
|
94
|
+
from ._storage import STORAGE_COLUMNS
|
|
95
|
+
from ._write_fences import ClickHouseFences
|
|
96
|
+
|
|
97
|
+
__all__ = ["STORAGE_COLUMNS", "ClickHouseEngine"]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# Bound from the one definition in sde.schema, so that DDL and DML cannot disagree about
|
|
101
|
+
# how an identifier is escaped.
|
|
102
|
+
_quote = QUOTE["clickhouse"]
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _as_utc(value: Any) -> Any:
|
|
106
|
+
"""A naive datetime is UTC. See the module docstring for the measurement behind this."""
|
|
107
|
+
if isinstance(value, _dt.datetime) and value.tzinfo is None:
|
|
108
|
+
return value.replace(tzinfo=_dt.UTC)
|
|
109
|
+
return value
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _query_parameter(name: str, value: Any, parameters: dict[str, Any]) -> str:
|
|
113
|
+
"""Bind a scalar without letting the driver's datetime formatter discard its fraction.
|
|
114
|
+
|
|
115
|
+
The plain datetime formatter emits seconds. Use a bound string and an explicit UTC
|
|
116
|
+
DateTime64 conversion for temporal comparisons, independent of the driver's formatting zone.
|
|
117
|
+
Data values still go through the driver's parameter binding; none are interpolated into SQL.
|
|
118
|
+
"""
|
|
119
|
+
placeholder = f"%({name})s"
|
|
120
|
+
if isinstance(value, _dt.datetime):
|
|
121
|
+
utc = _as_utc(value).astimezone(_dt.UTC)
|
|
122
|
+
parameters[name] = utc.replace(tzinfo=None).isoformat(" ", timespec="microseconds")
|
|
123
|
+
return f"toDateTime64({placeholder}, 6, 'UTC')"
|
|
124
|
+
parameters[name] = value
|
|
125
|
+
return placeholder
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _row(names: Sequence[str], types: Sequence[Any], row: Sequence[Any]) -> dict[str, Any]:
|
|
129
|
+
"""One result row as a dict, with the timezone put back on values that had one in the schema.
|
|
130
|
+
|
|
131
|
+
The write side of this module makes a naive datetime UTC. Without this, the read side would
|
|
132
|
+
not give it back: `clickhouse-connect` returns a *naive* datetime for a `DateTime64(3, 'UTC')`
|
|
133
|
+
column, while psycopg returns an aware one for a `timestamptz`. So the same entity read from
|
|
134
|
+
the two engines produces two datetimes that Python refuses to compare - ``can't compare
|
|
135
|
+
offset-naive and offset-aware datetimes`` - which is a `TypeError` in a client's code that
|
|
136
|
+
appears on the day a group is moved and not before.
|
|
137
|
+
|
|
138
|
+
The column type carries the zone, so it is read from there rather than assumed: a column
|
|
139
|
+
declared with a timezone yields aware values, one declared without yields naive, which is
|
|
140
|
+
exactly the distinction the neutral vocabulary makes between `timestamptz` and `timestamp`.
|
|
141
|
+
"""
|
|
142
|
+
out: dict[str, Any] = {}
|
|
143
|
+
for name, column_type, value in zip(names, types, row, strict=True):
|
|
144
|
+
zone = getattr(column_type, "tzinfo", None)
|
|
145
|
+
if zone is not None and isinstance(value, _dt.datetime) and value.tzinfo is None:
|
|
146
|
+
value = value.replace(tzinfo=zone)
|
|
147
|
+
out[name] = value
|
|
148
|
+
return out
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class _NoReplayTransport:
|
|
152
|
+
"""Keep the driver's TLS/proxy pool but forbid ambiguous transport replay for this client.
|
|
153
|
+
|
|
154
|
+
clickhouse-connect retries RemoteDisconnected even when its retry count is zero. In a live
|
|
155
|
+
probe that error followed an accepted INSERT: the driver replayed it and returned success.
|
|
156
|
+
Translate transport errors before that retry handler, and disable urllib3's own retry layer.
|
|
157
|
+
The shared pool is never patched; only this client's reference is wrapped.
|
|
158
|
+
"""
|
|
159
|
+
|
|
160
|
+
def __init__(self, pool: Any) -> None:
|
|
161
|
+
self.pool = pool
|
|
162
|
+
|
|
163
|
+
def __getattr__(self, name: str) -> Any:
|
|
164
|
+
return getattr(self.pool, name)
|
|
165
|
+
|
|
166
|
+
# The driver keys pool-expiration bookkeeping by manager. Preserve that identity instead of
|
|
167
|
+
# registering a new shared-pool owner on every connection (and clearing its active sockets).
|
|
168
|
+
def __hash__(self) -> int:
|
|
169
|
+
return hash(self.pool)
|
|
170
|
+
|
|
171
|
+
def __eq__(self, other: object) -> bool:
|
|
172
|
+
return self.pool is (other.pool if isinstance(other, _NoReplayTransport) else other)
|
|
173
|
+
|
|
174
|
+
def request(self, method: str, url: str, **options: Any) -> Any:
|
|
175
|
+
from urllib3.exceptions import HTTPError
|
|
176
|
+
|
|
177
|
+
options["retries"] = False
|
|
178
|
+
options["redirect"] = False
|
|
179
|
+
try:
|
|
180
|
+
return self.pool.request(method, url, **options)
|
|
181
|
+
except HTTPError as exc:
|
|
182
|
+
raise EngineError(
|
|
183
|
+
"ClickHouse transport failed; the operation was not replayed and its outcome "
|
|
184
|
+
"may be unknown"
|
|
185
|
+
) from exc
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
MIN_CLICKHOUSE_CONNECT = "1.7.2"
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _require_driver_version(installed: str) -> None:
|
|
192
|
+
parts = re.fullmatch(r"(\d+)\.(\d+)\.(\d+)(?:\.post\d+)?", installed)
|
|
193
|
+
if parts is None or tuple(int(part) for part in parts.groups()) < tuple(
|
|
194
|
+
int(part) for part in MIN_CLICKHOUSE_CONNECT.split(".")
|
|
195
|
+
):
|
|
196
|
+
raise EngineError(
|
|
197
|
+
f"ClickHouse requires stable clickhouse-connect >= {MIN_CLICKHOUSE_CONNECT}; "
|
|
198
|
+
f"installed {installed}. Older drivers do not meet the timestamp and structured "
|
|
199
|
+
"error-code contract. Install the SDK clickhouse extra in this environment."
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
MAX_CA_BYTES = 1024 * 1024
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _ca_context(path: str | None) -> ssl.SSLContext | None:
|
|
207
|
+
"""Read and qualify private per-client trust before creating any driver connection."""
|
|
208
|
+
if path is None:
|
|
209
|
+
return None
|
|
210
|
+
try:
|
|
211
|
+
flags = os.O_RDONLY | getattr(os, "O_NONBLOCK", 0) | getattr(os, "O_CLOEXEC", 0)
|
|
212
|
+
fd = os.open(path, flags)
|
|
213
|
+
try:
|
|
214
|
+
info = os.fstat(fd)
|
|
215
|
+
if not stat.S_ISREG(info.st_mode) or not 0 < info.st_size <= MAX_CA_BYTES:
|
|
216
|
+
raise ValueError("CA must be a bounded regular file")
|
|
217
|
+
pieces: list[bytes] = []
|
|
218
|
+
size = 0
|
|
219
|
+
while size <= MAX_CA_BYTES:
|
|
220
|
+
piece = os.read(fd, min(65536, MAX_CA_BYTES + 1 - size))
|
|
221
|
+
if not piece:
|
|
222
|
+
break
|
|
223
|
+
pieces.append(piece)
|
|
224
|
+
size += len(piece)
|
|
225
|
+
payload = b"".join(pieces)
|
|
226
|
+
finally:
|
|
227
|
+
os.close(fd)
|
|
228
|
+
if not payload or len(payload) > MAX_CA_BYTES:
|
|
229
|
+
raise ValueError("CA must be nonempty and at most 1 MiB")
|
|
230
|
+
context = ssl.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
|
|
231
|
+
context.load_verify_locations(cadata=payload.decode("ascii"))
|
|
232
|
+
if not context.cert_store_stats()["x509"]:
|
|
233
|
+
raise ValueError("empty CA store")
|
|
234
|
+
return context
|
|
235
|
+
except (OSError, ValueError, UnicodeError):
|
|
236
|
+
raise EngineError(
|
|
237
|
+
"ClickHouse CA file must contain readable valid PEM certificates within 1 MiB"
|
|
238
|
+
) from None
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _connect_client(parameters: ConnectionParameters) -> Any:
|
|
242
|
+
from clickhouse_connect.driver.httpclient import HttpClient
|
|
243
|
+
from clickhouse_connect.driver.query import TzSource
|
|
244
|
+
from urllib3 import Timeout
|
|
245
|
+
|
|
246
|
+
context = _ca_context(parameters.ca_cert)
|
|
247
|
+
|
|
248
|
+
class GuardedHttpClient(HttpClient):
|
|
249
|
+
def _init_common_settings(self, tz_source: TzSource) -> None:
|
|
250
|
+
# HttpClient calls this after constructing its native pool/backend and before the
|
|
251
|
+
# initial version/settings exchange. Change only this client's pool reference.
|
|
252
|
+
try:
|
|
253
|
+
self.http = cast(Any, _NoReplayTransport(self.http))
|
|
254
|
+
self.timeout = Timeout(
|
|
255
|
+
connect=parameters.connect_timeout, read=parameters.send_receive_timeout
|
|
256
|
+
)
|
|
257
|
+
if context is not None:
|
|
258
|
+
if not self._owns_pool_manager:
|
|
259
|
+
raise EngineError("The driver did not create an isolated CA pool")
|
|
260
|
+
# The native factory owns this pool; pin the already validated CA bytes so
|
|
261
|
+
# later sockets do not reread a changed path. Never alter the shared pool.
|
|
262
|
+
pool_options = self.http.connection_pool_kw
|
|
263
|
+
pool_options["ssl_context"] = context
|
|
264
|
+
pool_options.pop("ca_certs", None)
|
|
265
|
+
super()._init_common_settings(tz_source)
|
|
266
|
+
except BaseException as exc:
|
|
267
|
+
try:
|
|
268
|
+
self.close()
|
|
269
|
+
except BaseException:
|
|
270
|
+
exc.add_note("ClickHouse client cleanup also failed after startup.")
|
|
271
|
+
raise
|
|
272
|
+
|
|
273
|
+
# Driver 1.7.2 coerces timeout inputs to integers before constructing its backend. Positive
|
|
274
|
+
# placeholders permit construction; the scoped hook installs the exact requested fractions
|
|
275
|
+
# before I/O. The URI itself is never reparsed or used as an arbitrary kwargs channel.
|
|
276
|
+
host = f"[{parameters.host}]" if ":" in parameters.host else parameters.host
|
|
277
|
+
return GuardedHttpClient(
|
|
278
|
+
interface=parameters.interface,
|
|
279
|
+
host=host,
|
|
280
|
+
port=parameters.port,
|
|
281
|
+
username=parameters.username,
|
|
282
|
+
password=parameters.password,
|
|
283
|
+
database=parameters.database,
|
|
284
|
+
verify=True,
|
|
285
|
+
ca_cert=parameters.ca_cert,
|
|
286
|
+
query_retries=0,
|
|
287
|
+
connect_timeout=math.ceil(parameters.connect_timeout),
|
|
288
|
+
send_receive_timeout=math.ceil(parameters.send_receive_timeout),
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
class ClickHouseEngine:
|
|
293
|
+
"""A thin adapter over clickhouse-connect. Executes decisions, makes none."""
|
|
294
|
+
|
|
295
|
+
dialect = "clickhouse"
|
|
296
|
+
|
|
297
|
+
def __init__(self, dsn: str) -> None:
|
|
298
|
+
self._connection = parse_dsn(dsn)
|
|
299
|
+
try:
|
|
300
|
+
import clickhouse_connect # noqa: F401 -- validate the optional driver extra
|
|
301
|
+
except ImportError as exc: # pragma: no cover - depends on the install extra
|
|
302
|
+
raise EngineError(
|
|
303
|
+
"the ClickHouse adapter needs the 'clickhouse' extra: "
|
|
304
|
+
"pip install 'smart-data-engine-sdk[clickhouse]'. "
|
|
305
|
+
"The core library has no dependencies, because it goes into your application and "
|
|
306
|
+
"every dependency here would be one you inherit."
|
|
307
|
+
) from exc
|
|
308
|
+
_require_driver_version(version("clickhouse-connect"))
|
|
309
|
+
self._dsn = dsn
|
|
310
|
+
self._client: Any = None
|
|
311
|
+
self._usage = UsageGate()
|
|
312
|
+
|
|
313
|
+
# --- connection ------------------------------------------------------------------------
|
|
314
|
+
|
|
315
|
+
@guarded
|
|
316
|
+
def connect(self) -> None:
|
|
317
|
+
"""Open the client, with a bound on how long that may take.
|
|
318
|
+
|
|
319
|
+
Same measurement as the PostgreSQL adapter, and the same absence: a host that accepts the
|
|
320
|
+
connection and never answers hung for as long as the test would wait. Here the reason is
|
|
321
|
+
one layer up - the TCP connect succeeds, so ``connect_timeout`` never fires, and what is
|
|
322
|
+
left is ``send_receive_timeout``, whose driver default is 300 seconds.
|
|
323
|
+
|
|
324
|
+
Both are transport timers, with distinct defaults for opening and the initial exchange.
|
|
325
|
+
Exact caller-supplied seconds survive the driver's integer constructor coercion. Socket
|
|
326
|
+
inactivity is not an absolute query deadline: progress received from a long-running query
|
|
327
|
+
can keep that exchange alive under the driver's existing transport behavior.
|
|
328
|
+
"""
|
|
329
|
+
if self._client is None:
|
|
330
|
+
try:
|
|
331
|
+
self._client = _connect_client(self._connection)
|
|
332
|
+
except EngineError as exc:
|
|
333
|
+
raise EngineError("could not connect to ClickHouse: " + str(exc)) from None
|
|
334
|
+
except Exception:
|
|
335
|
+
# Driver/parser exceptions may echo URL, credentials or server content. The
|
|
336
|
+
# connection error conveys no proof of rollback and never weakens TLS to retry.
|
|
337
|
+
raise EngineError(
|
|
338
|
+
"could not connect to ClickHouse with the configured transport"
|
|
339
|
+
) from None
|
|
340
|
+
|
|
341
|
+
@guarded
|
|
342
|
+
def close(self) -> None:
|
|
343
|
+
if self._client is not None:
|
|
344
|
+
self._client.close()
|
|
345
|
+
self._client = None
|
|
346
|
+
|
|
347
|
+
def __enter__(self) -> ClickHouseEngine:
|
|
348
|
+
self.connect()
|
|
349
|
+
return self
|
|
350
|
+
|
|
351
|
+
def __exit__(self, *_: object) -> None:
|
|
352
|
+
self.close()
|
|
353
|
+
|
|
354
|
+
@property
|
|
355
|
+
def _cx(self) -> Any:
|
|
356
|
+
if self._client is None:
|
|
357
|
+
raise EngineError("not connected; call connect() first")
|
|
358
|
+
return self._client
|
|
359
|
+
|
|
360
|
+
def write_fence(self, table: str, *, project_id: str) -> WriteFence:
|
|
361
|
+
"""DDL capability; use a dedicated connection separate from application traffic."""
|
|
362
|
+
return WriteFence(ClickHouseFences(self._cx), table, project_id=project_id)
|
|
363
|
+
|
|
364
|
+
# --- schema ----------------------------------------------------------------------------
|
|
365
|
+
|
|
366
|
+
@guarded
|
|
367
|
+
def ensure_schema(
|
|
368
|
+
self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]]
|
|
369
|
+
) -> tuple[PhysicalFinding, ...]:
|
|
370
|
+
"""Create what is missing, change nothing that exists.
|
|
371
|
+
|
|
372
|
+
`ORDER BY` is the declared key, in declared order. That order is positional and carries
|
|
373
|
+
meaning: it decides which prefixes of the key can prune granules, so sorting it would
|
|
374
|
+
change the physical performance of the table while leaving the map looking identical.
|
|
375
|
+
|
|
376
|
+
Data-skipping indexes, when a layout declares them (map contract 5), are part of the
|
|
377
|
+
`CREATE TABLE` statement: `IF NOT EXISTS` never adds one to a table that already exists.
|
|
378
|
+
|
|
379
|
+
Returns how existing tables differ from the declared physical design - sort key, partition,
|
|
380
|
+
indexes - without refusing: the caller decides whether a difference is a refusal (a person
|
|
381
|
+
provisioning a map) or a report (a running session). Columns and types still refuse here.
|
|
382
|
+
"""
|
|
383
|
+
statements = schema_statements(layout, keys=keys, dialect=self.dialect)
|
|
384
|
+
|
|
385
|
+
for statement in statements:
|
|
386
|
+
try:
|
|
387
|
+
self._cx.command(statement)
|
|
388
|
+
except Exception as exc:
|
|
389
|
+
raise EngineError(f"schema statement failed: {statement}: {exc}") from exc
|
|
390
|
+
log("sde.schema.applied", engine=self.dialect, statements=len(statements))
|
|
391
|
+
self._verify_schema(layout)
|
|
392
|
+
return self._physical_findings(layout, keys)
|
|
393
|
+
|
|
394
|
+
@guarded
|
|
395
|
+
def validate_schema(
|
|
396
|
+
self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]] | None = None
|
|
397
|
+
) -> tuple[PhysicalFinding, ...]:
|
|
398
|
+
"""Check the existing physical columns without issuing DDL; report the physical design."""
|
|
399
|
+
self._verify_schema(layout)
|
|
400
|
+
return () if keys is None else self._physical_findings(layout, keys)
|
|
401
|
+
|
|
402
|
+
def _physical_findings(
|
|
403
|
+
self, layout: PhysicalLayout, keys: Mapping[str, Sequence[str]]
|
|
404
|
+
) -> tuple[PhysicalFinding, ...]:
|
|
405
|
+
"""Sort key, partition and data-skipping indexes as the catalogue reports them.
|
|
406
|
+
|
|
407
|
+
The catalogue formats expressions, so they are parsed into names rather than compared as
|
|
408
|
+
strings we predict (:func:`sde.physical.parse_identifier_list`). A key or partition the
|
|
409
|
+
parser does not understand is reported as it stands - an expression we cannot read is not
|
|
410
|
+
evidence that the table matches.
|
|
411
|
+
"""
|
|
412
|
+
declared = declared_tables(layout, keys)
|
|
413
|
+
if not declared:
|
|
414
|
+
return ()
|
|
415
|
+
tables = [entry.table for entry in declared]
|
|
416
|
+
described = self._cx.query(
|
|
417
|
+
"SELECT name, sorting_key, partition_key FROM system.tables "
|
|
418
|
+
"WHERE database = currentDatabase() AND name IN %(tables)s",
|
|
419
|
+
parameters={"tables": tables},
|
|
420
|
+
).result_rows
|
|
421
|
+
keys_found = {str(name): (str(sort), str(part)) for name, sort, part in described}
|
|
422
|
+
present: dict[str, dict[str, tuple[str, str, int]]] = {}
|
|
423
|
+
unreadable = ""
|
|
424
|
+
indexed = sorted(entry.table for entry in declared if entry.indexes)
|
|
425
|
+
# Read only when something is declared. `system.tables` and `system.columns` are filtered
|
|
426
|
+
# by the login's own table privileges, so a restricted runtime login reads them without a
|
|
427
|
+
# grant; `system.data_skipping_indices` needs an explicit one (measured, 24.8: code 497).
|
|
428
|
+
# Reading it for every table would have made every session of an existing deployment
|
|
429
|
+
# fail to start after an upgrade, for a check about performance.
|
|
430
|
+
if indexed:
|
|
431
|
+
try:
|
|
432
|
+
index_rows = self._cx.query(
|
|
433
|
+
"SELECT table, name, type_full, expr, granularity "
|
|
434
|
+
"FROM system.data_skipping_indices "
|
|
435
|
+
"WHERE database = currentDatabase() AND table IN %(tables)s",
|
|
436
|
+
parameters={"tables": indexed},
|
|
437
|
+
).result_rows
|
|
438
|
+
except Exception as exc:
|
|
439
|
+
if getattr(exc, "code", None) != 497:
|
|
440
|
+
raise
|
|
441
|
+
# Unverified, not matching: a running session reports this and goes on; a person
|
|
442
|
+
# provisioning refuses on it, because "could not look" is not "looked and agreed".
|
|
443
|
+
unreadable = (
|
|
444
|
+
"unverified: this login cannot read system.data_skipping_indices "
|
|
445
|
+
"(GRANT SELECT ON system.data_skipping_indices to verify it)"
|
|
446
|
+
)
|
|
447
|
+
index_rows = []
|
|
448
|
+
for table_name, index_name, type_full, expr, granularity in index_rows:
|
|
449
|
+
present.setdefault(str(table_name), {})[str(index_name)] = (
|
|
450
|
+
str(type_full),
|
|
451
|
+
str(expr),
|
|
452
|
+
int(granularity),
|
|
453
|
+
)
|
|
454
|
+
findings: list[PhysicalFinding] = []
|
|
455
|
+
for entry in declared:
|
|
456
|
+
if entry.table not in keys_found:
|
|
457
|
+
continue # _verify_schema has already refused a missing table
|
|
458
|
+
sorting, partition = keys_found[entry.table]
|
|
459
|
+
try:
|
|
460
|
+
found_key: tuple[str, ...] | str = parse_identifier_list(sorting)
|
|
461
|
+
except ValueError:
|
|
462
|
+
found_key = repr(sorting)
|
|
463
|
+
if found_key != entry.key:
|
|
464
|
+
findings.append(
|
|
465
|
+
PhysicalFinding(entry.table, "sort key", repr(list(entry.key)), str(found_key))
|
|
466
|
+
)
|
|
467
|
+
try:
|
|
468
|
+
found_partition: tuple[str, str] | str | None = parse_partition_key(partition)
|
|
469
|
+
except ValueError:
|
|
470
|
+
found_partition = repr(partition)
|
|
471
|
+
if found_partition != entry.partition:
|
|
472
|
+
findings.append(
|
|
473
|
+
PhysicalFinding(
|
|
474
|
+
entry.table, "partition", str(entry.partition), str(found_partition)
|
|
475
|
+
)
|
|
476
|
+
)
|
|
477
|
+
for index in entry.indexes:
|
|
478
|
+
got = present.get(entry.table, {}).get(index.name)
|
|
479
|
+
wanted = (
|
|
480
|
+
f"{index.type_full} on {list(index.columns)} granularity {index.granularity}"
|
|
481
|
+
)
|
|
482
|
+
if got is None:
|
|
483
|
+
findings.append(
|
|
484
|
+
PhysicalFinding(
|
|
485
|
+
entry.table, f"index {index.name}", wanted, unreadable or "absent"
|
|
486
|
+
)
|
|
487
|
+
)
|
|
488
|
+
continue
|
|
489
|
+
type_full, expr, granularity = got
|
|
490
|
+
try:
|
|
491
|
+
columns: tuple[str, ...] | str = parse_identifier_list(expr)
|
|
492
|
+
except ValueError:
|
|
493
|
+
columns = repr(expr)
|
|
494
|
+
if (
|
|
495
|
+
type_full != index.type_full
|
|
496
|
+
or columns != index.columns
|
|
497
|
+
or granularity != index.granularity
|
|
498
|
+
):
|
|
499
|
+
findings.append(
|
|
500
|
+
PhysicalFinding(
|
|
501
|
+
entry.table,
|
|
502
|
+
f"index {index.name}",
|
|
503
|
+
wanted,
|
|
504
|
+
f"{type_full} on {columns} granularity {granularity}",
|
|
505
|
+
)
|
|
506
|
+
)
|
|
507
|
+
return tuple(findings)
|
|
508
|
+
|
|
509
|
+
def _verify_schema(self, layout: PhysicalLayout) -> None:
|
|
510
|
+
"""The same check as the PostgreSQL adapter, for the same reason.
|
|
511
|
+
|
|
512
|
+
`CREATE TABLE IF NOT EXISTS` keeps whatever table is already there, so a leftover from an
|
|
513
|
+
older map is accepted in silence and the first insert fails naming a column rather than
|
|
514
|
+
the cause. Missing columns are refused, extra ones are logged: a client may have added one
|
|
515
|
+
outside SDE and the map has no opinion about it.
|
|
516
|
+
|
|
517
|
+
**Types too, and here they need no normalising at all.** Measured against every type this
|
|
518
|
+
library renders: `system.columns.type` returns the exact string we wrote, down to the space
|
|
519
|
+
in `Decimal(12, 2)` and the quotes in `DateTime64(6, 'UTC')`. So the comparison is literal,
|
|
520
|
+
and it is literal on purpose - a renderer that started emitting a different spelling of the
|
|
521
|
+
same type would fail this check, which is the right way round for a document we sign.
|
|
522
|
+
"""
|
|
523
|
+
expected = {
|
|
524
|
+
table: dict(layout.columns.get(entity, {}))
|
|
525
|
+
for entity, table in sorted(layout.tables.items())
|
|
526
|
+
}
|
|
527
|
+
if not expected:
|
|
528
|
+
return
|
|
529
|
+
|
|
530
|
+
result = self._cx.query(
|
|
531
|
+
"SELECT table, name, type FROM system.columns "
|
|
532
|
+
"WHERE database = currentDatabase() AND table IN %(tables)s",
|
|
533
|
+
parameters={"tables": sorted(expected)},
|
|
534
|
+
)
|
|
535
|
+
found: dict[str, dict[str, str]] = {}
|
|
536
|
+
for table_name, column_name, column_type in result.result_rows:
|
|
537
|
+
found.setdefault(str(table_name), {})[str(column_name)] = str(column_type)
|
|
538
|
+
|
|
539
|
+
for table, columns in sorted(expected.items()):
|
|
540
|
+
actual = found.get(table)
|
|
541
|
+
if actual is None:
|
|
542
|
+
raise EngineError(
|
|
543
|
+
f"{table!r} does not exist after applying the schema. The statement reported "
|
|
544
|
+
f"success, so this is a permissions or database-selection problem rather "
|
|
545
|
+
f"than a bad map."
|
|
546
|
+
)
|
|
547
|
+
missing = sorted(set(columns) - set(actual))
|
|
548
|
+
if missing:
|
|
549
|
+
raise EngineError(
|
|
550
|
+
f"{table!r} already existed with a different shape: the map needs "
|
|
551
|
+
f"{missing} and the table has {sorted(actual)}. `CREATE TABLE IF NOT "
|
|
552
|
+
f"EXISTS` keeps whatever "
|
|
553
|
+
f"is there, so this table came from somewhere else. Refusing here rather than "
|
|
554
|
+
f"at the first insert, which would fail in your request path with an error "
|
|
555
|
+
f"naming a column and not the cause."
|
|
556
|
+
)
|
|
557
|
+
for column, declared in sorted(columns.items()):
|
|
558
|
+
reported = actual[column]
|
|
559
|
+
if reported == declared:
|
|
560
|
+
continue
|
|
561
|
+
raise EngineError(
|
|
562
|
+
f"{table}.{column} is {reported!r} and this map declares it {declared!r}. "
|
|
563
|
+
f"`CREATE TABLE IF NOT EXISTS` keeps a table of that name whatever shape it "
|
|
564
|
+
f"is in, and this library never alters a column's type - so the table came "
|
|
565
|
+
f"from somewhere else, or from a map that rendered this column differently. "
|
|
566
|
+
f"Refusing rather than writing into it: with a timestamp the difference is "
|
|
567
|
+
f"usually precision, and a write that succeeds and comes back rounded is "
|
|
568
|
+
f"worse than one that fails."
|
|
569
|
+
)
|
|
570
|
+
extra = sorted(set(actual) - set(columns))
|
|
571
|
+
if extra:
|
|
572
|
+
log("sde.schema.extra_columns", table=table, columns=extra)
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
# --- data ------------------------------------------------------------------------------
|
|
576
|
+
|
|
577
|
+
@guarded
|
|
578
|
+
def explain_plan(self, sql: str) -> QueryPlan:
|
|
579
|
+
"""Plan an analyst's query without running it, with ``readonly=1`` on every statement.
|
|
580
|
+
|
|
581
|
+
Requirement 19.4. Four questions to the server, all under ``readonly=1`` - which ClickHouse
|
|
582
|
+
refuses a mutation under (code 164) and which leaves the rest working, both halves
|
|
583
|
+
measured, because a write protection that also blocked the measurement would be one nobody
|
|
584
|
+
would keep switched on.
|
|
585
|
+
|
|
586
|
+
The plan tree, ``EXPLAIN ESTIMATE`` for the numbers, ``EXPLAIN QUERY TREE`` for the table
|
|
587
|
+
names, and ``system.tables`` for each table's engine.
|
|
588
|
+
|
|
589
|
+
**The third question exists because of a measurement that broke the obvious design.** The
|
|
590
|
+
table names came from ``EXPLAIN ESTIMATE`` first, and for ``SELECT count() FROM t`` that
|
|
591
|
+
returns **no rows at all** - ClickHouse answers a trivial count from part metadata and
|
|
592
|
+
reads nothing, so there is no table in the estimate. That is precisely the query where
|
|
593
|
+
9.7's hazard is worst: a count over a ``ReplacingMergeTree`` without ``FINAL`` returns the
|
|
594
|
+
uncollapsed number. So the one query that most needs the warning was the one the estimate
|
|
595
|
+
could not see. ``EXPLAIN QUERY TREE`` names every table in the join tree whatever the read
|
|
596
|
+
optimisation does.
|
|
597
|
+
"""
|
|
598
|
+
settings = {"readonly": 1}
|
|
599
|
+
# Read it back. Same reason as the PostgreSQL side: without this, dropping the setting
|
|
600
|
+
# changes nothing observable - nothing EXPLAIN does would write - so the guard would be
|
|
601
|
+
# unverifiable and its removal silent.
|
|
602
|
+
confirmed = self._cx.query("SELECT getSetting('readonly')", settings=settings)
|
|
603
|
+
answer = confirmed.result_rows[0][0] if confirmed.result_rows else None
|
|
604
|
+
if int(answer or 0) < 1:
|
|
605
|
+
raise EngineError(
|
|
606
|
+
f"this server would not accept readonly=1 for planning: getSetting('readonly') "
|
|
607
|
+
f"came back {answer!r}. Refusing to plan anything: requirement 19.2 keeps this "
|
|
608
|
+
f"product out of the data path, and readonly=1 is the whole of what stops a "
|
|
609
|
+
f"statement here from writing - measured, it refuses a mutation with code 164 and "
|
|
610
|
+
f"still answers EXPLAIN ESTIMATE."
|
|
611
|
+
)
|
|
612
|
+
try:
|
|
613
|
+
tree = self._cx.query(f"EXPLAIN {sql}", settings=settings)
|
|
614
|
+
plan = tuple(str(row[0]) for row in tree.result_rows)
|
|
615
|
+
estimate = self._cx.query(f"EXPLAIN ESTIMATE {sql}", settings=settings)
|
|
616
|
+
except Exception as exc:
|
|
617
|
+
raise QueryPlanRefused(
|
|
618
|
+
f"{self.dialect} would not plan this query: {exc} Nothing was executed. This is "
|
|
619
|
+
f"what validating against a live engine means: the control plane checked the query "
|
|
620
|
+
f"against the schema it authored, and this is the schema that exists."
|
|
621
|
+
) from exc
|
|
622
|
+
|
|
623
|
+
columns = list(estimate.column_names)
|
|
624
|
+
rows = [dict(zip(columns, row, strict=False)) for row in estimate.result_rows]
|
|
625
|
+
values: dict[str, str] = {}
|
|
626
|
+
for name in ("parts", "rows", "marks"):
|
|
627
|
+
if name in columns:
|
|
628
|
+
values[name] = str(sum(int(row.get(name) or 0) for row in rows))
|
|
629
|
+
values["tables"] = str(len(rows))
|
|
630
|
+
|
|
631
|
+
named = self._tables_in(sql, settings=settings) or [
|
|
632
|
+
# The estimate's own names, as a fallback for a server whose analyzer does not answer
|
|
633
|
+
# EXPLAIN QUERY TREE. Both parts come from the answer rather than from this
|
|
634
|
+
# connection's default database: a query can read a table elsewhere, and the default
|
|
635
|
+
# would name the wrong one quietly, because system.tables would simply find nothing.
|
|
636
|
+
(str(row.get("table", "")), str(row.get("database", "")))
|
|
637
|
+
for row in rows
|
|
638
|
+
if row.get("table") and row.get("database")
|
|
639
|
+
]
|
|
640
|
+
engines: list[tuple[str, str]] = []
|
|
641
|
+
for table, database in named:
|
|
642
|
+
try:
|
|
643
|
+
answer = self._cx.query(
|
|
644
|
+
"SELECT engine FROM system.tables WHERE database = {db:String} "
|
|
645
|
+
"AND name = {tb:String}",
|
|
646
|
+
parameters={"db": database, "tb": table},
|
|
647
|
+
settings=settings,
|
|
648
|
+
)
|
|
649
|
+
except Exception: # pragma: no cover - a client without system.tables access
|
|
650
|
+
# Not fatal and not silent: the plan is still worth having, and the finding this
|
|
651
|
+
# would have produced is a property of the table rather than of the query, so its
|
|
652
|
+
# absence costs a warning rather than a guarantee.
|
|
653
|
+
continue
|
|
654
|
+
for row in answer.result_rows:
|
|
655
|
+
engines.append((table, str(row[0])))
|
|
656
|
+
|
|
657
|
+
return QueryPlan(
|
|
658
|
+
engine=self.dialect,
|
|
659
|
+
dialect=self.dialect,
|
|
660
|
+
plan=plan or ("(the engine returned an empty plan)",),
|
|
661
|
+
cost=Cost(
|
|
662
|
+
units="parts, rows and marks to read",
|
|
663
|
+
basis=(
|
|
664
|
+
"real counts rather than arbitrary units, and coarse below one granule: a "
|
|
665
|
+
"mark covers 8192 rows by default, so a small table reports every row and one "
|
|
666
|
+
"mark whether or not the key filter prunes anything - measured on a "
|
|
667
|
+
"thousand-row table. The figures are what the primary key can rule out before "
|
|
668
|
+
"reading, so they move when the query filters on a key prefix and not when it "
|
|
669
|
+
"filters on anything else. Zero across the board means the engine reads "
|
|
670
|
+
"nothing at all: a trivial count() is answered from part metadata, and the "
|
|
671
|
+
"estimate correctly reports no rows read - which is not the same as a query "
|
|
672
|
+
"that returns nothing."
|
|
673
|
+
),
|
|
674
|
+
values=values,
|
|
675
|
+
),
|
|
676
|
+
findings=replacing_merge_tree_finding(engines),
|
|
677
|
+
read_only_enforced=(
|
|
678
|
+
"every statement sent with readonly=1. ClickHouse refuses a mutation under it "
|
|
679
|
+
"(code 164, READONLY) and still answers EXPLAIN ESTIMATE - both measured. There "
|
|
680
|
+
"is no transaction here to roll back, which is the same absence transaction() "
|
|
681
|
+
"refuses rather than pretends about."
|
|
682
|
+
),
|
|
683
|
+
)
|
|
684
|
+
|
|
685
|
+
def _tables_in(self, sql: str, *, settings: Mapping[str, Any]) -> list[tuple[str, str]]:
|
|
686
|
+
"""Every table the query reads, from the analyzer's own tree. ``(table, database)`` pairs.
|
|
687
|
+
|
|
688
|
+
A regex over ``EXPLAIN QUERY TREE`` output, which is the engine's own rendering and could
|
|
689
|
+
change between versions. The cost of that is a **missing** warning rather than a wrong one:
|
|
690
|
+
the finding this feeds is a property of a table, so losing it costs an analyst a sentence
|
|
691
|
+
and costs no guarantee. Worth having anyway, because the alternative loses it for the one
|
|
692
|
+
query where it matters most - see :meth:`explain_plan`.
|
|
693
|
+
"""
|
|
694
|
+
import re
|
|
695
|
+
|
|
696
|
+
try:
|
|
697
|
+
tree = self._cx.query(f"EXPLAIN QUERY TREE {sql}", settings=dict(settings))
|
|
698
|
+
except Exception as exc:
|
|
699
|
+
# No pragma here any more, and that is the point. The marker that used to say "no test
|
|
700
|
+
# comes here" sat one line above a log() call whose event name was missing from the
|
|
701
|
+
# vocabulary, and log() raised on an unknown name - so the graceful path was the one
|
|
702
|
+
# that crashed, in the client's process, on an older server.
|
|
703
|
+
log("sde.explain.no_query_tree", reason=str(exc)[:120])
|
|
704
|
+
return []
|
|
705
|
+
found: list[tuple[str, str]] = []
|
|
706
|
+
for row in tree.result_rows:
|
|
707
|
+
for match in re.finditer(r"table_name:\s*(\S+)", str(row[0])):
|
|
708
|
+
qualified = match.group(1)
|
|
709
|
+
database, _, table = qualified.rpartition(".")
|
|
710
|
+
if table and (table, database) not in found:
|
|
711
|
+
found.append((table, database))
|
|
712
|
+
return found
|
|
713
|
+
|
|
714
|
+
@guarded
|
|
715
|
+
def insert(self, table: str, values: Mapping[str, Any]) -> None:
|
|
716
|
+
if not values:
|
|
717
|
+
raise EngineError("nothing to insert")
|
|
718
|
+
cols = sorted(values)
|
|
719
|
+
row = [_as_utc(values[c]) for c in cols]
|
|
720
|
+
try:
|
|
721
|
+
self._cx.insert(_quote(table), [row], column_names=cols)
|
|
722
|
+
except Exception as exc:
|
|
723
|
+
# Surfaced, not swallowed and not rerouted, as in the PostgreSQL adapter. A write that
|
|
724
|
+
# did not happen is not our internal problem to absorb.
|
|
725
|
+
log("sde.write.failed", table=table, error=type(exc).__name__)
|
|
726
|
+
raise EngineError(f"insert into {table} failed: {exc}") from exc
|
|
727
|
+
|
|
728
|
+
@guarded
|
|
729
|
+
def insert_many(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
|
|
730
|
+
"""One native batch; no promise of cross-block transactional atomicity."""
|
|
731
|
+
cols = batch_columns(rows)
|
|
732
|
+
if not cols:
|
|
733
|
+
return
|
|
734
|
+
data = [[_as_utc(row[c]) for c in cols] for row in rows]
|
|
735
|
+
try:
|
|
736
|
+
self._cx.insert(_quote(table), data, column_names=cols)
|
|
737
|
+
except Exception as exc:
|
|
738
|
+
log("sde.write.failed", table=table, error=type(exc).__name__)
|
|
739
|
+
raise EngineError(f"batch insert into {table} failed: {exc}") from exc
|
|
740
|
+
|
|
741
|
+
@guarded
|
|
742
|
+
def storage_sizes(self, tables: Sequence[str]) -> dict[str, tuple[int, int]]:
|
|
743
|
+
"""Each table's bytes and its secondary index bytes, from active parts. Numbers only.
|
|
744
|
+
|
|
745
|
+
One statement: ``system.tables`` says which of the names exist - readable by a runtime
|
|
746
|
+
login without any grant, and only for its own tables - and ``system.parts`` what their
|
|
747
|
+
active parts occupy, which needs the column grant in :data:`STORAGE_COLUMNS`. An empty
|
|
748
|
+
table has no parts and reads 0 bytes; a missing one is absent from the answer, because a
|
|
749
|
+
missing table is not an empty one.
|
|
750
|
+
"""
|
|
751
|
+
if not tables:
|
|
752
|
+
return {}
|
|
753
|
+
sql = (
|
|
754
|
+
"SELECT t.name, sum(p.bytes_on_disk), "
|
|
755
|
+
"sum(p.secondary_indices_compressed_bytes + p.secondary_indices_marks_bytes) "
|
|
756
|
+
"FROM system.tables AS t LEFT JOIN ("
|
|
757
|
+
"SELECT table, bytes_on_disk, secondary_indices_compressed_bytes, "
|
|
758
|
+
"secondary_indices_marks_bytes FROM system.parts "
|
|
759
|
+
"WHERE active AND database = currentDatabase()"
|
|
760
|
+
") AS p ON p.table = t.name "
|
|
761
|
+
"WHERE t.database = currentDatabase() AND t.name IN {tables:Array(String)} "
|
|
762
|
+
"GROUP BY t.name"
|
|
763
|
+
)
|
|
764
|
+
try:
|
|
765
|
+
rows = self._cx.query(
|
|
766
|
+
sql, parameters={"tables": list(tables)}, settings={"join_use_nulls": 0}
|
|
767
|
+
).result_rows
|
|
768
|
+
except Exception as exc:
|
|
769
|
+
raise EngineError(f"storage sizes could not be read: {exc}") from exc
|
|
770
|
+
return {str(name): (int(total), int(secondary)) for name, total, secondary in rows}
|
|
771
|
+
|
|
772
|
+
@guarded
|
|
773
|
+
def get(self, table: str, key: Mapping[str, Any]) -> dict[str, Any] | None:
|
|
774
|
+
"""One row by key, with `FINAL` so a superseded row is never returned.
|
|
775
|
+
|
|
776
|
+
Without `FINAL` a key that has been saved twice returns whichever duplicate the scan
|
|
777
|
+
reaches first until a merge happens - which is to say, nondeterministically the old value.
|
|
778
|
+
Paying for `FINAL` on a point read is the cheaper half of that trade.
|
|
779
|
+
"""
|
|
780
|
+
parameters: dict[str, Any] = {}
|
|
781
|
+
where = " AND ".join(
|
|
782
|
+
f"{_quote(column)} = {_query_parameter(f'key_{index}', key[column], parameters)}"
|
|
783
|
+
for index, column in enumerate(sorted(key))
|
|
784
|
+
)
|
|
785
|
+
sql = f"SELECT * FROM {_quote(table)} FINAL WHERE {where} LIMIT 1"
|
|
786
|
+
try:
|
|
787
|
+
result = self._cx.query(sql, parameters=parameters)
|
|
788
|
+
except Exception as exc:
|
|
789
|
+
raise EngineError(f"select from {table} failed: {exc}") from exc
|
|
790
|
+
if not result.result_rows:
|
|
791
|
+
return None
|
|
792
|
+
return _row(result.column_names, result.column_types, result.result_rows[0])
|
|
793
|
+
|
|
794
|
+
# --- rollback protection ------------------------------------------------------------------
|
|
795
|
+
#
|
|
796
|
+
# Append-only and `max()`, which is what makes this identical in both engines: no key to
|
|
797
|
+
# enforce, no row to update, nothing for this engine's lack of a unique constraint to spoil.
|
|
798
|
+
|
|
799
|
+
@guarded
|
|
800
|
+
def map_watermark(self) -> int | None:
|
|
801
|
+
"""The highest map version applied against this engine, creating the table if missing.
|
|
802
|
+
|
|
803
|
+
A plain `MergeTree` is right here, and it is the one place in this adapter where that is
|
|
804
|
+
true without qualification: the table is append-only by design and the answer is an
|
|
805
|
+
aggregate, so there are no duplicates to collapse and no reason to pay for `FINAL`.
|
|
806
|
+
"""
|
|
807
|
+
try:
|
|
808
|
+
existing = self._cx.query(f"EXISTS TABLE {_quote(WATERMARK_TABLE)}").result_rows
|
|
809
|
+
if not existing or existing[0][0] not in (0, 1):
|
|
810
|
+
raise EngineError("watermark catalog lookup returned no presence result")
|
|
811
|
+
if existing[0][0] == 0:
|
|
812
|
+
self._cx.command(
|
|
813
|
+
f"CREATE TABLE IF NOT EXISTS {_quote(WATERMARK_TABLE)} ("
|
|
814
|
+
f"{_quote('map_version')} Int64, "
|
|
815
|
+
f"{_quote('model_version')} String, "
|
|
816
|
+
f"{_quote('seen_at')} DateTime64(3, 'UTC') DEFAULT now64(3, 'UTC')) "
|
|
817
|
+
f"ENGINE = MergeTree ORDER BY ({_quote('map_version')})"
|
|
818
|
+
)
|
|
819
|
+
result = self._cx.query(
|
|
820
|
+
f"SELECT max({_quote('map_version')}) FROM {_quote(WATERMARK_TABLE)}"
|
|
821
|
+
)
|
|
822
|
+
except Exception as exc:
|
|
823
|
+
raise EngineError(f"reading {WATERMARK_TABLE} failed: {exc}") from exc
|
|
824
|
+
if not result.result_rows or result.result_rows[0][0] is None:
|
|
825
|
+
return None
|
|
826
|
+
highest = int(result.result_rows[0][0])
|
|
827
|
+
# An empty MergeTree answers max() with 0 rather than with null, so zero here means either
|
|
828
|
+
# "no map has been applied" or "map version 0 has been". Map version 0 is what this library
|
|
829
|
+
# reads when a hand-written map omits the field, and a hand-written map is unsigned - so it
|
|
830
|
+
# never reaches this path. Reported as absent, which is the honest reading of the two.
|
|
831
|
+
return highest if highest > 0 else None
|
|
832
|
+
|
|
833
|
+
@guarded
|
|
834
|
+
def record_map_version(self, version: int, *, model_version: str) -> None:
|
|
835
|
+
try:
|
|
836
|
+
self._cx.insert(
|
|
837
|
+
WATERMARK_TABLE,
|
|
838
|
+
[[version, model_version]],
|
|
839
|
+
column_names=["map_version", "model_version"],
|
|
840
|
+
)
|
|
841
|
+
except Exception as exc:
|
|
842
|
+
raise EngineError(
|
|
843
|
+
f"recording a map version in {WATERMARK_TABLE} failed: {exc}"
|
|
844
|
+
) from exc
|
|
845
|
+
|
|
846
|
+
@guarded
|
|
847
|
+
def select_rows(self, table: str, plan: ReadPlan) -> list[dict[str, Any]]:
|
|
848
|
+
parameters: dict[str, Any] = {}
|
|
849
|
+
def parameter(value: Any) -> str:
|
|
850
|
+
return _query_parameter(f"read_{len(parameters)}", value, parameters)
|
|
851
|
+
statement = read_sql(table, plan, dialect=self.dialect, parameter=parameter)
|
|
852
|
+
try:
|
|
853
|
+
result = self._cx.query(statement, parameters=parameters)
|
|
854
|
+
rows = [_row(result.column_names, result.column_types, row)
|
|
855
|
+
for row in result.result_rows]
|
|
856
|
+
for row in rows:
|
|
857
|
+
for column in plan.columns:
|
|
858
|
+
value = row[column.name]
|
|
859
|
+
if value is not None and column.type in ("timestamp", "timestamptz"):
|
|
860
|
+
instant = (
|
|
861
|
+
value.replace(tzinfo=_dt.UTC)
|
|
862
|
+
if value.tzinfo is None else value.astimezone(_dt.UTC)
|
|
863
|
+
)
|
|
864
|
+
row[column.name] = (
|
|
865
|
+
instant.replace(tzinfo=None) if column.type == "timestamp" else instant
|
|
866
|
+
)
|
|
867
|
+
return [read_row(plan.columns, row) for row in rows]
|
|
868
|
+
except Exception as exc:
|
|
869
|
+
raise EngineError(f"logical scan of {table} failed: {exc}") from exc
|
|
870
|
+
|
|
871
|
+
@guarded
|
|
872
|
+
def count_rows(self, table: str, plan: ReadPlan) -> int:
|
|
873
|
+
parameters: dict[str, Any] = {}
|
|
874
|
+
def parameter(value: Any) -> str:
|
|
875
|
+
return _query_parameter(f"read_{len(parameters)}", value, parameters)
|
|
876
|
+
statement = read_sql(table, plan, dialect=self.dialect, parameter=parameter, count=True)
|
|
877
|
+
try:
|
|
878
|
+
result = self._cx.query(statement, parameters=parameters)
|
|
879
|
+
if not result.result_rows:
|
|
880
|
+
raise EngineError("count query returned no result")
|
|
881
|
+
return int(result.result_rows[0][0])
|
|
882
|
+
except Exception as exc:
|
|
883
|
+
raise EngineError(f"logical count of {table} failed: {exc}") from exc
|
|
884
|
+
|
|
885
|
+
@guarded
|
|
886
|
+
def summarize_rows(
|
|
887
|
+
self, table: str, plan: ReadPlan, column: ReadColumn,
|
|
888
|
+
) -> Mapping[str, Any]:
|
|
889
|
+
parameters: dict[str, Any] = {}
|
|
890
|
+
def parameter(value: Any) -> str:
|
|
891
|
+
return _query_parameter(f"read_{len(parameters)}", value, parameters)
|
|
892
|
+
statement = summary_sql(table, plan, column, dialect=self.dialect, parameter=parameter)
|
|
893
|
+
try:
|
|
894
|
+
result = self._cx.query(statement, parameters=parameters)
|
|
895
|
+
if not result.result_rows:
|
|
896
|
+
raise EngineError("summary query returned no result")
|
|
897
|
+
return dict(zip(result.column_names, result.result_rows[0], strict=True))
|
|
898
|
+
except Exception as exc:
|
|
899
|
+
raise EngineError(f"logical summary of {table} failed: {exc}") from exc
|
|
900
|
+
|
|
901
|
+
@guarded
|
|
902
|
+
def range(
|
|
903
|
+
self,
|
|
904
|
+
table: str,
|
|
905
|
+
column: str,
|
|
906
|
+
*,
|
|
907
|
+
low: Any = None,
|
|
908
|
+
high: Any = None,
|
|
909
|
+
limit: int | None = None,
|
|
910
|
+
) -> list[dict[str, Any]]:
|
|
911
|
+
clauses: list[str] = []
|
|
912
|
+
parameters: dict[str, Any] = {}
|
|
913
|
+
if low is not None:
|
|
914
|
+
bound = _query_parameter("low", low, parameters)
|
|
915
|
+
clauses.append(f"{_quote(column)} >= {bound}")
|
|
916
|
+
if high is not None:
|
|
917
|
+
bound = _query_parameter("high", high, parameters)
|
|
918
|
+
clauses.append(f"{_quote(column)} < {bound}")
|
|
919
|
+
where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
|
|
920
|
+
cap = ""
|
|
921
|
+
if limit is not None:
|
|
922
|
+
cap = " LIMIT %(limit)s"
|
|
923
|
+
parameters["limit"] = int(limit)
|
|
924
|
+
sql = f"SELECT * FROM {_quote(table)} FINAL{where} ORDER BY {_quote(column)}{cap}"
|
|
925
|
+
try:
|
|
926
|
+
result = self._cx.query(sql, parameters=parameters)
|
|
927
|
+
except Exception as exc:
|
|
928
|
+
raise EngineError(f"range select from {table} failed: {exc}") from exc
|
|
929
|
+
return [_row(result.column_names, result.column_types, row) for row in result.result_rows]
|
|
930
|
+
|
|
931
|
+
@guarded
|
|
932
|
+
def count(self, table: str) -> int:
|
|
933
|
+
"""`FINAL` here too, so this counts entities rather than stored rows.
|
|
934
|
+
|
|
935
|
+
The two differ between a save and the next merge. A count that drifts and then settles is
|
|
936
|
+
worse than a slower count, because it makes a test flaky and a dashboard untrustworthy in
|
|
937
|
+
the same way.
|
|
938
|
+
"""
|
|
939
|
+
try:
|
|
940
|
+
result = self._cx.query(f"SELECT count() FROM {_quote(table)} FINAL")
|
|
941
|
+
except Exception as exc:
|
|
942
|
+
raise EngineError(f"count on {table} failed: {exc}") from exc
|
|
943
|
+
return int(result.result_rows[0][0]) if result.result_rows else 0
|
|
944
|
+
|
|
945
|
+
# --- migration ------------------------------------------------------------------------------
|
|
946
|
+
#
|
|
947
|
+
# `sde.migration.Migratable`, the same optional protocol the PostgreSQL adapter satisfies. Two
|
|
948
|
+
# things are different here and neither is smoothed over: reads take `FINAL`, because a key
|
|
949
|
+
# saved twice is an overwrite in this engine rather than an error, and `copy_in` has no
|
|
950
|
+
# conflict clause to write - the collapse *is* the idempotence.
|
|
951
|
+
|
|
952
|
+
@guarded
|
|
953
|
+
def key_range(
|
|
954
|
+
self,
|
|
955
|
+
table: str,
|
|
956
|
+
order: Sequence[str],
|
|
957
|
+
*,
|
|
958
|
+
after: Sequence[Any] | None = None,
|
|
959
|
+
upto: Sequence[Any] | None = None,
|
|
960
|
+
limit: int | None = None,
|
|
961
|
+
) -> list[dict[str, Any]]:
|
|
962
|
+
"""Rows in key order, strictly after one key and up to another inclusive, with `FINAL`.
|
|
963
|
+
|
|
964
|
+
`FINAL` is what makes keyset pagination correct here rather than merely fast enough. Without
|
|
965
|
+
it a key saved twice returns two rows until a merge happens, and the next page starts
|
|
966
|
+
strictly after that key - so one of the duplicates is read and the other is not, which for a
|
|
967
|
+
backfill means copying a row this engine considers superseded.
|
|
968
|
+
"""
|
|
969
|
+
cols = key_columns(order, table)
|
|
970
|
+
clauses: list[str] = []
|
|
971
|
+
parameters: dict[str, Any] = {}
|
|
972
|
+
tuple_expr = f"({', '.join(_quote(c) for c in cols)})"
|
|
973
|
+
if after is not None:
|
|
974
|
+
same_width(after, cols, "after")
|
|
975
|
+
bound = [
|
|
976
|
+
_query_parameter(f"after_{index}", value, parameters)
|
|
977
|
+
for index, value in enumerate(after)
|
|
978
|
+
]
|
|
979
|
+
clauses.append(f"{tuple_expr} > ({', '.join(bound)})")
|
|
980
|
+
if upto is not None:
|
|
981
|
+
same_width(upto, cols, "upto")
|
|
982
|
+
bound = [
|
|
983
|
+
_query_parameter(f"upto_{index}", value, parameters)
|
|
984
|
+
for index, value in enumerate(upto)
|
|
985
|
+
]
|
|
986
|
+
clauses.append(f"{tuple_expr} <= ({', '.join(bound)})")
|
|
987
|
+
where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
|
|
988
|
+
cap = ""
|
|
989
|
+
if limit is not None:
|
|
990
|
+
cap = " LIMIT %(row_limit)s"
|
|
991
|
+
parameters["row_limit"] = int(limit)
|
|
992
|
+
sql = f"SELECT * FROM {_quote(table)} FINAL{where} ORDER BY {tuple_expr}{cap}"
|
|
993
|
+
try:
|
|
994
|
+
result = self._cx.query(sql, parameters=parameters)
|
|
995
|
+
except Exception as exc:
|
|
996
|
+
raise EngineError(f"key range select from {table} failed: {exc}") from exc
|
|
997
|
+
return [_row(result.column_names, result.column_types, row) for row in result.result_rows]
|
|
998
|
+
|
|
999
|
+
@guarded
|
|
1000
|
+
def nth_key(
|
|
1001
|
+
self, table: str, order: Sequence[str], *, position: int
|
|
1002
|
+
) -> tuple[Any, ...] | None:
|
|
1003
|
+
"""The key of the ``position``-th row in key order, one-based, or None if there is none."""
|
|
1004
|
+
cols = key_columns(order, table)
|
|
1005
|
+
if position < 1:
|
|
1006
|
+
raise EngineError(f"position is one-based; {position} is not a row")
|
|
1007
|
+
projection = ", ".join(_quote(c) for c in cols)
|
|
1008
|
+
sql = (
|
|
1009
|
+
f"SELECT {projection} FROM {_quote(table)} FINAL ORDER BY ({projection}) "
|
|
1010
|
+
f"LIMIT 1 OFFSET %(skip)s"
|
|
1011
|
+
)
|
|
1012
|
+
try:
|
|
1013
|
+
result = self._cx.query(sql, parameters={"skip": position - 1})
|
|
1014
|
+
except Exception as exc:
|
|
1015
|
+
raise EngineError(f"reading row {position} of {table} failed: {exc}") from exc
|
|
1016
|
+
if not result.result_rows:
|
|
1017
|
+
return None
|
|
1018
|
+
row = _row(result.column_names, result.column_types, result.result_rows[0])
|
|
1019
|
+
return tuple(row[column] for column in cols)
|
|
1020
|
+
|
|
1021
|
+
@guarded
|
|
1022
|
+
def copy_in(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
|
|
1023
|
+
"""Insert rows. Duplicates are collapsed by the table rather than rejected by it.
|
|
1024
|
+
|
|
1025
|
+
There is no `ON CONFLICT` to write and none is needed: the tables this library creates here
|
|
1026
|
+
are `ReplacingMergeTree` ordered by the key, so a row copied twice leaves two parts that
|
|
1027
|
+
collapse to the newest at merge time and read as one under `FINAL`. That is the same
|
|
1028
|
+
idempotence the PostgreSQL path gets from a conflict clause, arrived at from the opposite
|
|
1029
|
+
direction - and it is why a recopied chunk is free in both engines.
|
|
1030
|
+
"""
|
|
1031
|
+
if not rows:
|
|
1032
|
+
return
|
|
1033
|
+
cols = sorted(rows[0])
|
|
1034
|
+
for row in rows:
|
|
1035
|
+
if sorted(row) != cols:
|
|
1036
|
+
raise EngineError(
|
|
1037
|
+
f"copy_in into {table} was given rows with different columns "
|
|
1038
|
+
f"({cols} and {sorted(row)}). A chunk comes from one table, so this is a "
|
|
1039
|
+
f"caller assembling it from two."
|
|
1040
|
+
)
|
|
1041
|
+
data = [[_as_utc(row[c]) for c in cols] for row in rows]
|
|
1042
|
+
try:
|
|
1043
|
+
self._cx.insert(_quote(table), data, column_names=cols)
|
|
1044
|
+
except Exception as exc:
|
|
1045
|
+
log("sde.write.failed", table=table, error=type(exc).__name__)
|
|
1046
|
+
raise EngineError(f"copying {len(rows)} rows into {table} failed: {exc}") from exc
|
|
1047
|
+
|
|
1048
|
+
@guarded
|
|
1049
|
+
def backfill_marker(self, *, materialization: str, entity: str) -> int:
|
|
1050
|
+
"""How many rows of this entity have been copied into this engine. Zero if none.
|
|
1051
|
+
|
|
1052
|
+
A plain `MergeTree` and `max()`, as the map watermark is - append-only by design, so there
|
|
1053
|
+
are no duplicates to collapse and no reason to pay for `FINAL`. The quirk that needed a
|
|
1054
|
+
comment there is harmless here: an empty aggregate answers 0 rather than null, and 0 is
|
|
1055
|
+
exactly what "nothing has been copied" means, so the two readings coincide.
|
|
1056
|
+
"""
|
|
1057
|
+
try:
|
|
1058
|
+
self._cx.command(
|
|
1059
|
+
f"CREATE TABLE IF NOT EXISTS {_quote(BACKFILL_TABLE)} ("
|
|
1060
|
+
f"{_quote('materialization')} String, "
|
|
1061
|
+
f"{_quote('entity')} String, "
|
|
1062
|
+
f"{_quote('rows_copied')} Int64, "
|
|
1063
|
+
f"{_quote('at')} DateTime64(3, 'UTC') DEFAULT now64(3, 'UTC')) "
|
|
1064
|
+
f"ENGINE = MergeTree ORDER BY ({_quote('materialization')}, {_quote('entity')})"
|
|
1065
|
+
)
|
|
1066
|
+
result = self._cx.query(
|
|
1067
|
+
f"SELECT max({_quote('rows_copied')}) FROM {_quote(BACKFILL_TABLE)} "
|
|
1068
|
+
f"WHERE {_quote('materialization')} = %(m)s AND {_quote('entity')} = %(e)s",
|
|
1069
|
+
parameters={"m": materialization, "e": entity},
|
|
1070
|
+
)
|
|
1071
|
+
except Exception as exc:
|
|
1072
|
+
raise EngineError(f"reading {BACKFILL_TABLE} failed: {exc}") from exc
|
|
1073
|
+
if not result.result_rows or result.result_rows[0][0] is None:
|
|
1074
|
+
return 0
|
|
1075
|
+
return int(result.result_rows[0][0])
|
|
1076
|
+
|
|
1077
|
+
@guarded
|
|
1078
|
+
def record_backfill_marker(self, *, materialization: str, entity: str, rows: int) -> None:
|
|
1079
|
+
"""Append the new marker. Never update, so an interrupted run leaves a readable trail."""
|
|
1080
|
+
try:
|
|
1081
|
+
self._cx.insert(
|
|
1082
|
+
BACKFILL_TABLE,
|
|
1083
|
+
[[materialization, entity, int(rows)]],
|
|
1084
|
+
column_names=["materialization", "entity", "rows_copied"],
|
|
1085
|
+
)
|
|
1086
|
+
except Exception as exc:
|
|
1087
|
+
raise EngineError(
|
|
1088
|
+
f"recording backfill progress in {BACKFILL_TABLE} failed: {exc}"
|
|
1089
|
+
) from exc
|
|
1090
|
+
|
|
1091
|
+
# --- transactions ----------------------------------------------------------------------
|
|
1092
|
+
|
|
1093
|
+
def transaction(self) -> Iterator[ClickHouseEngine]:
|
|
1094
|
+
"""Refuses. There is no transaction here to give you.
|
|
1095
|
+
|
|
1096
|
+
A no-op context manager would be the friendlier signature and the worse library: the
|
|
1097
|
+
caller would believe a group of writes was atomic, and would find out otherwise from the
|
|
1098
|
+
state of the data rather than from an exception.
|
|
1099
|
+
|
|
1100
|
+
The way out is not a flag. Declare the atomicity - `atomic_with` on the entity - and the
|
|
1101
|
+
planner is then obliged to place those entities in one group, and one group is one engine,
|
|
1102
|
+
so it will not be this one.
|
|
1103
|
+
|
|
1104
|
+
Not decorated with `@contextmanager`, unlike the PostgreSQL one. A decorated generator
|
|
1105
|
+
would need a `yield` after the `raise` to keep the type honest, and that statement is
|
|
1106
|
+
unreachable - `mypy --strict` says so, correctly. A plain method that raises fails at the
|
|
1107
|
+
call, which is one frame earlier and reads better in a traceback.
|
|
1108
|
+
"""
|
|
1109
|
+
raise EngineError(
|
|
1110
|
+
"ClickHouse has no multi-statement transactions, so this adapter will not pretend to "
|
|
1111
|
+
"start one. If these writes have to commit together, declare it: `atomic_with` on the "
|
|
1112
|
+
"entities makes them one colocation group, one group is one engine, and the planner is "
|
|
1113
|
+
"then not permitted to put them here. A silent no-op context manager would let the "
|
|
1114
|
+
"writes proceed and let you believe they were atomic."
|
|
1115
|
+
)
|