smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +226 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/engines/__init__.py +0 -0
- sde/engines/clickhouse.py +689 -0
- sde/engines/orderbook.py +454 -0
- sde/engines/postgres.py +672 -0
- sde/entity.py +170 -0
- sde/errors.py +88 -0
- sde/explain.py +300 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/infer.py +461 -0
- sde/internal.py +90 -0
- sde/layout.py +660 -0
- sde/logging.py +132 -0
- sde/migration.py +820 -0
- sde/model.py +482 -0
- sde/placement.py +818 -0
- sde/py.typed +0 -0
- sde/routing.py +85 -0
- sde/schema.py +370 -0
- sde/session.py +507 -0
- sde/shapes.py +153 -0
- sde/telemetry.py +736 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +318 -0
- sde/types.py +228 -0
- sde/watermark.py +222 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/METADATA +152 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/RECORD +35 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/NOTICE +13 -0
|
@@ -0,0 +1,689 @@
|
|
|
1
|
+
"""ClickHouse adapter.
|
|
2
|
+
|
|
3
|
+
The second engine, and the reason there is a product at all: between PostgreSQL and ClickHouse
|
|
4
|
+
lies the decision clients get wrong once, at the start, and never revisit. With one adapter the
|
|
5
|
+
planner chooses from a set of size one.
|
|
6
|
+
|
|
7
|
+
Three things here are genuinely different from the PostgreSQL adapter, and none of them is a
|
|
8
|
+
detail. Each is a place where "the same operation" means something else because the engine is
|
|
9
|
+
different, and the product's position is that such a difference must be *stated* rather than
|
|
10
|
+
smoothed over.
|
|
11
|
+
|
|
12
|
+
**A naive datetime is UTC. Always, explicitly, here.**
|
|
13
|
+
|
|
14
|
+
A fix for a divergence that was measured rather than imagined. `datetime(2026, 8, 27, 12, 0)` with
|
|
15
|
+
no tzinfo, written through this library:
|
|
16
|
+
|
|
17
|
+
- into PostgreSQL `timestamptz`, it arrives as ``12:00:00+00:00`` - into ClickHouse
|
|
18
|
+
``DateTime64(3, 'UTC')`` with the driver left to its own devices, it arrived as
|
|
19
|
+
``10:00:00`` - because `clickhouse-connect` reads a naive datetime as *local* time and converts
|
|
20
|
+
|
|
21
|
+
Two hours apart, from one call, in one library, with no error anywhere. After a migration from one
|
|
22
|
+
engine to the other every timestamp in the client's analytics would shift by the offset of
|
|
23
|
+
whichever machine happened to write the row. So naive datetimes are given `timezone.utc` before
|
|
24
|
+
they reach the driver, which is what the PostgreSQL path already does in effect. If you mean a
|
|
25
|
+
different zone, pass an aware datetime and it is respected.
|
|
26
|
+
|
|
27
|
+
**There are no transactions, and this adapter says so instead of pretending.**
|
|
28
|
+
|
|
29
|
+
`transaction()` raises. Not a no-op context manager: a caller who believes they are in a
|
|
30
|
+
transaction and is not has been lied to at the worst possible moment. A client who needs two
|
|
31
|
+
entities to commit together declares that, and the planner then places them in the same group and
|
|
32
|
+
therefore the same engine - the requirement becomes a placement constraint. This method existing
|
|
33
|
+
and refusing is how that constraint is discovered at the point of use rather than inferred from a
|
|
34
|
+
rollback that never happened.
|
|
35
|
+
|
|
36
|
+
**Keys are not enforced, so the table is a `ReplacingMergeTree` and point reads use `FINAL`.**
|
|
37
|
+
|
|
38
|
+
A MergeTree does not enforce uniqueness on its `ORDER BY`. So a second `save()` of the same key
|
|
39
|
+
does not raise the way PostgreSQL's primary key does - and that divergence cannot be removed, only
|
|
40
|
+
chosen. Plain `MergeTree` would leave two rows and make every aggregate over that group quietly
|
|
41
|
+
wrong; `ReplacingMergeTree` keeps the newest and collapses the rest at merge time. Of the two,
|
|
42
|
+
"the newest row for a key wins" is something a client can reason about. `FINAL` on point reads and
|
|
43
|
+
counts is what makes the collapse visible immediately rather than eventually, and it is not free -
|
|
44
|
+
which is the trade, stated.
|
|
45
|
+
|
|
46
|
+
The residue is worth naming plainly, because a client will meet it: in PostgreSQL, saving the same
|
|
47
|
+
key twice is an error; here it is an overwrite. That belongs in a placement decision rather than
|
|
48
|
+
in an adapter, and there is no declaration for it yet.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
from __future__ import annotations
|
|
52
|
+
|
|
53
|
+
import datetime as _dt
|
|
54
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
55
|
+
from typing import Any
|
|
56
|
+
|
|
57
|
+
from ..errors import EngineError
|
|
58
|
+
from ..explain import (
|
|
59
|
+
Cost,
|
|
60
|
+
QueryPlan,
|
|
61
|
+
QueryPlanRefused,
|
|
62
|
+
replacing_merge_tree_finding,
|
|
63
|
+
)
|
|
64
|
+
from ..logging import log
|
|
65
|
+
from ..migration import key_columns, same_width
|
|
66
|
+
from ..placement import BACKFILL_TABLE, WATERMARK_TABLE, PhysicalLayout
|
|
67
|
+
from ..schema import QUOTE, schema_statements
|
|
68
|
+
|
|
69
|
+
__all__ = ["ClickHouseEngine"]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# Bound from the one definition in sde.schema, so that DDL and DML cannot disagree about
|
|
73
|
+
# how an identifier is escaped.
|
|
74
|
+
_quote = QUOTE["clickhouse"]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _as_utc(value: Any) -> Any:
|
|
78
|
+
"""A naive datetime is UTC. See the module docstring for the measurement behind this."""
|
|
79
|
+
if isinstance(value, _dt.datetime) and value.tzinfo is None:
|
|
80
|
+
return value.replace(tzinfo=_dt.UTC)
|
|
81
|
+
return value
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _row(names: Sequence[str], types: Sequence[Any], row: Sequence[Any]) -> dict[str, Any]:
|
|
85
|
+
"""One result row as a dict, with the timezone put back on values that had one in the schema.
|
|
86
|
+
|
|
87
|
+
The write side of this module makes a naive datetime UTC. Without this, the read side would
|
|
88
|
+
not give it back: `clickhouse-connect` returns a *naive* datetime for a `DateTime64(3, 'UTC')`
|
|
89
|
+
column, while psycopg returns an aware one for a `timestamptz`. So the same entity read from
|
|
90
|
+
the two engines produces two datetimes that Python refuses to compare - ``can't compare
|
|
91
|
+
offset-naive and offset-aware datetimes`` - which is a `TypeError` in a client's code that
|
|
92
|
+
appears on the day a group is moved and not before.
|
|
93
|
+
|
|
94
|
+
The column type carries the zone, so it is read from there rather than assumed: a column
|
|
95
|
+
declared with a timezone yields aware values, one declared without yields naive, which is
|
|
96
|
+
exactly the distinction the neutral vocabulary makes between `timestamptz` and `timestamp`.
|
|
97
|
+
"""
|
|
98
|
+
out: dict[str, Any] = {}
|
|
99
|
+
for name, column_type, value in zip(names, types, row, strict=True):
|
|
100
|
+
zone = getattr(column_type, "tzinfo", None)
|
|
101
|
+
if zone is not None and isinstance(value, _dt.datetime) and value.tzinfo is None:
|
|
102
|
+
value = value.replace(tzinfo=zone)
|
|
103
|
+
out[name] = value
|
|
104
|
+
return out
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
# Seconds, and the same argument as the PostgreSQL adapter's constant of the same name.
|
|
108
|
+
CONNECT_TIMEOUT_SECONDS = 10
|
|
109
|
+
# Seconds, for the version handshake `get_client` performs. The driver's own default is 300, which
|
|
110
|
+
# is a sensible ceiling for a query and a very long time to sit inside a caller's request path
|
|
111
|
+
# waiting for a host that has already accepted the socket and said nothing.
|
|
112
|
+
HANDSHAKE_TIMEOUT_SECONDS = 15
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
class ClickHouseEngine:
|
|
116
|
+
"""A thin adapter over clickhouse-connect. Executes decisions, makes none."""
|
|
117
|
+
|
|
118
|
+
dialect = "clickhouse"
|
|
119
|
+
|
|
120
|
+
def __init__(self, dsn: str) -> None:
|
|
121
|
+
try:
|
|
122
|
+
import clickhouse_connect
|
|
123
|
+
except ImportError as exc: # pragma: no cover - depends on the install extra
|
|
124
|
+
raise EngineError(
|
|
125
|
+
"the ClickHouse adapter needs the 'clickhouse' extra: "
|
|
126
|
+
"pip install 'smart-data-engine-sdk[clickhouse]'. "
|
|
127
|
+
"The core library has no dependencies, because it goes into your application and "
|
|
128
|
+
"every dependency here would be one you inherit."
|
|
129
|
+
) from exc
|
|
130
|
+
self._module = clickhouse_connect
|
|
131
|
+
self._dsn = dsn
|
|
132
|
+
self._client: Any = None
|
|
133
|
+
|
|
134
|
+
# --- connection ------------------------------------------------------------------------
|
|
135
|
+
|
|
136
|
+
def connect(self) -> None:
|
|
137
|
+
"""Open the client, with a bound on how long that may take.
|
|
138
|
+
|
|
139
|
+
Same measurement as the PostgreSQL adapter, and the same absence: a host that accepts the
|
|
140
|
+
connection and never answers hung for as long as the test would wait. Here the reason is
|
|
141
|
+
one layer up - the TCP connect succeeds, so ``connect_timeout`` never fires, and what is
|
|
142
|
+
left is ``send_receive_timeout``, whose driver default is 300 seconds.
|
|
143
|
+
|
|
144
|
+
So both are set, and they are set to different things on purpose. Opening is bounded
|
|
145
|
+
tightly. Reading is **not**, past this handshake: an analytical query legitimately takes
|
|
146
|
+
minutes and a library that timed it out would be deciding something about the caller's
|
|
147
|
+
workload. The handshake is the one exchange whose duration this library knows anything
|
|
148
|
+
about.
|
|
149
|
+
"""
|
|
150
|
+
if self._client is None:
|
|
151
|
+
options: dict[str, Any] = {}
|
|
152
|
+
if "connect_timeout" not in self._dsn:
|
|
153
|
+
options["connect_timeout"] = CONNECT_TIMEOUT_SECONDS
|
|
154
|
+
if "send_receive_timeout" not in self._dsn:
|
|
155
|
+
options["send_receive_timeout"] = HANDSHAKE_TIMEOUT_SECONDS
|
|
156
|
+
try:
|
|
157
|
+
self._client = self._module.get_client(dsn=self._dsn, **options)
|
|
158
|
+
except Exception as exc:
|
|
159
|
+
raise EngineError(f"could not connect to ClickHouse: {exc}") from exc
|
|
160
|
+
|
|
161
|
+
def close(self) -> None:
|
|
162
|
+
if self._client is not None:
|
|
163
|
+
self._client.close()
|
|
164
|
+
self._client = None
|
|
165
|
+
|
|
166
|
+
def __enter__(self) -> ClickHouseEngine:
|
|
167
|
+
self.connect()
|
|
168
|
+
return self
|
|
169
|
+
|
|
170
|
+
def __exit__(self, *_: object) -> None:
|
|
171
|
+
self.close()
|
|
172
|
+
|
|
173
|
+
@property
|
|
174
|
+
def _cx(self) -> Any:
|
|
175
|
+
if self._client is None:
|
|
176
|
+
raise EngineError("not connected; call connect() first")
|
|
177
|
+
return self._client
|
|
178
|
+
|
|
179
|
+
# --- schema ----------------------------------------------------------------------------
|
|
180
|
+
|
|
181
|
+
def ensure_schema(self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]]) -> None:
|
|
182
|
+
"""Create what is missing, change nothing that exists.
|
|
183
|
+
|
|
184
|
+
`ORDER BY` is the declared key, in declared order. That order is positional and carries
|
|
185
|
+
meaning: it decides which prefixes of the key can prune granules, so sorting it would
|
|
186
|
+
change the physical performance of the table while leaving the map looking identical.
|
|
187
|
+
|
|
188
|
+
No indexes are created. `layout.indexes` is empty for this dialect by construction - see
|
|
189
|
+
`sde.layout` - because ClickHouse's `CREATE INDEX` builds a data-skipping index that needs
|
|
190
|
+
a type and a granularity, and choosing those is a planner decision with a cost attached.
|
|
191
|
+
"""
|
|
192
|
+
statements = schema_statements(layout, keys=keys, dialect=self.dialect)
|
|
193
|
+
|
|
194
|
+
for statement in statements:
|
|
195
|
+
try:
|
|
196
|
+
self._cx.command(statement)
|
|
197
|
+
except Exception as exc:
|
|
198
|
+
raise EngineError(f"schema statement failed: {statement}: {exc}") from exc
|
|
199
|
+
log("sde.schema.applied", engine=self.dialect, statements=len(statements))
|
|
200
|
+
self._verify_schema(layout)
|
|
201
|
+
|
|
202
|
+
def _verify_schema(self, layout: PhysicalLayout) -> None:
|
|
203
|
+
"""The same check as the PostgreSQL adapter, for the same reason.
|
|
204
|
+
|
|
205
|
+
`CREATE TABLE IF NOT EXISTS` keeps whatever table is already there, so a leftover from an
|
|
206
|
+
older map is accepted in silence and the first insert fails naming a column rather than
|
|
207
|
+
the cause. Missing columns are refused, extra ones are logged: a client may have added one
|
|
208
|
+
outside SDE and the map has no opinion about it.
|
|
209
|
+
|
|
210
|
+
**Types too, and here they need no normalising at all.** Measured against every type this
|
|
211
|
+
library renders: `system.columns.type` returns the exact string we wrote, down to the space
|
|
212
|
+
in `Decimal(12, 2)` and the quotes in `DateTime64(6, 'UTC')`. So the comparison is literal,
|
|
213
|
+
and it is literal on purpose - a renderer that started emitting a different spelling of the
|
|
214
|
+
same type would fail this check, which is the right way round for a document we sign.
|
|
215
|
+
"""
|
|
216
|
+
expected = {
|
|
217
|
+
table: dict(layout.columns.get(entity, {}))
|
|
218
|
+
for entity, table in sorted(layout.tables.items())
|
|
219
|
+
}
|
|
220
|
+
if not expected:
|
|
221
|
+
return
|
|
222
|
+
|
|
223
|
+
result = self._cx.query(
|
|
224
|
+
"SELECT table, name, type FROM system.columns "
|
|
225
|
+
"WHERE database = currentDatabase() AND table IN %(tables)s",
|
|
226
|
+
parameters={"tables": sorted(expected)},
|
|
227
|
+
)
|
|
228
|
+
found: dict[str, dict[str, str]] = {}
|
|
229
|
+
for table_name, column_name, column_type in result.result_rows:
|
|
230
|
+
found.setdefault(str(table_name), {})[str(column_name)] = str(column_type)
|
|
231
|
+
|
|
232
|
+
for table, columns in sorted(expected.items()):
|
|
233
|
+
actual = found.get(table)
|
|
234
|
+
if actual is None:
|
|
235
|
+
raise EngineError(
|
|
236
|
+
f"{table!r} does not exist after applying the schema. The statement reported "
|
|
237
|
+
f"success, so this is a permissions or database-selection problem rather "
|
|
238
|
+
f"than a bad map."
|
|
239
|
+
)
|
|
240
|
+
missing = sorted(set(columns) - set(actual))
|
|
241
|
+
if missing:
|
|
242
|
+
raise EngineError(
|
|
243
|
+
f"{table!r} already existed with a different shape: the map needs "
|
|
244
|
+
f"{missing} and the table has {sorted(actual)}. `CREATE TABLE IF NOT "
|
|
245
|
+
f"EXISTS` keeps whatever "
|
|
246
|
+
f"is there, so this table came from somewhere else. Refusing here rather than "
|
|
247
|
+
f"at the first insert, which would fail in your request path with an error "
|
|
248
|
+
f"naming a column and not the cause."
|
|
249
|
+
)
|
|
250
|
+
for column, declared in sorted(columns.items()):
|
|
251
|
+
reported = actual[column]
|
|
252
|
+
if reported == declared:
|
|
253
|
+
continue
|
|
254
|
+
raise EngineError(
|
|
255
|
+
f"{table}.{column} is {reported!r} and this map declares it {declared!r}. "
|
|
256
|
+
f"`CREATE TABLE IF NOT EXISTS` keeps a table of that name whatever shape it "
|
|
257
|
+
f"is in, and this library never alters a column's type - so the table came "
|
|
258
|
+
f"from somewhere else, or from a map that rendered this column differently. "
|
|
259
|
+
f"Refusing rather than writing into it: with a timestamp the difference is "
|
|
260
|
+
f"usually precision, and a write that succeeds and comes back rounded is "
|
|
261
|
+
f"worse than one that fails."
|
|
262
|
+
)
|
|
263
|
+
extra = sorted(set(actual) - set(columns))
|
|
264
|
+
if extra:
|
|
265
|
+
log("sde.schema.extra_columns", table=table, columns=extra)
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
# --- data ------------------------------------------------------------------------------
|
|
269
|
+
|
|
270
|
+
def explain_plan(self, sql: str) -> QueryPlan:
|
|
271
|
+
"""Plan an analyst's query without running it, with ``readonly=1`` on every statement.
|
|
272
|
+
|
|
273
|
+
Requirement 19.4. Four questions to the server, all under ``readonly=1`` - which ClickHouse
|
|
274
|
+
refuses a mutation under (code 164) and which leaves the rest working, both halves
|
|
275
|
+
measured, because a write protection that also blocked the measurement would be one nobody
|
|
276
|
+
would keep switched on.
|
|
277
|
+
|
|
278
|
+
The plan tree, ``EXPLAIN ESTIMATE`` for the numbers, ``EXPLAIN QUERY TREE`` for the table
|
|
279
|
+
names, and ``system.tables`` for each table's engine.
|
|
280
|
+
|
|
281
|
+
**The third question exists because of a measurement that broke the obvious design.** The
|
|
282
|
+
table names came from ``EXPLAIN ESTIMATE`` first, and for ``SELECT count() FROM t`` that
|
|
283
|
+
returns **no rows at all** - ClickHouse answers a trivial count from part metadata and
|
|
284
|
+
reads nothing, so there is no table in the estimate. That is precisely the query where
|
|
285
|
+
9.7's hazard is worst: a count over a ``ReplacingMergeTree`` without ``FINAL`` returns the
|
|
286
|
+
uncollapsed number. So the one query that most needs the warning was the one the estimate
|
|
287
|
+
could not see. ``EXPLAIN QUERY TREE`` names every table in the join tree whatever the read
|
|
288
|
+
optimisation does.
|
|
289
|
+
"""
|
|
290
|
+
settings = {"readonly": 1}
|
|
291
|
+
# Read it back. Same reason as the PostgreSQL side: without this, dropping the setting
|
|
292
|
+
# changes nothing observable - nothing EXPLAIN does would write - so the guard would be
|
|
293
|
+
# unverifiable and its removal silent.
|
|
294
|
+
confirmed = self._cx.query("SELECT getSetting('readonly')", settings=settings)
|
|
295
|
+
answer = confirmed.result_rows[0][0] if confirmed.result_rows else None
|
|
296
|
+
if int(answer or 0) < 1:
|
|
297
|
+
raise EngineError(
|
|
298
|
+
f"this server would not accept readonly=1 for planning: getSetting('readonly') "
|
|
299
|
+
f"came back {answer!r}. Refusing to plan anything: requirement 19.2 keeps this "
|
|
300
|
+
f"product out of the data path, and readonly=1 is the whole of what stops a "
|
|
301
|
+
f"statement here from writing - measured, it refuses a mutation with code 164 and "
|
|
302
|
+
f"still answers EXPLAIN ESTIMATE."
|
|
303
|
+
)
|
|
304
|
+
try:
|
|
305
|
+
tree = self._cx.query(f"EXPLAIN {sql}", settings=settings)
|
|
306
|
+
plan = tuple(str(row[0]) for row in tree.result_rows)
|
|
307
|
+
estimate = self._cx.query(f"EXPLAIN ESTIMATE {sql}", settings=settings)
|
|
308
|
+
except Exception as exc:
|
|
309
|
+
raise QueryPlanRefused(
|
|
310
|
+
f"{self.dialect} would not plan this query: {exc} Nothing was executed. This is "
|
|
311
|
+
f"what validating against a live engine means: the control plane checked the query "
|
|
312
|
+
f"against the schema it authored, and this is the schema that exists."
|
|
313
|
+
) from exc
|
|
314
|
+
|
|
315
|
+
columns = list(estimate.column_names)
|
|
316
|
+
rows = [dict(zip(columns, row, strict=False)) for row in estimate.result_rows]
|
|
317
|
+
values: dict[str, str] = {}
|
|
318
|
+
for name in ("parts", "rows", "marks"):
|
|
319
|
+
if name in columns:
|
|
320
|
+
values[name] = str(sum(int(row.get(name) or 0) for row in rows))
|
|
321
|
+
values["tables"] = str(len(rows))
|
|
322
|
+
|
|
323
|
+
named = self._tables_in(sql, settings=settings) or [
|
|
324
|
+
# The estimate's own names, as a fallback for a server whose analyzer does not answer
|
|
325
|
+
# EXPLAIN QUERY TREE. Both parts come from the answer rather than from this
|
|
326
|
+
# connection's default database: a query can read a table elsewhere, and the default
|
|
327
|
+
# would name the wrong one quietly, because system.tables would simply find nothing.
|
|
328
|
+
(str(row.get("table", "")), str(row.get("database", "")))
|
|
329
|
+
for row in rows
|
|
330
|
+
if row.get("table") and row.get("database")
|
|
331
|
+
]
|
|
332
|
+
engines: list[tuple[str, str]] = []
|
|
333
|
+
for table, database in named:
|
|
334
|
+
try:
|
|
335
|
+
answer = self._cx.query(
|
|
336
|
+
"SELECT engine FROM system.tables WHERE database = {db:String} "
|
|
337
|
+
"AND name = {tb:String}",
|
|
338
|
+
parameters={"db": database, "tb": table},
|
|
339
|
+
settings=settings,
|
|
340
|
+
)
|
|
341
|
+
except Exception: # pragma: no cover - a client without system.tables access
|
|
342
|
+
# Not fatal and not silent: the plan is still worth having, and the finding this
|
|
343
|
+
# would have produced is a property of the table rather than of the query, so its
|
|
344
|
+
# absence costs a warning rather than a guarantee.
|
|
345
|
+
continue
|
|
346
|
+
for row in answer.result_rows:
|
|
347
|
+
engines.append((table, str(row[0])))
|
|
348
|
+
|
|
349
|
+
return QueryPlan(
|
|
350
|
+
engine=self.dialect,
|
|
351
|
+
dialect=self.dialect,
|
|
352
|
+
plan=plan or ("(the engine returned an empty plan)",),
|
|
353
|
+
cost=Cost(
|
|
354
|
+
units="parts, rows and marks to read",
|
|
355
|
+
basis=(
|
|
356
|
+
"real counts rather than arbitrary units, and coarse below one granule: a "
|
|
357
|
+
"mark covers 8192 rows by default, so a small table reports every row and one "
|
|
358
|
+
"mark whether or not the key filter prunes anything - measured on a "
|
|
359
|
+
"thousand-row table. The figures are what the primary key can rule out before "
|
|
360
|
+
"reading, so they move when the query filters on a key prefix and not when it "
|
|
361
|
+
"filters on anything else. Zero across the board means the engine reads "
|
|
362
|
+
"nothing at all: a trivial count() is answered from part metadata, and the "
|
|
363
|
+
"estimate correctly reports no rows read - which is not the same as a query "
|
|
364
|
+
"that returns nothing."
|
|
365
|
+
),
|
|
366
|
+
values=values,
|
|
367
|
+
),
|
|
368
|
+
findings=replacing_merge_tree_finding(engines),
|
|
369
|
+
read_only_enforced=(
|
|
370
|
+
"every statement sent with readonly=1. ClickHouse refuses a mutation under it "
|
|
371
|
+
"(code 164, READONLY) and still answers EXPLAIN ESTIMATE - both measured. There "
|
|
372
|
+
"is no transaction here to roll back, which is the same absence transaction() "
|
|
373
|
+
"refuses rather than pretends about."
|
|
374
|
+
),
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
def _tables_in(self, sql: str, *, settings: Mapping[str, Any]) -> list[tuple[str, str]]:
|
|
378
|
+
"""Every table the query reads, from the analyzer's own tree. ``(table, database)`` pairs.
|
|
379
|
+
|
|
380
|
+
A regex over ``EXPLAIN QUERY TREE`` output, which is the engine's own rendering and could
|
|
381
|
+
change between versions. The cost of that is a **missing** warning rather than a wrong one:
|
|
382
|
+
the finding this feeds is a property of a table, so losing it costs an analyst a sentence
|
|
383
|
+
and costs no guarantee. Worth having anyway, because the alternative loses it for the one
|
|
384
|
+
query where it matters most - see :meth:`explain_plan`.
|
|
385
|
+
"""
|
|
386
|
+
import re
|
|
387
|
+
|
|
388
|
+
try:
|
|
389
|
+
tree = self._cx.query(f"EXPLAIN QUERY TREE {sql}", settings=dict(settings))
|
|
390
|
+
except Exception as exc:
|
|
391
|
+
# No pragma here any more, and that is the point. The marker that used to say "no test
|
|
392
|
+
# comes here" sat one line above a log() call whose event name was missing from the
|
|
393
|
+
# vocabulary, and log() raised on an unknown name - so the graceful path was the one
|
|
394
|
+
# that crashed, in the client's process, on an older server.
|
|
395
|
+
log("sde.explain.no_query_tree", reason=str(exc)[:120])
|
|
396
|
+
return []
|
|
397
|
+
found: list[tuple[str, str]] = []
|
|
398
|
+
for row in tree.result_rows:
|
|
399
|
+
for match in re.finditer(r"table_name:\s*(\S+)", str(row[0])):
|
|
400
|
+
qualified = match.group(1)
|
|
401
|
+
database, _, table = qualified.rpartition(".")
|
|
402
|
+
if table and (table, database) not in found:
|
|
403
|
+
found.append((table, database))
|
|
404
|
+
return found
|
|
405
|
+
|
|
406
|
+
def insert(self, table: str, values: Mapping[str, Any]) -> None:
|
|
407
|
+
if not values:
|
|
408
|
+
raise EngineError("nothing to insert")
|
|
409
|
+
cols = sorted(values)
|
|
410
|
+
row = [_as_utc(values[c]) for c in cols]
|
|
411
|
+
try:
|
|
412
|
+
self._cx.insert(table, [row], column_names=cols)
|
|
413
|
+
except Exception as exc:
|
|
414
|
+
# Surfaced, not swallowed and not rerouted, as in the PostgreSQL adapter. A write that
|
|
415
|
+
# did not happen is not our internal problem to absorb.
|
|
416
|
+
log("sde.write.failed", table=table, error=type(exc).__name__)
|
|
417
|
+
raise EngineError(f"insert into {table} failed: {exc}") from exc
|
|
418
|
+
|
|
419
|
+
def get(self, table: str, key: Mapping[str, Any]) -> dict[str, Any] | None:
|
|
420
|
+
"""One row by key, with `FINAL` so a superseded row is never returned.
|
|
421
|
+
|
|
422
|
+
Without `FINAL` a key that has been saved twice returns whichever duplicate the scan
|
|
423
|
+
reaches first until a merge happens - which is to say, nondeterministically the old value.
|
|
424
|
+
Paying for `FINAL` on a point read is the cheaper half of that trade.
|
|
425
|
+
"""
|
|
426
|
+
columns = sorted(key)
|
|
427
|
+
where = " AND ".join(f"{_quote(c)} = %({c})s" for c in columns)
|
|
428
|
+
sql = f"SELECT * FROM {_quote(table)} FINAL WHERE {where} LIMIT 1"
|
|
429
|
+
try:
|
|
430
|
+
result = self._cx.query(sql, parameters={c: _as_utc(key[c]) for c in columns})
|
|
431
|
+
except Exception as exc:
|
|
432
|
+
raise EngineError(f"select from {table} failed: {exc}") from exc
|
|
433
|
+
if not result.result_rows:
|
|
434
|
+
return None
|
|
435
|
+
return _row(result.column_names, result.column_types, result.result_rows[0])
|
|
436
|
+
|
|
437
|
+
# --- rollback protection ------------------------------------------------------------------
|
|
438
|
+
#
|
|
439
|
+
# Append-only and `max()`, which is what makes this identical in both engines: no key to
|
|
440
|
+
# enforce, no row to update, nothing for this engine's lack of a unique constraint to spoil.
|
|
441
|
+
|
|
442
|
+
def map_watermark(self) -> int | None:
|
|
443
|
+
"""The highest map version applied against this engine, creating the table if missing.
|
|
444
|
+
|
|
445
|
+
A plain `MergeTree` is right here, and it is the one place in this adapter where that is
|
|
446
|
+
true without qualification: the table is append-only by design and the answer is an
|
|
447
|
+
aggregate, so there are no duplicates to collapse and no reason to pay for `FINAL`.
|
|
448
|
+
"""
|
|
449
|
+
try:
|
|
450
|
+
self._cx.command(
|
|
451
|
+
f"CREATE TABLE IF NOT EXISTS {_quote(WATERMARK_TABLE)} ("
|
|
452
|
+
f"{_quote('map_version')} Int64, "
|
|
453
|
+
f"{_quote('model_version')} String, "
|
|
454
|
+
f"{_quote('seen_at')} DateTime64(3, 'UTC') DEFAULT now64(3, 'UTC')) "
|
|
455
|
+
f"ENGINE = MergeTree ORDER BY ({_quote('map_version')})"
|
|
456
|
+
)
|
|
457
|
+
result = self._cx.query(
|
|
458
|
+
f"SELECT max({_quote('map_version')}) FROM {_quote(WATERMARK_TABLE)}"
|
|
459
|
+
)
|
|
460
|
+
except Exception as exc:
|
|
461
|
+
raise EngineError(f"reading {WATERMARK_TABLE} failed: {exc}") from exc
|
|
462
|
+
if not result.result_rows or result.result_rows[0][0] is None:
|
|
463
|
+
return None
|
|
464
|
+
highest = int(result.result_rows[0][0])
|
|
465
|
+
# An empty MergeTree answers max() with 0 rather than with null, so zero here means either
|
|
466
|
+
# "no map has been applied" or "map version 0 has been". Map version 0 is what this library
|
|
467
|
+
# reads when a hand-written map omits the field, and a hand-written map is unsigned - so it
|
|
468
|
+
# never reaches this path. Reported as absent, which is the honest reading of the two.
|
|
469
|
+
return highest if highest > 0 else None
|
|
470
|
+
|
|
471
|
+
def record_map_version(self, version: int, *, model_version: str) -> None:
|
|
472
|
+
try:
|
|
473
|
+
self._cx.insert(
|
|
474
|
+
WATERMARK_TABLE,
|
|
475
|
+
[[version, model_version]],
|
|
476
|
+
column_names=["map_version", "model_version"],
|
|
477
|
+
)
|
|
478
|
+
except Exception as exc:
|
|
479
|
+
raise EngineError(
|
|
480
|
+
f"recording a map version in {WATERMARK_TABLE} failed: {exc}"
|
|
481
|
+
) from exc
|
|
482
|
+
|
|
483
|
+
def range(
|
|
484
|
+
self,
|
|
485
|
+
table: str,
|
|
486
|
+
column: str,
|
|
487
|
+
*,
|
|
488
|
+
low: Any = None,
|
|
489
|
+
high: Any = None,
|
|
490
|
+
limit: int | None = None,
|
|
491
|
+
) -> list[dict[str, Any]]:
|
|
492
|
+
clauses: list[str] = []
|
|
493
|
+
parameters: dict[str, Any] = {}
|
|
494
|
+
if low is not None:
|
|
495
|
+
clauses.append(f"{_quote(column)} >= %(low)s")
|
|
496
|
+
parameters["low"] = _as_utc(low)
|
|
497
|
+
if high is not None:
|
|
498
|
+
clauses.append(f"{_quote(column)} < %(high)s")
|
|
499
|
+
parameters["high"] = _as_utc(high)
|
|
500
|
+
where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
|
|
501
|
+
cap = ""
|
|
502
|
+
if limit is not None:
|
|
503
|
+
cap = " LIMIT %(limit)s"
|
|
504
|
+
parameters["limit"] = int(limit)
|
|
505
|
+
sql = f"SELECT * FROM {_quote(table)} FINAL{where} ORDER BY {_quote(column)}{cap}"
|
|
506
|
+
try:
|
|
507
|
+
result = self._cx.query(sql, parameters=parameters)
|
|
508
|
+
except Exception as exc:
|
|
509
|
+
raise EngineError(f"range select from {table} failed: {exc}") from exc
|
|
510
|
+
return [
|
|
511
|
+
_row(result.column_names, result.column_types, row)
|
|
512
|
+
for row in result.result_rows
|
|
513
|
+
]
|
|
514
|
+
|
|
515
|
+
def count(self, table: str) -> int:
|
|
516
|
+
"""`FINAL` here too, so this counts entities rather than stored rows.
|
|
517
|
+
|
|
518
|
+
The two differ between a save and the next merge. A count that drifts and then settles is
|
|
519
|
+
worse than a slower count, because it makes a test flaky and a dashboard untrustworthy in
|
|
520
|
+
the same way.
|
|
521
|
+
"""
|
|
522
|
+
try:
|
|
523
|
+
result = self._cx.query(f"SELECT count() FROM {_quote(table)} FINAL")
|
|
524
|
+
except Exception as exc:
|
|
525
|
+
raise EngineError(f"count on {table} failed: {exc}") from exc
|
|
526
|
+
return int(result.result_rows[0][0]) if result.result_rows else 0
|
|
527
|
+
|
|
528
|
+
# --- migration ------------------------------------------------------------------------------
|
|
529
|
+
#
|
|
530
|
+
# `sde.migration.Migratable`, the same optional protocol the PostgreSQL adapter satisfies. Two
|
|
531
|
+
# things are different here and neither is smoothed over: reads take `FINAL`, because a key
|
|
532
|
+
# saved twice is an overwrite in this engine rather than an error, and `copy_in` has no
|
|
533
|
+
# conflict clause to write - the collapse *is* the idempotence.
|
|
534
|
+
|
|
535
|
+
def key_range(
|
|
536
|
+
self,
|
|
537
|
+
table: str,
|
|
538
|
+
order: Sequence[str],
|
|
539
|
+
*,
|
|
540
|
+
after: Sequence[Any] | None = None,
|
|
541
|
+
upto: Sequence[Any] | None = None,
|
|
542
|
+
limit: int | None = None,
|
|
543
|
+
) -> list[dict[str, Any]]:
|
|
544
|
+
"""Rows in key order, strictly after one key and up to another inclusive, with `FINAL`.
|
|
545
|
+
|
|
546
|
+
`FINAL` is what makes keyset pagination correct here rather than merely fast enough. Without
|
|
547
|
+
it a key saved twice returns two rows until a merge happens, and the next page starts
|
|
548
|
+
strictly after that key - so one of the duplicates is read and the other is not, which for a
|
|
549
|
+
backfill means copying a row this engine considers superseded.
|
|
550
|
+
"""
|
|
551
|
+
cols = key_columns(order, table)
|
|
552
|
+
clauses: list[str] = []
|
|
553
|
+
parameters: dict[str, Any] = {}
|
|
554
|
+
tuple_expr = f"({', '.join(_quote(c) for c in cols)})"
|
|
555
|
+
if after is not None:
|
|
556
|
+
same_width(after, cols, "after")
|
|
557
|
+
names = [f"after_{i}" for i in range(len(cols))]
|
|
558
|
+
clauses.append(f"{tuple_expr} > ({', '.join(f'%({n})s' for n in names)})")
|
|
559
|
+
parameters.update(zip(names, (_as_utc(v) for v in after), strict=True))
|
|
560
|
+
if upto is not None:
|
|
561
|
+
same_width(upto, cols, "upto")
|
|
562
|
+
names = [f"upto_{i}" for i in range(len(cols))]
|
|
563
|
+
clauses.append(f"{tuple_expr} <= ({', '.join(f'%({n})s' for n in names)})")
|
|
564
|
+
parameters.update(zip(names, (_as_utc(v) for v in upto), strict=True))
|
|
565
|
+
where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
|
|
566
|
+
cap = ""
|
|
567
|
+
if limit is not None:
|
|
568
|
+
cap = " LIMIT %(row_limit)s"
|
|
569
|
+
parameters["row_limit"] = int(limit)
|
|
570
|
+
sql = f"SELECT * FROM {_quote(table)} FINAL{where} ORDER BY {tuple_expr}{cap}"
|
|
571
|
+
try:
|
|
572
|
+
result = self._cx.query(sql, parameters=parameters)
|
|
573
|
+
except Exception as exc:
|
|
574
|
+
raise EngineError(f"key range select from {table} failed: {exc}") from exc
|
|
575
|
+
return [_row(result.column_names, result.column_types, row) for row in result.result_rows]
|
|
576
|
+
|
|
577
|
+
def nth_key(
|
|
578
|
+
self, table: str, order: Sequence[str], *, position: int
|
|
579
|
+
) -> tuple[Any, ...] | None:
|
|
580
|
+
"""The key of the ``position``-th row in key order, one-based, or None if there is none."""
|
|
581
|
+
cols = key_columns(order, table)
|
|
582
|
+
if position < 1:
|
|
583
|
+
raise EngineError(f"position is one-based; {position} is not a row")
|
|
584
|
+
projection = ", ".join(_quote(c) for c in cols)
|
|
585
|
+
sql = (
|
|
586
|
+
f"SELECT {projection} FROM {_quote(table)} FINAL ORDER BY ({projection}) "
|
|
587
|
+
f"LIMIT 1 OFFSET %(skip)s"
|
|
588
|
+
)
|
|
589
|
+
try:
|
|
590
|
+
result = self._cx.query(sql, parameters={"skip": position - 1})
|
|
591
|
+
except Exception as exc:
|
|
592
|
+
raise EngineError(f"reading row {position} of {table} failed: {exc}") from exc
|
|
593
|
+
if not result.result_rows:
|
|
594
|
+
return None
|
|
595
|
+
row = _row(result.column_names, result.column_types, result.result_rows[0])
|
|
596
|
+
return tuple(row[column] for column in cols)
|
|
597
|
+
|
|
598
|
+
def copy_in(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
|
|
599
|
+
"""Insert rows. Duplicates are collapsed by the table rather than rejected by it.
|
|
600
|
+
|
|
601
|
+
There is no `ON CONFLICT` to write and none is needed: the tables this library creates here
|
|
602
|
+
are `ReplacingMergeTree` ordered by the key, so a row copied twice leaves two parts that
|
|
603
|
+
collapse to the newest at merge time and read as one under `FINAL`. That is the same
|
|
604
|
+
idempotence the PostgreSQL path gets from a conflict clause, arrived at from the opposite
|
|
605
|
+
direction - and it is why a recopied chunk is free in both engines.
|
|
606
|
+
"""
|
|
607
|
+
if not rows:
|
|
608
|
+
return
|
|
609
|
+
cols = sorted(rows[0])
|
|
610
|
+
for row in rows:
|
|
611
|
+
if sorted(row) != cols:
|
|
612
|
+
raise EngineError(
|
|
613
|
+
f"copy_in into {table} was given rows with different columns "
|
|
614
|
+
f"({cols} and {sorted(row)}). A chunk comes from one table, so this is a "
|
|
615
|
+
f"caller assembling it from two."
|
|
616
|
+
)
|
|
617
|
+
data = [[_as_utc(row[c]) for c in cols] for row in rows]
|
|
618
|
+
try:
|
|
619
|
+
self._cx.insert(table, data, column_names=cols)
|
|
620
|
+
except Exception as exc:
|
|
621
|
+
log("sde.write.failed", table=table, error=type(exc).__name__)
|
|
622
|
+
raise EngineError(f"copying {len(rows)} rows into {table} failed: {exc}") from exc
|
|
623
|
+
|
|
624
|
+
def backfill_marker(self, *, materialization: str, entity: str) -> int:
|
|
625
|
+
"""How many rows of this entity have been copied into this engine. Zero if none.
|
|
626
|
+
|
|
627
|
+
A plain `MergeTree` and `max()`, as the map watermark is - append-only by design, so there
|
|
628
|
+
are no duplicates to collapse and no reason to pay for `FINAL`. The quirk that needed a
|
|
629
|
+
comment there is harmless here: an empty aggregate answers 0 rather than null, and 0 is
|
|
630
|
+
exactly what "nothing has been copied" means, so the two readings coincide.
|
|
631
|
+
"""
|
|
632
|
+
try:
|
|
633
|
+
self._cx.command(
|
|
634
|
+
f"CREATE TABLE IF NOT EXISTS {_quote(BACKFILL_TABLE)} ("
|
|
635
|
+
f"{_quote('materialization')} String, "
|
|
636
|
+
f"{_quote('entity')} String, "
|
|
637
|
+
f"{_quote('rows_copied')} Int64, "
|
|
638
|
+
f"{_quote('at')} DateTime64(3, 'UTC') DEFAULT now64(3, 'UTC')) "
|
|
639
|
+
f"ENGINE = MergeTree ORDER BY ({_quote('materialization')}, {_quote('entity')})"
|
|
640
|
+
)
|
|
641
|
+
result = self._cx.query(
|
|
642
|
+
f"SELECT max({_quote('rows_copied')}) FROM {_quote(BACKFILL_TABLE)} "
|
|
643
|
+
f"WHERE {_quote('materialization')} = %(m)s AND {_quote('entity')} = %(e)s",
|
|
644
|
+
parameters={"m": materialization, "e": entity},
|
|
645
|
+
)
|
|
646
|
+
except Exception as exc:
|
|
647
|
+
raise EngineError(f"reading {BACKFILL_TABLE} failed: {exc}") from exc
|
|
648
|
+
if not result.result_rows or result.result_rows[0][0] is None:
|
|
649
|
+
return 0
|
|
650
|
+
return int(result.result_rows[0][0])
|
|
651
|
+
|
|
652
|
+
def record_backfill_marker(self, *, materialization: str, entity: str, rows: int) -> None:
|
|
653
|
+
"""Append the new marker. Never update, so an interrupted run leaves a readable trail."""
|
|
654
|
+
try:
|
|
655
|
+
self._cx.insert(
|
|
656
|
+
BACKFILL_TABLE,
|
|
657
|
+
[[materialization, entity, int(rows)]],
|
|
658
|
+
column_names=["materialization", "entity", "rows_copied"],
|
|
659
|
+
)
|
|
660
|
+
except Exception as exc:
|
|
661
|
+
raise EngineError(
|
|
662
|
+
f"recording backfill progress in {BACKFILL_TABLE} failed: {exc}"
|
|
663
|
+
) from exc
|
|
664
|
+
|
|
665
|
+
# --- transactions ----------------------------------------------------------------------
|
|
666
|
+
|
|
667
|
+
def transaction(self) -> Iterator[ClickHouseEngine]:
|
|
668
|
+
"""Refuses. There is no transaction here to give you.
|
|
669
|
+
|
|
670
|
+
A no-op context manager would be the friendlier signature and the worse library: the
|
|
671
|
+
caller would believe a group of writes was atomic, and would find out otherwise from the
|
|
672
|
+
state of the data rather than from an exception.
|
|
673
|
+
|
|
674
|
+
The way out is not a flag. Declare the atomicity - `atomic_with` on the entity - and the
|
|
675
|
+
planner is then obliged to place those entities in one group, and one group is one engine,
|
|
676
|
+
so it will not be this one.
|
|
677
|
+
|
|
678
|
+
Not decorated with `@contextmanager`, unlike the PostgreSQL one. A decorated generator
|
|
679
|
+
would need a `yield` after the `raise` to keep the type honest, and that statement is
|
|
680
|
+
unreachable - `mypy --strict` says so, correctly. A plain method that raises fails at the
|
|
681
|
+
call, which is one frame earlier and reads better in a traceback.
|
|
682
|
+
"""
|
|
683
|
+
raise EngineError(
|
|
684
|
+
"ClickHouse has no multi-statement transactions, so this adapter will not pretend to "
|
|
685
|
+
"start one. If these writes have to commit together, declare it: `atomic_with` on the "
|
|
686
|
+
"entities makes them one colocation group, one group is one engine, and the planner is "
|
|
687
|
+
"then not permitted to put them here. A silent no-op context manager would let the "
|
|
688
|
+
"writes proceed and let you believe they were atomic."
|
|
689
|
+
)
|